{"meta":{"query_hash":"faf12f59571d","filters":{"topic":"Natural Language Processing Techniques"},"cohort_total":3084,"direct_labels_cover":10,"predictions_cover":3084,"exported":3084,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/faf12f59571d","api":"https://metacan.xera.ac/api/v1/cohort?topic=Natural+Language+Processing+Techniques"},"results":[{"id":"W10031117","doi":"10.5220/0002292201400145","title":"DNA AND NATURAL LANGUAGES - Text Mining","year":2009,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Natural language processing; Natural (archaeology); Biomedical text mining; Natural language; Artificial intelligence; Text mining; History","score_opus":0.006523990818350113,"score_gpt":0.2694687897516756,"score_spread":0.2629447989333255,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W10031117","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022257078,0.04152391,0.89736015,0.008025825,0.0012767602,0.00031767364,0.0031756142,0.003374051,0.022688977],"genre_scores_gemma":[0.23126186,0.037724454,0.6808744,0.0033971278,0.0016869141,0.0005338859,0.008366044,0.0007027546,0.03545257],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989748,0.00031398732,0.0000783554,0.0002445865,0.0003453024,0.000042918513],"domain_scores_gemma":[0.99717927,0.0018128807,0.00022246136,0.00031302954,0.0003497186,0.00012268694],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010722935,0.00047304307,0.0008803152,0.0034400912,0.0007963652,0.0023350816,0.0013318869,0.0009451133,0.0056180703],"category_scores_gemma":[0.0058875233,0.00041679118,0.00058672036,0.0049648946,0.0017659384,0.0044767708,0.0014327174,0.0015349793,0.0038873872],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017034833,0.00022827626,0.004065696,0.0016953807,0.00010380465,0.00045741152,0.00050602044,0.006650528,0.008314972,0.18926492,0.020580227,0.76796234],"study_design_scores_gemma":[0.000047368307,0.000067854686,0.001998483,0.00053041085,0.0000715266,0.00088868913,0.0006466726,0.036247067,0.015525939,0.7894392,0.15448551,0.00005134857],"about_ca_topic_score_codex":0.00095367717,"about_ca_topic_score_gemma":0.0008174154,"teacher_disagreement_score":0.0056180703,"about_ca_system_score_codex":0.0006928104,"about_ca_system_score_gemma":0.0011547287,"threshold_uncertainty_score":0.018794298},"labels":[],"label_agreement":null},{"id":"W102119915","doi":"","title":"Automatic Evaluation of Relation Extraction Systems on Large-scale","year":2012,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Relationship extraction; Relation (database); Information extraction; Scale (ratio); Field (mathematics); Natural language processing; Information retrieval; Natural language; Extraction (chemistry); Artificial intelligence; Data mining; Data science; Mathematics; Geography","score_opus":0.030946327094080153,"score_gpt":0.3357781684581911,"score_spread":0.30483184136411096,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W102119915","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6819848,0.0038018066,0.23910992,0.0012400986,0.00079469464,0.0035739897,0.012577709,0.03935935,0.017557612],"genre_scores_gemma":[0.6830407,0.00083821506,0.2607399,0.0003467813,0.00020567523,0.0020832175,0.04515084,0.0016823809,0.0059123198],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9758672,0.011126475,0.0021695439,0.0036444662,0.0064909654,0.00070132787],"domain_scores_gemma":[0.9415668,0.036586333,0.0022047367,0.008379573,0.010263323,0.000999285],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.018183164,0.002061674,0.0022739582,0.004018854,0.0021517752,0.003196294,0.002996961,0.0022743077,0.004372733],"category_scores_gemma":[0.04515624,0.00085955067,0.00096081273,0.003439736,0.0014978809,0.0070505473,0.0044221305,0.0016460551,0.0032365771],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0040282486,0.005872052,0.021210453,0.0035890313,0.001456746,0.0018307812,0.0024422905,0.0780697,0.124265395,0.007455179,0.06398802,0.68579215],"study_design_scores_gemma":[0.0009806891,0.0020790275,0.042676535,0.00019999685,0.0004975513,0.0010137034,0.0020340544,0.7086716,0.19600554,0.010142649,0.03540484,0.00029379022],"about_ca_topic_score_codex":0.004874224,"about_ca_topic_score_gemma":0.0070136185,"teacher_disagreement_score":0.018183164,"about_ca_system_score_codex":0.0017250018,"about_ca_system_score_gemma":0.0018851794,"threshold_uncertainty_score":0.096162856},"labels":[],"label_agreement":null},{"id":"W1030963941","doi":"10.1163/9789401206884_013","title":"Corpus linguistics and language documentation: challenges for collaboration","year":2011,"lang":"en","type":"book-chapter","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Documentation; Corpus linguistics; Linguistics; Computer science; Text corpus; Applied linguistics; Natural language processing; Philosophy; Programming language","score_opus":0.027290815105451654,"score_gpt":0.29271301825987117,"score_spread":0.2654222031544195,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1030963941","genre_codex":"commentary","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008768062,0.1644137,0.30975905,0.37957478,0.0042275744,0.00036977985,0.00032497148,0.001384283,0.1311778],"genre_scores_gemma":[0.18525685,0.15609491,0.57137364,0.02805096,0.00843735,0.002291377,0.0011577926,0.0029938375,0.044343233],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.8942908,0.08178619,0.0048575713,0.00459878,0.013228423,0.0012382065],"domain_scores_gemma":[0.7287289,0.2244194,0.004178811,0.025883863,0.012683332,0.00410577],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.10171087,0.00088805874,0.0023210216,0.011066794,0.009708311,0.039774403,0.0063635074,0.0081343055,0.007157788],"category_scores_gemma":[0.15129046,0.0014723712,0.0009119318,0.016643384,0.027626833,0.060643874,0.019725304,0.013412011,0.00311806],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000020007015,0.00004283614,0.0004811558,0.0006594538,0.000021414016,0.00017516383,0.019923782,0.000560133,0.00017933603,0.7188155,0.040556893,0.2185644],"study_design_scores_gemma":[0.000020201329,0.000015952122,0.00027818157,0.0027037547,0.000009047949,0.00025620434,0.011210151,0.00113705,0.00018288736,0.65161216,0.33252376,0.00005063776],"about_ca_topic_score_codex":0.0061719366,"about_ca_topic_score_gemma":0.006448732,"teacher_disagreement_score":0.10171087,"about_ca_system_score_codex":0.008505989,"about_ca_system_score_gemma":0.027173765,"threshold_uncertainty_score":0.5379049},"labels":[],"label_agreement":null},{"id":"W1042283287","doi":"10.21248/hpsg.2003.18","title":"In search of epistemic primitives in the English Resource Grammar (or Why HPSG can't live without higher-order datatypes)","year":2003,"lang":"en","type":"article","venue":"Proceedings of the International Conference on Head-Driven Phrase Structure Grammar","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Head-driven phrase structure grammar; Grammar; Computer science; Order (exchange); Resource (disambiguation); Programming language; Natural language processing; Artificial intelligence; Linguistics; Generative grammar","score_opus":0.03157530896605859,"score_gpt":0.3026997935168341,"score_spread":0.27112448455077554,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1042283287","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02029283,0.0002862562,0.9386218,0.0038698674,0.00018764369,0.00003804981,0.00014025693,0.00059250847,0.035970848],"genre_scores_gemma":[0.7161535,0.0006471406,0.26689747,0.0012859603,0.00012824814,0.00012979627,0.0002281283,0.0007612591,0.013768508],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9976332,0.0012356464,0.00016451372,0.0003869527,0.00038782682,0.00019190178],"domain_scores_gemma":[0.9965946,0.001725208,0.000248056,0.00086080795,0.00047677968,0.00009451137],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006010984,0.0004887834,0.00058693264,0.0005465637,0.0014436336,0.0033888135,0.0015996477,0.0013393533,0.0053787306],"category_scores_gemma":[0.006610093,0.0006301906,0.00073817844,0.0007201933,0.0061283377,0.012858081,0.0024471588,0.002976306,0.0014246561],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000011513021,0.0000022151744,0.00013275383,0.00003304082,0.0000028360464,0.000063098574,0.0009776866,0.0006387239,0.0004551324,0.99201584,0.00057748077,0.0050896415],"study_design_scores_gemma":[0.000009781293,0.00001717368,0.0001441404,0.00005985463,0.0000151125505,0.00018247886,0.0008087896,0.004523305,0.0025690894,0.9487271,0.042922862,0.000020296497],"about_ca_topic_score_codex":0.0028426289,"about_ca_topic_score_gemma":0.0044207033,"teacher_disagreement_score":0.006010984,"about_ca_system_score_codex":0.0013519472,"about_ca_system_score_gemma":0.0016717758,"threshold_uncertainty_score":0.03178948},"labels":[],"label_agreement":null},{"id":"W104995555","doi":"10.63317/5n34rmh7y5bw","title":"Cost-Sensitive Learning in Answer Extraction","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Thomson Reuters (Canada)","funders":"Universität des Saarlandes; Bundesministerium für Bildung und Forschung; Deutsche Forschungsgemeinschaft","keywords":"Computer science; Machine learning; Artificial intelligence; Class (philosophy); Task (project management); Set (abstract data type); Training set; Data mining","score_opus":0.027385510615035833,"score_gpt":0.3014164446641877,"score_spread":0.27403093404915185,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W104995555","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.040573247,0.0009992484,0.95456004,0.0011227782,0.000052458927,0.00017363255,0.00018199364,0.0006912488,0.0016453607],"genre_scores_gemma":[0.6378802,0.00051493366,0.35698745,0.00078281737,0.00022055955,0.0004575324,0.0007484025,0.00020821823,0.0021997704],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98804164,0.006808445,0.0006340812,0.0016705781,0.0023838838,0.0004614319],"domain_scores_gemma":[0.9551492,0.036629308,0.0017678273,0.0034775995,0.0024994782,0.00047662357],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.014246734,0.0012121241,0.00308128,0.0029754427,0.0010493957,0.0028146182,0.0032219174,0.002618737,0.0022078909],"category_scores_gemma":[0.06251541,0.0010878241,0.0011764718,0.003091133,0.002152532,0.0066511678,0.003706938,0.0034162677,0.0004983425],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00070297974,0.00052711216,0.0071980148,0.00037344143,0.00024842922,0.0002579166,0.0005831053,0.46967608,0.0023641966,0.08487715,0.005925495,0.42726612],"study_design_scores_gemma":[0.00002544435,0.000047112382,0.0005985199,0.000017746712,0.000020984895,0.00006397313,0.00003519554,0.895222,0.0013008412,0.10185752,0.00079085067,0.000019821315],"about_ca_topic_score_codex":0.0024276588,"about_ca_topic_score_gemma":0.0019178771,"teacher_disagreement_score":0.014246734,"about_ca_system_score_codex":0.0027703238,"about_ca_system_score_gemma":0.0015781687,"threshold_uncertainty_score":0.0753448},"labels":[],"label_agreement":null},{"id":"W106673770","doi":"","title":"Voicing Difference in Language.","year":2000,"lang":"en","type":"article","venue":"Essays on Canadian writing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Voice; Linguistics; Philosophy","score_opus":0.00798274605322057,"score_gpt":0.24492223124485257,"score_spread":0.236939485191632,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W106673770","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.37923497,0.02322699,0.023951577,0.019123688,0.00300667,0.000073602714,0.0008725651,0.00028783502,0.55022216],"genre_scores_gemma":[0.9669512,0.0016308649,0.0024495844,0.0004584387,0.00020432496,0.000007704528,0.00011190927,0.000037504757,0.028148403],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9991725,0.00017582405,0.00004173333,0.00017876839,0.00033401328,0.00009713853],"domain_scores_gemma":[0.9970969,0.0014617769,0.00015867989,0.00016617938,0.0009633041,0.00015316375],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008870748,0.00019105039,0.00019665423,0.0009115313,0.0014856714,0.00288154,0.00048235914,0.00069697853,0.006533617],"category_scores_gemma":[0.011243415,0.000114258386,0.0001329666,0.0010919501,0.0027544287,0.0015266844,0.0005058002,0.0012917639,0.000419868],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001293074,0.00007489968,0.013134254,0.0007588574,0.00009331031,0.0010485168,0.052177723,0.0015134871,0.058184855,0.2723814,0.046689842,0.5526498],"study_design_scores_gemma":[0.00008011193,0.00023521877,0.28890872,0.00078976527,0.00015193471,0.0018210401,0.06574697,0.006912106,0.027892293,0.13822949,0.46902248,0.00020986474],"about_ca_topic_score_codex":0.17340048,"about_ca_topic_score_gemma":0.22149716,"teacher_disagreement_score":0.17340048,"about_ca_system_score_codex":0.003911545,"about_ca_system_score_gemma":0.0031644881,"threshold_uncertainty_score":0.3447823},"labels":[],"label_agreement":null},{"id":"W107142879","doi":"10.1007/978-3-540-68825-9_12","title":"Recognizing Biomedical Named Entities in Chinese Research Abstracts","year":2008,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Conditional random field; Computer science; Natural language processing; CRFS; Artificial intelligence; Named-entity recognition; Task (project management); F1 score; Information retrieval","score_opus":0.03653476046084891,"score_gpt":0.32768426447626575,"score_spread":0.29114950401541684,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W107142879","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4409599,0.050547883,0.25750902,0.0047618374,0.0033243916,0.001443316,0.17160402,0.009163323,0.060686376],"genre_scores_gemma":[0.5543133,0.016947156,0.22797458,0.0006512743,0.0013092582,0.00059354474,0.17109393,0.0005575741,0.026559474],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9992305,0.00008176312,0.00018351476,0.0002513114,0.000179702,0.000073098556],"domain_scores_gemma":[0.9983644,0.00067572994,0.00019837203,0.00012705171,0.0005125945,0.000121871046],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011831699,0.0010789668,0.0008155012,0.008063174,0.0012044711,0.0016128286,0.0009959167,0.00073853415,0.0067750276],"category_scores_gemma":[0.0032947154,0.0003460744,0.0013174139,0.008962601,0.00047472172,0.0025263978,0.001179683,0.0005396383,0.0037435277],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010779997,0.00016484887,0.045867812,0.0061220173,0.00043602876,0.0054849153,0.0025792727,0.005561247,0.11078072,0.022172991,0.116632424,0.6831198],"study_design_scores_gemma":[0.00024342841,0.0005673274,0.20606336,0.0011399135,0.003140982,0.007097121,0.00337919,0.09322657,0.10968446,0.033342626,0.5417663,0.0003488273],"about_ca_topic_score_codex":0.011607293,"about_ca_topic_score_gemma":0.013019456,"teacher_disagreement_score":0.011607293,"about_ca_system_score_codex":0.0012622187,"about_ca_system_score_gemma":0.0027827413,"threshold_uncertainty_score":0.023079455},"labels":[],"label_agreement":null},{"id":"W108071511","doi":"","title":"A Novel Discriminative Framework for Sentence-Level Discourse Analysis","year":2012,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":76,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Discriminative model; Conditional random field; Computer science; Parsing; Sentence; Artificial intelligence; Natural language processing; Probabilistic logic; Classifier (UML); Margin (machine learning); Binary number; CRFS; Speech recognition; Machine learning; Mathematics","score_opus":0.06085311550217667,"score_gpt":0.36461547151025225,"score_spread":0.3037623560080756,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W108071511","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0026982066,0.00019352061,0.9931765,0.00015176488,0.000037911377,0.000039429004,0.00044374808,0.0019971798,0.0012617371],"genre_scores_gemma":[0.14497781,0.0003198753,0.84373,0.00023797006,0.00025943885,0.00026048408,0.0029549864,0.0007089496,0.0065505495],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9982712,0.00051339244,0.000079856116,0.0006121313,0.0003968885,0.00012651163],"domain_scores_gemma":[0.998216,0.00068752945,0.00021689925,0.00037378582,0.00038936417,0.000116433104],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00181255,0.0011055624,0.0009857992,0.0026939078,0.0010014414,0.0017874696,0.0020015836,0.0010678199,0.0056915306],"category_scores_gemma":[0.0043109185,0.000737927,0.00082682667,0.002003462,0.001273315,0.0031798799,0.0020308401,0.002146936,0.0034107848],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024458187,0.00025589633,0.002949347,0.0005329649,0.000115626535,0.00036343362,0.00094186905,0.034305643,0.06903259,0.13904914,0.022491686,0.7297172],"study_design_scores_gemma":[0.000030050676,0.00011848527,0.0025205894,0.000063047846,0.00007078918,0.0005093998,0.0001528112,0.8190605,0.022273105,0.11401822,0.041100025,0.00008298508],"about_ca_topic_score_codex":0.003324478,"about_ca_topic_score_gemma":0.005308742,"teacher_disagreement_score":0.0056915306,"about_ca_system_score_codex":0.00080803194,"about_ca_system_score_gemma":0.0019444682,"threshold_uncertainty_score":0.019040048},"labels":[],"label_agreement":null},{"id":"W108353034","doi":"10.1007/978-3-642-21916-0_48","title":"Towards Automatic Acquisition of a Fully Sense Tagged Corpus for Persian","year":2011,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Natural language processing; Word-sense disambiguation; Persian; Artificial intelligence; Economic shortage; SemEval; Word (group theory); Natural language understanding; Natural language; Linguistics; WordNet","score_opus":0.016255368524319837,"score_gpt":0.25616247304639156,"score_spread":0.23990710452207173,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W108353034","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15816101,0.0028487202,0.6571883,0.002047391,0.002673045,0.0010871371,0.09421061,0.047659986,0.034123857],"genre_scores_gemma":[0.24365497,0.0010703274,0.54884267,0.00064925355,0.0006210921,0.0010836156,0.18644895,0.00737848,0.010250639],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99778205,0.0005336964,0.00032271515,0.0007918509,0.00039313236,0.00017659266],"domain_scores_gemma":[0.9913051,0.0030349728,0.00052840135,0.0017000635,0.0030434928,0.0003880035],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020708996,0.0020115739,0.001450436,0.0047463123,0.0018786445,0.0032213249,0.0018583216,0.0014820548,0.014149533],"category_scores_gemma":[0.0071944245,0.0014063905,0.00081304694,0.0031812857,0.0009765136,0.005175275,0.005017799,0.0025764632,0.016934002],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011044304,0.00047713393,0.0050664837,0.0030165429,0.0001652241,0.0036606758,0.0047292677,0.003972398,0.27496332,0.018444408,0.14554343,0.5388566],"study_design_scores_gemma":[0.00044604435,0.00095952343,0.024202935,0.0009101777,0.0004995062,0.007873267,0.008093698,0.09623854,0.24019186,0.04063042,0.57949626,0.00045780733],"about_ca_topic_score_codex":0.0026608824,"about_ca_topic_score_gemma":0.005322967,"teacher_disagreement_score":0.014149533,"about_ca_system_score_codex":0.0007568186,"about_ca_system_score_gemma":0.0032006816,"threshold_uncertainty_score":0.04733497},"labels":[],"label_agreement":null},{"id":"W110373586","doi":"","title":"Voting between Dictionary-Based and Subword Tagging Models for Chinese Word Segmentation","year":2006,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Conditional random field; Computer science; Text segmentation; Word (group theory); Artificial intelligence; Voting; Natural language processing; Segmentation; Speech recognition; Matching (statistics); Pattern recognition (psychology); Mathematics; Statistics","score_opus":0.012373086473969837,"score_gpt":0.2713870272343266,"score_spread":0.2590139407603568,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W110373586","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06600429,0.00037488953,0.9238212,0.0002144037,0.0001818408,0.00012743151,0.00030878079,0.00532576,0.003641342],"genre_scores_gemma":[0.644163,0.00024984137,0.3467211,0.00023423223,0.00007466151,0.00015688091,0.0016521116,0.0007200882,0.0060280086],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99767894,0.00084141665,0.00017749454,0.0007395564,0.0003686481,0.00019403394],"domain_scores_gemma":[0.9966281,0.0016113432,0.00017604997,0.00077225437,0.0006933609,0.0001188887],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003633315,0.0007612994,0.0012579897,0.0010748329,0.00087955897,0.0012291471,0.0021068407,0.000821398,0.003357579],"category_scores_gemma":[0.0064184098,0.00057585083,0.0009346646,0.0015647708,0.00055874564,0.0038321905,0.001257328,0.0010615127,0.0020026916],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014286161,0.00020629172,0.0067576584,0.0002934678,0.00029244594,0.00021623909,0.00060184253,0.080071375,0.023744985,0.016422952,0.00899134,0.8609729],"study_design_scores_gemma":[0.000050272978,0.000103829334,0.000888018,0.000016986696,0.00008178925,0.000093178496,0.00009103486,0.9647642,0.01742418,0.012108667,0.004331763,0.000046058518],"about_ca_topic_score_codex":0.005568124,"about_ca_topic_score_gemma":0.010832578,"teacher_disagreement_score":0.005568124,"about_ca_system_score_codex":0.0009336887,"about_ca_system_score_gemma":0.0012408793,"threshold_uncertainty_score":0.019215047},"labels":[],"label_agreement":null},{"id":"W1138336476","doi":"10.71781/13708","title":"Le repérage automatique des entités nommées dans la langue arabe : vers la création d'un système à base de règles","year":2009,"lang":"fr","type":"dissertation","venue":"Open MIND","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Humanities; Art; Philosophy; Political science","score_opus":0.016920996348984273,"score_gpt":0.3020869959005775,"score_spread":0.2851659995515933,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1138336476","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.23680845,0.0022733693,0.7030636,0.0029535885,0.00034109823,0.00032829575,0.0026279895,0.033381138,0.01822244],"genre_scores_gemma":[0.5003503,0.00093460834,0.46094579,0.00034855018,0.00008641172,0.00018124258,0.0027358932,0.001562329,0.032854818],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9984938,0.00040841213,0.00009509973,0.0005051308,0.00037456566,0.00012294114],"domain_scores_gemma":[0.99507385,0.0027418314,0.00026385984,0.00068281614,0.0010552893,0.00018245497],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0045013,0.0006540016,0.0007891445,0.0015412061,0.0016091173,0.005722539,0.0012669162,0.0011243765,0.005706535],"category_scores_gemma":[0.012258526,0.0007352189,0.0008264139,0.0011675796,0.0014104156,0.0054522036,0.00172001,0.0013846963,0.0031151164],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015126091,0.00023528704,0.016965544,0.0008187661,0.00023106254,0.0006982876,0.01358392,0.01393477,0.06953841,0.05070412,0.020799592,0.8109776],"study_design_scores_gemma":[0.00025046224,0.00044635078,0.025282517,0.0004715141,0.00058155024,0.0013262044,0.0066260663,0.36599493,0.21574745,0.0315788,0.35131413,0.000380123],"about_ca_topic_score_codex":0.11999837,"about_ca_topic_score_gemma":0.10240512,"teacher_disagreement_score":0.11999837,"about_ca_system_score_codex":0.0022922952,"about_ca_system_score_gemma":0.0038310417,"threshold_uncertainty_score":0.23859978},"labels":[],"label_agreement":null},{"id":"W115395619","doi":"10.63317/3q59sesfixax","title":"Rapid Deployment of a New METIS Language Pair: Catalan-English","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Catalan; Metis; Computer science; Software deployment; Architecture; Linguistics; Natural language processing; Artificial intelligence; History; Software engineering; World Wide Web","score_opus":0.017913791831646944,"score_gpt":0.25119890155129665,"score_spread":0.23328510971964972,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W115395619","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.71902287,0.0005629966,0.14874546,0.0015273544,0.001177917,0.0022507992,0.0051387735,0.055266425,0.06630748],"genre_scores_gemma":[0.7406327,0.00020546127,0.21800579,0.00070524274,0.00010250277,0.0013549353,0.009673867,0.0045954743,0.02472399],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9978891,0.00057755667,0.00019739423,0.0006435866,0.0004995834,0.00019274585],"domain_scores_gemma":[0.9969501,0.0005197806,0.00010175092,0.0008269708,0.0012723773,0.00032905326],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021851573,0.0009847076,0.0006858387,0.000555472,0.00066500634,0.0015927835,0.0013667301,0.0005835982,0.0088612735],"category_scores_gemma":[0.0048005427,0.0006274392,0.0004166911,0.00046758942,0.000512894,0.0027606308,0.002661953,0.0014148639,0.0056307083],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00718273,0.0010765482,0.02393119,0.0013507453,0.00032169704,0.005968482,0.008950044,0.008218867,0.4129444,0.012901948,0.07277985,0.44437343],"study_design_scores_gemma":[0.0016073307,0.0034703582,0.030003116,0.00017851277,0.0003312852,0.0032905482,0.0046058986,0.1045394,0.42470482,0.0034102628,0.42344734,0.00041129635],"about_ca_topic_score_codex":0.0065923445,"about_ca_topic_score_gemma":0.0075023994,"teacher_disagreement_score":0.0088612735,"about_ca_system_score_codex":0.0009776615,"about_ca_system_score_gemma":0.0009597245,"threshold_uncertainty_score":0.029643893},"labels":[],"label_agreement":null},{"id":"W117861810","doi":"","title":"Using monolingual source-language data to improve MT performance.","year":2006,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":37,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"Advanced Research Projects Agency; Defense Advanced Research Projects Agency","keywords":"Computer science; Natural language processing; Machine translation; Artificial intelligence; Phrase; Translation (biology); Focus (optics); Task (project management); Domain (mathematical analysis); Example-based machine translation; Training set; Source text","score_opus":0.031119709411602255,"score_gpt":0.3021307769701553,"score_spread":0.27101106755855303,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W117861810","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20714559,0.008247134,0.71765465,0.00198727,0.00086991495,0.0009247744,0.008965819,0.025785001,0.028419744],"genre_scores_gemma":[0.41345146,0.002354586,0.53417236,0.0006752025,0.00038653446,0.0006114234,0.031662334,0.0017533927,0.014932709],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9958846,0.002661508,0.00020915813,0.00054730277,0.0006071465,0.0000903178],"domain_scores_gemma":[0.9862096,0.0064907237,0.00062064675,0.0032387376,0.0032400447,0.0002002751],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004847022,0.0015965749,0.001181282,0.002817627,0.00084378483,0.0014512851,0.0013037837,0.0011551803,0.0090044625],"category_scores_gemma":[0.02106567,0.00050013774,0.00051771937,0.003096931,0.0003963843,0.0032689916,0.0020087797,0.001294346,0.014226932],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011308229,0.00083238055,0.009930779,0.0021991753,0.00057175296,0.0006097491,0.0005993291,0.017710343,0.1397744,0.0024623817,0.031714186,0.79246473],"study_design_scores_gemma":[0.00049754453,0.0021240949,0.041882113,0.00080117834,0.0008917675,0.003436518,0.00087937526,0.3861772,0.36289513,0.016839478,0.1831832,0.00039233506],"about_ca_topic_score_codex":0.0018304503,"about_ca_topic_score_gemma":0.0057830806,"teacher_disagreement_score":0.0090044625,"about_ca_system_score_codex":0.0004958053,"about_ca_system_score_gemma":0.00069376174,"threshold_uncertainty_score":0.030122936},"labels":[],"label_agreement":null},{"id":"W117940898","doi":"10.7202/1029339ar","title":"La recherche d’information multilingue","year":2015,"lang":"fr","type":"article","venue":"Documentation et bibliothèques","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Humanities; Political science; Art","score_opus":0.19111254966309607,"score_gpt":0.4330038621560142,"score_spread":0.24189131249291815,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W117940898","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021544347,0.07487455,0.724736,0.020257777,0.0032695,0.00016273995,0.0013469958,0.0026532232,0.15115492],"genre_scores_gemma":[0.22832456,0.0794161,0.5947076,0.0036124487,0.003179957,0.0004109971,0.0034183112,0.0025496574,0.084380336],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.98736924,0.0038417163,0.0010101029,0.0022012072,0.0051891278,0.00038866067],"domain_scores_gemma":[0.97614664,0.014819635,0.0009081672,0.0037444849,0.0041266,0.00025449894],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0071776295,0.0009792467,0.00131933,0.0071939547,0.0024329373,0.012557526,0.0017137611,0.0026830584,0.012900675],"category_scores_gemma":[0.030778948,0.00097542483,0.001491116,0.007829727,0.0061811823,0.014858721,0.0037782767,0.005068621,0.0051844893],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000120572244,0.000045637706,0.000995222,0.0027927244,0.00012419833,0.0006246115,0.0054132016,0.0023231753,0.010803384,0.6236491,0.016902832,0.33620533],"study_design_scores_gemma":[0.000029298413,0.000059822305,0.0013441067,0.0013726902,0.00009661522,0.0017723886,0.0014941571,0.0064460966,0.015965328,0.15214318,0.8191735,0.000102785576],"about_ca_topic_score_codex":0.009811984,"about_ca_topic_score_gemma":0.0047357646,"teacher_disagreement_score":0.012900675,"about_ca_system_score_codex":0.005194888,"about_ca_system_score_gemma":0.004727741,"threshold_uncertainty_score":0.0431571},"labels":[],"label_agreement":null},{"id":"W119124879","doi":"","title":"Implantation de grammaires de propriétés en CHR.","year":2004,"lang":"fr","type":"article","venue":"JFPLC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Humanities; Philosophy","score_opus":0.016199934260877766,"score_gpt":0.3100763791400885,"score_spread":0.2938764448792107,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W119124879","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01742074,0.0003105793,0.95941967,0.0005110935,0.00017212183,0.0002041204,0.00078782276,0.008807505,0.012366304],"genre_scores_gemma":[0.1567447,0.000571839,0.8136938,0.0004969486,0.00014133843,0.00027115492,0.0019461225,0.0030646673,0.0230695],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9958106,0.001353288,0.0005203541,0.00079290895,0.0012967328,0.00022611613],"domain_scores_gemma":[0.99223983,0.0037292882,0.0005429778,0.0018818332,0.0014604067,0.00014567586],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004571531,0.0009313107,0.00072198757,0.0016315044,0.0010707695,0.0031726977,0.001499757,0.0017681102,0.011549311],"category_scores_gemma":[0.009735617,0.0010991944,0.0014257568,0.0014018227,0.0023012112,0.006357546,0.0029310933,0.0025976347,0.0070394496],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004903915,0.00015900903,0.0046703583,0.0013066677,0.00011588801,0.0011705732,0.0045720655,0.010953113,0.044194102,0.6616989,0.008310654,0.26235837],"study_design_scores_gemma":[0.000102673526,0.00048362205,0.0026610352,0.0005775265,0.00027649902,0.0021782022,0.0018265826,0.094002046,0.09754065,0.33430827,0.4657853,0.00025755833],"about_ca_topic_score_codex":0.0025707604,"about_ca_topic_score_gemma":0.0033126385,"teacher_disagreement_score":0.011549311,"about_ca_system_score_codex":0.0011356004,"about_ca_system_score_gemma":0.0018969589,"threshold_uncertainty_score":0.038636327},"labels":[],"label_agreement":null},{"id":"W1196909593","doi":"10.20381/ruor-12790","title":"Semantic relations across syntactic levels","year":2004,"lang":"en","type":"dissertation","venue":"uO Research (University of Ottawa)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Utterance; Verb; Relation (database); Linguistics; Point (geometry); Computer science; Order (exchange); Natural language processing; Artificial intelligence; Mathematics; Philosophy","score_opus":0.04780527354515794,"score_gpt":0.3627621691719937,"score_spread":0.31495689562683576,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1196909593","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.095227495,0.0080316,0.5098074,0.0131183015,0.0010533414,0.0005700591,0.0033071719,0.0019204482,0.3669641],"genre_scores_gemma":[0.85355854,0.002374512,0.12384316,0.001914765,0.00068552484,0.0006944992,0.0035348777,0.0009146487,0.012479425],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99105644,0.003969951,0.0008303743,0.0016367554,0.0018351345,0.00067124737],"domain_scores_gemma":[0.9915564,0.0047734566,0.0007199708,0.00122369,0.0015203282,0.00020616828],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004412442,0.0013124183,0.00092330505,0.0050935657,0.0034960767,0.01032348,0.0018592637,0.003204292,0.014275357],"category_scores_gemma":[0.015070052,0.0012640685,0.0018036205,0.0048612542,0.008776523,0.028781522,0.0055698566,0.005975502,0.003784508],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007845696,0.000027121161,0.0014351844,0.00036387518,0.00006559914,0.0003576866,0.016150475,0.00039787096,0.003967933,0.945439,0.0030380448,0.02867868],"study_design_scores_gemma":[0.000021262062,0.000038846567,0.0033937332,0.0002595407,0.00013999276,0.0003880319,0.00662879,0.0022409982,0.0020127175,0.9184297,0.066379294,0.00006712969],"about_ca_topic_score_codex":0.003222204,"about_ca_topic_score_gemma":0.0017099255,"teacher_disagreement_score":0.014275357,"about_ca_system_score_codex":0.0038430167,"about_ca_system_score_gemma":0.0020590934,"threshold_uncertainty_score":0.047755778},"labels":[],"label_agreement":null},{"id":"W123161345","doi":"10.1007/978-3-540-85760-0_37","title":"Bilingual Question Answering Using CINDI_QA at QA@CLEF 2007","year":2008,"lang":"en","type":"article","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Clef; Question answering; Computer science; Information retrieval; Natural language processing; Artificial intelligence; Engineering","score_opus":0.018965334212091404,"score_gpt":0.29230686669175526,"score_spread":0.27334153247966386,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W123161345","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06617764,0.002073642,0.42306623,0.005614067,0.0028551752,0.0021666544,0.09385871,0.23129384,0.17289402],"genre_scores_gemma":[0.24236093,0.00048614616,0.40291083,0.00141454,0.00066142995,0.0012980957,0.2799314,0.019553147,0.05138349],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9949302,0.002200206,0.00033755865,0.0013530899,0.00076650485,0.00041242948],"domain_scores_gemma":[0.99104965,0.0035748759,0.00019044879,0.0018815757,0.0028648868,0.00043861804],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005067366,0.002186734,0.0021476327,0.0039976267,0.0028662742,0.0033549152,0.002516798,0.0027175338,0.16732566],"category_scores_gemma":[0.016459627,0.0015700243,0.0012567012,0.0028356481,0.00084248616,0.006394774,0.0042945794,0.0034917668,0.0606297],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018490403,0.00079103553,0.0013231222,0.0016072433,0.00014379404,0.0005799907,0.0008836943,0.0047033173,0.02454102,0.02404855,0.66648704,0.27304217],"study_design_scores_gemma":[0.0015039084,0.0005036879,0.0037213112,0.00030929717,0.0002373215,0.0013339561,0.0011325172,0.13492554,0.10011345,0.04165852,0.71421015,0.0003503072],"about_ca_topic_score_codex":0.009203124,"about_ca_topic_score_gemma":0.009314253,"teacher_disagreement_score":0.16732566,"about_ca_system_score_codex":0.0027644865,"about_ca_system_score_gemma":0.0040615895,"threshold_uncertainty_score":0.5597601},"labels":[],"label_agreement":null},{"id":"W124635070","doi":"","title":"Language identification of names with SVMs","year":2010,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Language identification; Natural language processing; Artificial intelligence; Identification (biology); Language model; Transliteration; Natural language; Support vector machine; Task (project management); Cache language model; Machine translation; Universal Networking Language; Process (computing); Character (mathematics); Comprehension approach; Programming language","score_opus":0.004026076663038785,"score_gpt":0.2520028099435107,"score_spread":0.24797673328047193,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W124635070","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13094935,0.00063430745,0.8509447,0.00072111125,0.00026050129,0.00011474081,0.00074214546,0.012508122,0.0031250257],"genre_scores_gemma":[0.72541404,0.00021144368,0.26640388,0.00024291081,0.00017531453,0.00013463115,0.0023161543,0.00038210972,0.0047195484],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9978879,0.00085820205,0.00019417584,0.00049199373,0.0003586017,0.00020915236],"domain_scores_gemma":[0.9956696,0.0020145315,0.00050425576,0.00054903113,0.0011034997,0.00015909933],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002779977,0.00095146673,0.0010721643,0.0019986958,0.0007576979,0.0018248413,0.0011267242,0.001349874,0.0025374903],"category_scores_gemma":[0.008634915,0.00039710067,0.0009935427,0.001227401,0.0005081456,0.0035239123,0.001288194,0.0018857815,0.0033713484],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009607158,0.00046259913,0.014608785,0.00024352537,0.00015949413,0.00043684166,0.00043368648,0.080281205,0.031399354,0.00871316,0.018308189,0.8439925],"study_design_scores_gemma":[0.00001477679,0.000049592647,0.0014541995,0.00001429879,0.000015640957,0.000100621495,0.00010114488,0.9819562,0.007065268,0.007470214,0.0017369512,0.000021141557],"about_ca_topic_score_codex":0.001696127,"about_ca_topic_score_gemma":0.0014900004,"teacher_disagreement_score":0.002779977,"about_ca_system_score_codex":0.00058950623,"about_ca_system_score_gemma":0.000702159,"threshold_uncertainty_score":0.014702082},"labels":[],"label_agreement":null},{"id":"W125248212","doi":"","title":"SegCV : traitement efficace de CV avec analyse et correction d'erreurs","year":2013,"lang":"fr","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Humanities; Parsing; Artificial intelligence; Computer science; Art","score_opus":0.017105503091324054,"score_gpt":0.2613504558504024,"score_spread":0.24424495275907834,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W125248212","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19334821,0.0015262305,0.64078975,0.00066435104,0.0017245864,0.00020156967,0.0027326734,0.14951411,0.009498482],"genre_scores_gemma":[0.6365951,0.00025783546,0.33132297,0.00040287102,0.00018131029,0.00016470213,0.0040039914,0.0086121475,0.018459165],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9980463,0.00027930844,0.00012891286,0.0005622365,0.00081109814,0.00017210073],"domain_scores_gemma":[0.9960025,0.0015781944,0.00018638106,0.00085656345,0.0012305657,0.00014572879],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011431897,0.0022389288,0.0011152453,0.0018624832,0.0007263034,0.0019393751,0.0015682143,0.001972851,0.011096575],"category_scores_gemma":[0.008400842,0.0005269033,0.0007170437,0.0010899454,0.00057109055,0.0012768705,0.0012347909,0.0011074946,0.0043056817],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0032589526,0.0002729575,0.008048001,0.0004943849,0.00027195853,0.0010362383,0.0004395189,0.030423999,0.16371012,0.0018731346,0.035644915,0.7545258],"study_design_scores_gemma":[0.00025733004,0.0010264535,0.016855951,0.00009939168,0.00021864366,0.0021331336,0.0004349086,0.5714159,0.36390677,0.0045035756,0.038912024,0.00023599053],"about_ca_topic_score_codex":0.0068293116,"about_ca_topic_score_gemma":0.0058475207,"teacher_disagreement_score":0.011096575,"about_ca_system_score_codex":0.0004571915,"about_ca_system_score_gemma":0.001026235,"threshold_uncertainty_score":0.037121713},"labels":[],"label_agreement":null},{"id":"W125575914","doi":"","title":"Database-text alignment via structured multilabel classification","year":2007,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":38,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Computer science; Classifier (UML); Artificial intelligence; Task (project management); Matching (statistics); Sentence; Natural language processing; Semantics (computer science); Natural language; Database; Information retrieval","score_opus":0.021404367206959467,"score_gpt":0.29762037263246827,"score_spread":0.2762160054255088,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W125575914","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019635124,0.00022335292,0.9710198,0.00045976066,0.00010430403,0.00018788356,0.0006533691,0.0057811146,0.0019353712],"genre_scores_gemma":[0.29114112,0.00022251306,0.69387096,0.0005531016,0.00024268343,0.0005683856,0.0063695544,0.00063299184,0.0063986694],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9969848,0.0009792767,0.00017247109,0.0010591643,0.00060157984,0.00020265646],"domain_scores_gemma":[0.99580103,0.0018081665,0.00057482265,0.0007865449,0.0008967885,0.00013260962],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021003035,0.0010626636,0.0011236357,0.0029652836,0.0014161053,0.0018427953,0.0027214952,0.0017791094,0.0047925464],"category_scores_gemma":[0.008198317,0.00047192565,0.001142978,0.002917278,0.0008548488,0.004537277,0.002058389,0.0018190847,0.003617144],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006403922,0.0005999885,0.0034304573,0.00032376582,0.00010785645,0.00032329993,0.0005620466,0.08219036,0.015957512,0.01671993,0.02533124,0.85381323],"study_design_scores_gemma":[0.000033850774,0.000112273236,0.00070529676,0.000029393997,0.000041365893,0.00015607261,0.00024152004,0.9207425,0.014884018,0.05463995,0.008374941,0.000038722043],"about_ca_topic_score_codex":0.0033054843,"about_ca_topic_score_gemma":0.005846911,"teacher_disagreement_score":0.0047925464,"about_ca_system_score_codex":0.0012438056,"about_ca_system_score_gemma":0.0018985231,"threshold_uncertainty_score":0.016032636},"labels":[],"label_agreement":null},{"id":"W127041021","doi":"10.29173/cais316","title":"Measuring and Comparing Aggregation Inconsistency for Chinese Titles in Two Library Catalogues","year":2013,"lang":"en","type":"article","venue":"Proceedings of the Annual Conference of CAIS / Actes du congrès annuel de l ACSI","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Library science; Linguistics; Computer science; Philosophy","score_opus":0.024782623100585557,"score_gpt":0.2542776304345394,"score_spread":0.22949500733395384,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W127041021","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9922454,0.0007516457,0.003346963,0.000063314656,0.000020888037,0.000117792035,0.0017287759,0.00009439688,0.0016308281],"genre_scores_gemma":[0.9843315,0.00033801445,0.008284128,0.000028146751,0.000037850317,0.00018319013,0.0060684956,0.000040914045,0.0006876834],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.98169327,0.0033095307,0.005265389,0.0025909801,0.006324979,0.00081587926],"domain_scores_gemma":[0.89907396,0.044752795,0.026518948,0.008039793,0.020555083,0.001059439],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013378535,0.00049917423,0.0010513038,0.020037495,0.0014509216,0.003215998,0.001158563,0.0006946023,0.00095591496],"category_scores_gemma":[0.070026085,0.00046655667,0.0009802128,0.028022934,0.0014457378,0.0024196266,0.0026790802,0.00054548227,0.00038282017],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036896628,0.00003817543,0.9465846,0.0005072822,0.0003236872,0.0003623267,0.006519832,0.0011958036,0.0027993834,0.00041969132,0.0006949561,0.040185317],"study_design_scores_gemma":[0.000014466889,0.000082662125,0.9878193,0.000056100424,0.0002072645,0.0002835124,0.0036043592,0.003441154,0.0024350968,0.0002288632,0.0017821768,0.000045161607],"about_ca_topic_score_codex":0.032413278,"about_ca_topic_score_gemma":0.03168305,"teacher_disagreement_score":0.032413278,"about_ca_system_score_codex":0.0024473267,"about_ca_system_score_gemma":0.0023303663,"threshold_uncertainty_score":0.07075328},"labels":[],"label_agreement":null},{"id":"W12836875","doi":"10.63317/3gh6v2hd4z55","title":"From TreeBank to PropBank","year":2002,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":539,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Treebank; Computer science; Natural language processing; Artificial intelligence; Annotation","score_opus":0.01830638065933082,"score_gpt":0.252359065373384,"score_spread":0.23405268471405316,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W12836875","genre_codex":"dataset","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006398774,0.004100794,0.007627193,0.0021119157,0.000810364,0.00016520369,0.615924,0.027375596,0.3354861],"genre_scores_gemma":[0.015225936,0.0039377697,0.010947383,0.00068697304,0.0002055435,0.00024040899,0.5238849,0.014399841,0.43047124],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99967504,0.00003183394,0.000033576365,0.00013257528,0.000097921526,0.000029045843],"domain_scores_gemma":[0.9992343,0.00018203819,0.00008174527,0.00015398399,0.00019064397,0.00015741178],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0004588611,0.00083865796,0.0008439979,0.002591837,0.00074934255,0.002936829,0.00107744,0.0007573557,0.58965397],"category_scores_gemma":[0.0017092956,0.00073584943,0.0002627937,0.004954773,0.00028960174,0.0027474326,0.0016941653,0.0011727029,0.4851124],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00040844,0.00007270627,0.00084284745,0.00069840736,0.000026084252,0.0002169646,0.00026279723,0.00023745939,0.0023523832,0.004182485,0.7922772,0.1984223],"study_design_scores_gemma":[0.000033831056,0.000010474338,0.0012610496,0.00010851987,0.0000058745286,0.00006035699,0.000054897,0.0001483977,0.0003105346,0.0015155877,0.9964805,0.000010046536],"about_ca_topic_score_codex":0.0038553197,"about_ca_topic_score_gemma":0.004996414,"teacher_disagreement_score":0.58965397,"about_ca_system_score_codex":0.0005855115,"about_ca_system_score_gemma":0.0006869199,"threshold_uncertainty_score":0.5853088},"labels":[],"label_agreement":null},{"id":"W12884002","doi":"10.1007/s00439-003-0984-7","title":"Structure de l'enonce oral spontane en thai standard (siamois) : etude prosodique et enonciative","year":2001,"lang":"en","type":"dissertation","venue":"Human Genetics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Humanities; Philosophy; Political science","score_opus":0.009884293945772825,"score_gpt":0.31779035434397757,"score_spread":0.30790606039820473,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W12884002","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9740744,0.013512879,0.004067485,0.00014213764,0.000078497804,0.000079491765,0.00030407688,0.000058569967,0.007682569],"genre_scores_gemma":[0.9688081,0.011259568,0.00432502,0.00006690735,0.000084652005,0.00007452886,0.0005111278,0.0000507148,0.014819356],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9995307,0.00006360653,0.000033203065,0.000165821,0.00014511148,0.000061576524],"domain_scores_gemma":[0.99934405,0.00020388946,0.00015579749,0.00006529725,0.000116067844,0.000114908544],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00073167455,0.000692578,0.00044779238,0.0019106488,0.00091988884,0.0020802042,0.0004684216,0.0009751507,0.0054230196],"category_scores_gemma":[0.0013876165,0.00046029076,0.00053363614,0.0013476735,0.0015797534,0.00065392273,0.00047512297,0.0010253974,0.0013845164],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0024845675,0.0005835297,0.21492934,0.0012757116,0.00039050492,0.057863317,0.008640516,0.0018039407,0.22705989,0.010175349,0.0013731983,0.47342017],"study_design_scores_gemma":[0.00014013458,0.0022557252,0.48031592,0.00022625818,0.00040802517,0.28488684,0.00622235,0.003129767,0.06482429,0.0015698303,0.15585457,0.00016634323],"about_ca_topic_score_codex":0.011504852,"about_ca_topic_score_gemma":0.0059948033,"teacher_disagreement_score":0.011504852,"about_ca_system_score_codex":0.0013457392,"about_ca_system_score_gemma":0.00078075874,"threshold_uncertainty_score":0.022875786},"labels":[],"label_agreement":null},{"id":"W131108392","doi":"10.21437/interspeech.2008-601","title":"Towards domain independence in machine aided human translation","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Independence (probability theory); Computer science; Machine translation; Translation (biology); Domain (mathematical analysis); Artificial intelligence; Natural language processing; Mathematics; Statistics","score_opus":0.025473597164371805,"score_gpt":0.2905562772338025,"score_spread":0.2650826800694307,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W131108392","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009520557,0.00091089605,0.9840719,0.0006011229,0.000062248975,0.000055996756,0.00010595049,0.0016127013,0.0030585907],"genre_scores_gemma":[0.29002222,0.001849673,0.69797623,0.00088474824,0.0003553893,0.00029992507,0.0012081381,0.0006555638,0.0067480993],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99112344,0.005277528,0.00040674064,0.0011292776,0.0018087692,0.00025424542],"domain_scores_gemma":[0.9849331,0.007798734,0.0007216153,0.0039239815,0.0024348919,0.00018768317],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0073666936,0.001063372,0.0012167437,0.0017295866,0.0008877289,0.002208808,0.0014868765,0.0015667239,0.0029472231],"category_scores_gemma":[0.016503664,0.0010392175,0.0011761681,0.0016657701,0.0018620401,0.004597528,0.0038044436,0.00308528,0.004956508],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00081957295,0.0003194785,0.0031623954,0.0008396753,0.00034521404,0.00045586724,0.0008818259,0.10269016,0.079577655,0.081206836,0.008263799,0.7214376],"study_design_scores_gemma":[0.0001178416,0.00035189392,0.0020683315,0.00011549435,0.00017974302,0.00070233556,0.00030317163,0.7397719,0.07145145,0.15342465,0.031398173,0.00011506768],"about_ca_topic_score_codex":0.00128736,"about_ca_topic_score_gemma":0.0015149413,"teacher_disagreement_score":0.0073666936,"about_ca_system_score_codex":0.00068062014,"about_ca_system_score_gemma":0.0011696189,"threshold_uncertainty_score":0.038959205},"labels":[],"label_agreement":null},{"id":"W131127834","doi":"","title":"TALP at TAC 2008: A Semantic Approach to Recognizing Textual Entailment","year":2008,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Textual entailment; Natural language processing; Computer science; Logical consequence; Artificial intelligence; Classifier (UML); AdaBoost; Baseline (sea); Semantic role labeling; Sentence","score_opus":0.013547297476799204,"score_gpt":0.25204040553220847,"score_spread":0.23849310805540927,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W131127834","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10803089,0.0017457919,0.76972747,0.0018971884,0.00078605604,0.0029405549,0.017220883,0.07478358,0.022867618],"genre_scores_gemma":[0.19264919,0.00047897844,0.75266886,0.0007077566,0.00026025518,0.0010100335,0.044511102,0.0017758469,0.0059379893],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99526703,0.0015599402,0.0003837928,0.0010673017,0.0014610256,0.00026094363],"domain_scores_gemma":[0.99477565,0.0021483207,0.00035195344,0.0009832456,0.0013519563,0.00038875674],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004343576,0.0018928185,0.0014510273,0.004895802,0.0022035975,0.0025925892,0.003429956,0.002728878,0.012730353],"category_scores_gemma":[0.011881492,0.0006273294,0.00179128,0.002588916,0.0010781002,0.0075841793,0.003116051,0.0032829207,0.006313428],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021450764,0.0023416346,0.0068993857,0.0026212037,0.00047934792,0.0010092211,0.0013711883,0.013362956,0.06683865,0.027844924,0.1634833,0.7116031],"study_design_scores_gemma":[0.0008338464,0.001897756,0.009335454,0.0002145624,0.00054466847,0.0026095437,0.0017899302,0.68690926,0.091381155,0.055256296,0.14895168,0.00027581275],"about_ca_topic_score_codex":0.0066002593,"about_ca_topic_score_gemma":0.008993453,"teacher_disagreement_score":0.012730353,"about_ca_system_score_codex":0.0014572338,"about_ca_system_score_gemma":0.0025403642,"threshold_uncertainty_score":0.04258728},"labels":[],"label_agreement":null},{"id":"W133184457","doi":"","title":"Basic Concepts of Lexical Resource Semantics","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":40,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Underspecification; Head-driven phrase structure grammar; Natural language processing; Generative grammar; Artificial intelligence; Linguistics; Semantics (computer science); Programming language","score_opus":0.020669437584995547,"score_gpt":0.28731959773484855,"score_spread":0.266650160149853,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W133184457","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0056851343,0.00967909,0.7471751,0.013484359,0.001271359,0.00039842777,0.0016310724,0.0012586095,0.2194169],"genre_scores_gemma":[0.35790217,0.01252397,0.5618302,0.007543319,0.005253525,0.0026814349,0.003351552,0.0013952192,0.04751862],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99494296,0.0019570356,0.0006583038,0.0009327998,0.0011594554,0.0003493902],"domain_scores_gemma":[0.9958087,0.002145051,0.00035483035,0.00086161186,0.0006296399,0.00020010954],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0048501464,0.0017344747,0.0019349037,0.0073309187,0.0032083003,0.010711959,0.004625892,0.0038483907,0.015312307],"category_scores_gemma":[0.010980131,0.0012333785,0.0023653202,0.007236432,0.016279599,0.026595274,0.0060482016,0.0061465944,0.0059100487],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000003062514,0.0000045416423,0.000028922983,0.000043605065,0.0000060033663,0.000028530652,0.0002509973,0.00013654558,0.00006732753,0.9943393,0.0011761558,0.003915128],"study_design_scores_gemma":[0.0000043032005,0.000002882878,0.00003188909,0.000045693043,0.00000365564,0.00005689032,0.00009846091,0.0005347522,0.00006613707,0.97243506,0.026713151,0.0000071119584],"about_ca_topic_score_codex":0.003038326,"about_ca_topic_score_gemma":0.0018902556,"teacher_disagreement_score":0.015312307,"about_ca_system_score_codex":0.0035850706,"about_ca_system_score_gemma":0.0028200813,"threshold_uncertainty_score":0.05122477},"labels":[],"label_agreement":null},{"id":"W135244556","doi":"10.21437/speechprosody.2002-79","title":"The prosody of questions in natural discourse","year":2002,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":51,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Prosody; Natural (archaeology); Computer science; Linguistics; Natural language processing; Natural language; Artificial intelligence; Speech recognition; History; Philosophy","score_opus":0.009899327242053137,"score_gpt":0.28189695095893785,"score_spread":0.27199762371688474,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W135244556","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9646386,0.0031074297,0.016914649,0.0004570711,0.00009297509,0.00007642046,0.0023688138,0.00010526138,0.012238841],"genre_scores_gemma":[0.9901528,0.000531113,0.005433656,0.000097947785,0.0000519799,0.00008470937,0.0019610801,0.000042660806,0.0016441287],"study_design_codex":"bench_or_experimental","study_design_gemma":"observational","domain_scores_codex":[0.9984226,0.00089785934,0.000114178154,0.0002699211,0.00023320437,0.00006225123],"domain_scores_gemma":[0.9943051,0.004467101,0.0004519311,0.00021655588,0.00046801506,0.00009130626],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011606341,0.00027859295,0.00026405777,0.001044803,0.0006536124,0.0014401034,0.0002830893,0.0005205351,0.0023914876],"category_scores_gemma":[0.0076541165,0.00018768948,0.00015535938,0.0011657004,0.0009730634,0.0013843853,0.0007754065,0.0003590348,0.0004127444],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019066714,0.00025551752,0.08900517,0.004832314,0.00020602178,0.0035193975,0.25097275,0.0022454183,0.32207137,0.047873292,0.008282383,0.2688297],"study_design_scores_gemma":[0.00011974274,0.00048140195,0.47447178,0.0006886692,0.00016654411,0.010868161,0.065956905,0.012100958,0.067153655,0.032603342,0.33516183,0.00022693479],"about_ca_topic_score_codex":0.0005827053,"about_ca_topic_score_gemma":0.0008070036,"teacher_disagreement_score":0.0023914876,"about_ca_system_score_codex":0.00032433475,"about_ca_system_score_gemma":0.00016533102,"threshold_uncertainty_score":0.008000314},"labels":[],"label_agreement":null},{"id":"W135764212","doi":"","title":"Selective analysis for automatic abstracting: evaluating indicativeness and acceptability","year":2000,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Task (project management); Information retrieval; Template; Natural language processing; Artificial intelligence; Programming language","score_opus":0.026144439670993447,"score_gpt":0.35989645926788094,"score_spread":0.3337520195968875,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W135764212","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4428969,0.0012405923,0.5348731,0.0006759993,0.00007575626,0.0007401517,0.0017331953,0.007605741,0.010158516],"genre_scores_gemma":[0.7704302,0.00030427135,0.2246493,0.00012381555,0.00006649952,0.00035205344,0.002090067,0.0006318017,0.0013520296],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9824284,0.008602264,0.0016728003,0.001677537,0.0052478234,0.0003713098],"domain_scores_gemma":[0.8943119,0.08235355,0.007541695,0.003958387,0.010670784,0.0011636687],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012446365,0.0011554313,0.00092807226,0.005013799,0.0009713271,0.0041362597,0.0012126941,0.0013997629,0.0035592099],"category_scores_gemma":[0.07087143,0.00060114625,0.00104428,0.0020178584,0.00119061,0.0039971448,0.0027235472,0.0011081282,0.0011400562],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0026808581,0.00030125846,0.06284769,0.0022172758,0.00046873905,0.00062520924,0.009018322,0.01143638,0.10134395,0.0129866805,0.0071299654,0.78894365],"study_design_scores_gemma":[0.0005099905,0.0023622455,0.1375147,0.0007521252,0.0010866597,0.0028514885,0.009548813,0.530598,0.21698172,0.058724735,0.038478415,0.0005911589],"about_ca_topic_score_codex":0.002307146,"about_ca_topic_score_gemma":0.0024125103,"teacher_disagreement_score":0.012446365,"about_ca_system_score_codex":0.001093734,"about_ca_system_score_gemma":0.0013028645,"threshold_uncertainty_score":0.065823436},"labels":[],"label_agreement":null},{"id":"W135924241","doi":"10.63317/5f2krbiappcc","title":"Combining Multiple Models for Speech Information Retrieval","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Task (project management); Set (abstract data type); Information retrieval; Training set; Artificial intelligence; Data retrieval; Natural language processing; Test set; Data set; Speech recognition","score_opus":0.029225484760872614,"score_gpt":0.26228958080988374,"score_spread":0.23306409604901113,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W135924241","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011959614,0.0018810286,0.9806393,0.00040470407,0.00014815104,0.00016407493,0.00014502472,0.0022355276,0.0024225668],"genre_scores_gemma":[0.50036454,0.0027916087,0.4844737,0.00053150894,0.00067485165,0.0007983355,0.0010102248,0.0006023843,0.008752874],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99680746,0.0012419014,0.00023335706,0.0005240639,0.0010030548,0.00019023247],"domain_scores_gemma":[0.99510115,0.003325281,0.00022572243,0.0005366665,0.0007161147,0.0000950544],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0042313663,0.002097459,0.002300422,0.0032060503,0.0007359855,0.0024218617,0.002195511,0.0021856949,0.0036893368],"category_scores_gemma":[0.0153622385,0.0010392531,0.0027609826,0.0020508969,0.0006458373,0.0046327054,0.0018139879,0.0019664462,0.003444074],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000645577,0.0002885037,0.0018754003,0.0004598293,0.0008202219,0.00044280497,0.0002912763,0.4609215,0.009051092,0.013221388,0.004828069,0.5071544],"study_design_scores_gemma":[0.00002959629,0.00010533482,0.0003080976,0.00001864035,0.0001436118,0.00014397493,0.000024983923,0.9842211,0.0018860636,0.011346029,0.0017284219,0.000044075514],"about_ca_topic_score_codex":0.004585361,"about_ca_topic_score_gemma":0.0046603824,"teacher_disagreement_score":0.004585361,"about_ca_system_score_codex":0.0012210686,"about_ca_system_score_gemma":0.00073257386,"threshold_uncertainty_score":0.022377849},"labels":[],"label_agreement":null},{"id":"W137061674","doi":"10.63317/2tjb7zqm2qsv","title":"Using the Web as a Linguistic Resource to Automatically Correct Lexico-Syntactic Errors","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":32,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Natural language processing; Sentence; Focus (optics); Artificial intelligence; Context (archaeology); Grammar; Linguistics","score_opus":0.03857461495009921,"score_gpt":0.31909493801435806,"score_spread":0.28052032306425884,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W137061674","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.068279356,0.00078480603,0.8240986,0.0006696267,0.00018952157,0.00040818754,0.002822162,0.095638864,0.007109018],"genre_scores_gemma":[0.102054805,0.00033656135,0.8849696,0.00022865218,0.00007252452,0.00023881419,0.0045350357,0.0022918084,0.005272197],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99879396,0.00030923708,0.00012564017,0.00027100852,0.00044651082,0.000053683463],"domain_scores_gemma":[0.9945575,0.0028267084,0.00044460333,0.0006048762,0.0014666895,0.00009966784],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011520002,0.000960077,0.0008907817,0.0036564402,0.0007323334,0.0016560758,0.0011823446,0.0011695994,0.004094083],"category_scores_gemma":[0.0068982365,0.0005728126,0.0006000887,0.0016354814,0.00055667834,0.0027748698,0.0010779143,0.00082959153,0.004527486],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025592904,0.00022956083,0.0078961905,0.0004011009,0.00012915125,0.0007414213,0.0004886626,0.007810798,0.02683261,0.003186646,0.027312564,0.92471534],"study_design_scores_gemma":[0.00021439396,0.00024258561,0.013722479,0.00026150493,0.00023258441,0.0033777123,0.00097094645,0.7040708,0.16294014,0.020909255,0.09281408,0.0002435389],"about_ca_topic_score_codex":0.0046199323,"about_ca_topic_score_gemma":0.005988286,"teacher_disagreement_score":0.0046199323,"about_ca_system_score_codex":0.0004435185,"about_ca_system_score_gemma":0.0013502827,"threshold_uncertainty_score":0.013696134},"labels":[],"label_agreement":null},{"id":"W137102759","doi":"","title":"Predicting the Semantic Compositionality of Prefix Verbs","year":2010,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Prefix; Principle of compositionality; Computer science; Natural language processing; Verb; Artificial intelligence; Focus (optics); Classifier (UML); Linguistics","score_opus":0.0086444183264944,"score_gpt":0.263025302831988,"score_spread":0.2543808845054936,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W137102759","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7282318,0.0010184323,0.2535212,0.0005594144,0.00015183326,0.0002500814,0.0033230537,0.0037198798,0.009224347],"genre_scores_gemma":[0.83999026,0.000443899,0.15149961,0.00007606916,0.00007846821,0.0001306165,0.005589399,0.00021428063,0.0019773098],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995871,0.00010649044,0.000043745997,0.00015666401,0.000074126314,0.000031870688],"domain_scores_gemma":[0.9979488,0.0011495923,0.00029349947,0.0001596175,0.00037407034,0.00007445548],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00071817456,0.0005837398,0.00041791314,0.0027070593,0.00083743385,0.0010699654,0.00037005538,0.0009927909,0.002521682],"category_scores_gemma":[0.0035576473,0.0003109619,0.0005966071,0.0014120002,0.0006451753,0.0028424123,0.00078488316,0.000682247,0.0016729445],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009061015,0.0003429297,0.09141096,0.00076173904,0.00011862233,0.0012420127,0.0012842345,0.014128038,0.12186271,0.017504208,0.00970689,0.7407315],"study_design_scores_gemma":[0.000063447005,0.00030512008,0.059731554,0.00015414118,0.00014877527,0.0018340374,0.0015658325,0.80908954,0.051996112,0.05143475,0.023583015,0.00009378076],"about_ca_topic_score_codex":0.002003257,"about_ca_topic_score_gemma":0.0044558384,"teacher_disagreement_score":0.0027070593,"about_ca_system_score_codex":0.00052479166,"about_ca_system_score_gemma":0.0005904889,"threshold_uncertainty_score":0.008435905},"labels":[],"label_agreement":null},{"id":"W14107570","doi":"","title":"Extracting Synonyms from Dictionary Definitions","year":2009,"lang":"en","type":"article","venue":"Recent Advances in Natural Language Processing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Natural language processing; Synonym (taxonomy); Artificial intelligence; Lexicon; Bilingual dictionary; Lemmatisation; Information retrieval","score_opus":0.013372814188784323,"score_gpt":0.2994146677861813,"score_spread":0.286041853597397,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W14107570","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10747654,0.003495861,0.87149024,0.00073485816,0.0003736606,0.00052806776,0.0032788145,0.0019252981,0.010696642],"genre_scores_gemma":[0.19234328,0.0019915057,0.7953666,0.0001878141,0.0001206755,0.0002477057,0.006198232,0.00035580402,0.003188408],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9973924,0.0008701538,0.0004804314,0.0005695848,0.00061995175,0.000067411354],"domain_scores_gemma":[0.99150497,0.004706785,0.0010305393,0.0012395896,0.0013823644,0.0001357828],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016925136,0.00078318163,0.00093841687,0.0060366997,0.0007368438,0.0018837175,0.0011625272,0.00070961326,0.0038418083],"category_scores_gemma":[0.014098249,0.00053539895,0.00079328596,0.005171304,0.0007602481,0.00480203,0.0021987276,0.0008128908,0.0022626885],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015084239,0.0001751619,0.010100598,0.0032937033,0.00022938484,0.0015297537,0.0026251406,0.004501699,0.03878283,0.0500707,0.010372473,0.87816775],"study_design_scores_gemma":[0.00019635251,0.00066030776,0.02486468,0.0014767309,0.0004524961,0.011682722,0.0077560274,0.17757018,0.17285423,0.15925564,0.44291523,0.00031545636],"about_ca_topic_score_codex":0.00046020062,"about_ca_topic_score_gemma":0.0011191617,"teacher_disagreement_score":0.0060366997,"about_ca_system_score_codex":0.00042102108,"about_ca_system_score_gemma":0.000975082,"threshold_uncertainty_score":0.012852132},"labels":[],"label_agreement":null},{"id":"W142283382","doi":"","title":"Les systèmes de résumé automatique sont-ils vraiment des mauvais élèves ?","year":2008,"lang":"fr","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Humanities; Automatic summarization; Latent semantic analysis; Computer science; Political science; Philosophy; Artificial intelligence","score_opus":0.022505797886826183,"score_gpt":0.2618247802426477,"score_spread":0.23931898235582152,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W142283382","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08639579,0.017167883,0.6933136,0.04129062,0.0028738633,0.0004060939,0.003607141,0.089926876,0.065018155],"genre_scores_gemma":[0.5581403,0.008444226,0.31526586,0.0033767903,0.003127424,0.0002813085,0.005237882,0.010418699,0.09570759],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.993962,0.0022499503,0.0005063015,0.0012166341,0.0016791992,0.00038588225],"domain_scores_gemma":[0.9656659,0.019408973,0.0015234011,0.007271585,0.005191128,0.00093898474],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0102566825,0.0011632866,0.0016489939,0.0020258566,0.0016382999,0.011096178,0.0023892666,0.0032068274,0.029996466],"category_scores_gemma":[0.054042872,0.0010956378,0.0010941055,0.0021762475,0.0025148452,0.01982569,0.001958531,0.0037072687,0.020443339],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021335557,0.0003036444,0.007382317,0.001472173,0.00025668493,0.0005146218,0.0035517795,0.012848343,0.022779586,0.090558805,0.07192148,0.78627706],"study_design_scores_gemma":[0.00043390156,0.00042990415,0.010079613,0.0008054902,0.0004576114,0.0012800487,0.0027575558,0.17314981,0.069050774,0.1523767,0.58884555,0.00033292823],"about_ca_topic_score_codex":0.009721165,"about_ca_topic_score_gemma":0.007039062,"teacher_disagreement_score":0.029996466,"about_ca_system_score_codex":0.0018831948,"about_ca_system_score_gemma":0.0025010996,"threshold_uncertainty_score":0.100348115},"labels":[],"label_agreement":null},{"id":"W1433257690","doi":"10.1007/978-3-642-41578-4_1","title":"The Role of Universal Constraints in Language Acquisition","year":2013,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Parsing; Rule-based machine translation; Grammar; Constraint (computer-aided design); Artificial intelligence; Natural language processing; Language acquisition; Second-language acquisition; Programming language; Linguistics","score_opus":0.005341153284112133,"score_gpt":0.22875116573588175,"score_spread":0.2234100124517696,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1433257690","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0949522,0.016110519,0.33994317,0.00576183,0.0003158987,0.000099624056,0.0004289097,0.0008330696,0.5415548],"genre_scores_gemma":[0.9002856,0.0068373745,0.06861292,0.0005320975,0.00028128625,0.00013730332,0.00045924337,0.00061416003,0.022240037],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9985989,0.0005087295,0.0000929729,0.00032035774,0.00031419442,0.00016488622],"domain_scores_gemma":[0.9940275,0.0046092235,0.00019579317,0.00063997123,0.00039950147,0.00012814227],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018385856,0.0005145062,0.00067201874,0.0009568987,0.0012158933,0.003619229,0.0017216728,0.001060061,0.011661433],"category_scores_gemma":[0.008762084,0.0013729354,0.0005573898,0.0011898391,0.0069953045,0.014174185,0.0027308292,0.0038293079,0.0015044586],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003226185,0.000016837674,0.0005836084,0.00018985588,0.000015367954,0.00017386487,0.0015117647,0.0008140618,0.0016374962,0.90537673,0.0020688623,0.08757928],"study_design_scores_gemma":[0.000008253095,0.000016885839,0.0010358307,0.00009358928,0.000011918348,0.00023790842,0.00032514197,0.0023654536,0.001869864,0.975654,0.018358288,0.0000228768],"about_ca_topic_score_codex":0.0032148764,"about_ca_topic_score_gemma":0.00371284,"teacher_disagreement_score":0.011661433,"about_ca_system_score_codex":0.0016181113,"about_ca_system_score_gemma":0.0015677357,"threshold_uncertainty_score":0.03901142},"labels":[],"label_agreement":null},{"id":"W14394360","doi":"","title":"User-relevant access to textual information through flexible identification of terms: a semi-automatic method and software based on a combination of n-grams and surface linguistic filters","year":2000,"lang":"en","type":"article","venue":"RIAO Conference","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Trois-Rivières","funders":"","keywords":"Computer science; Identification (biology); Personalization; Software; Term (time); Task (project management); Domain (mathematical analysis); Human–computer interaction; Artificial intelligence; Natural language processing; Information retrieval; World Wide Web; Programming language; Engineering","score_opus":0.02007228474740815,"score_gpt":0.31646219443931606,"score_spread":0.2963899096919079,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W14394360","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0064530736,0.00007455193,0.971575,0.00010324464,0.000025579468,0.00012379176,0.00029431557,0.020648556,0.00070187135],"genre_scores_gemma":[0.031372804,0.00006378341,0.9646185,0.00008121584,0.000025790421,0.00018975731,0.0006764954,0.0012695459,0.0017021126],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99727315,0.0009226844,0.00028811456,0.0005667583,0.00085330446,0.00009593992],"domain_scores_gemma":[0.98733866,0.0084513305,0.00071663514,0.0015215796,0.0016407948,0.0003310721],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030074397,0.0010701413,0.0013086562,0.0033611422,0.00082837825,0.0028502021,0.0014576981,0.0011312794,0.0054384344],"category_scores_gemma":[0.012722277,0.0006660587,0.0010105689,0.0019568224,0.00097579707,0.0030992622,0.0019195416,0.0011145673,0.0052118744],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005942557,0.00016896534,0.0015969462,0.0005758499,0.00012237854,0.00037793297,0.0017375991,0.0025736194,0.10188511,0.0077901906,0.012305024,0.87027204],"study_design_scores_gemma":[0.0002395674,0.00041102595,0.006570974,0.00039269924,0.00025702,0.0021632328,0.0015416099,0.59383345,0.25632632,0.043259297,0.09447565,0.0005291951],"about_ca_topic_score_codex":0.0018353158,"about_ca_topic_score_gemma":0.0028498105,"teacher_disagreement_score":0.0054384344,"about_ca_system_score_codex":0.0004657271,"about_ca_system_score_gemma":0.0014209604,"threshold_uncertainty_score":0.018193424},"labels":[],"label_agreement":null},{"id":"W144978919","doi":"","title":"Some Arguments for Coordination in Categorial Grammar and Combinatory Logic.","year":2005,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Trois-Rivières","funders":"","keywords":"Combinatory categorial grammar; Categorial grammar; Conjunction (astronomy); Computer science; Grammar; Programming language; Generative grammar; Combinatory logic; Theoretical computer science; Mildly context-sensitive grammar formalism; Linguistics; Artificial intelligence; Natural language processing; Emergent grammar","score_opus":0.012722177368837885,"score_gpt":0.27516748895140763,"score_spread":0.26244531158256973,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W144978919","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02680137,0.037198965,0.4180919,0.09278292,0.0023647638,0.00014569511,0.0007332088,0.0005400741,0.4213411],"genre_scores_gemma":[0.79980266,0.016247662,0.109262675,0.014443622,0.0051082224,0.00044926553,0.00084478763,0.00053127715,0.053309795],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9980089,0.0011067729,0.000096044954,0.00023448319,0.00030508643,0.000248804],"domain_scores_gemma":[0.9968111,0.0024401394,0.00016914139,0.00023790897,0.0002013744,0.00014037098],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030560626,0.0007802448,0.0005597968,0.0028243822,0.0032875766,0.0050923843,0.0016102178,0.0032601524,0.014203892],"category_scores_gemma":[0.0055971127,0.0006050792,0.0018898553,0.0025939853,0.0101306075,0.011247483,0.0027737315,0.004717758,0.0011600317],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000004607789,0.000004405118,0.000041639916,0.000014345755,0.000004145727,0.00005389859,0.00016701108,0.0001835348,0.000039779836,0.99654347,0.0014811383,0.0014619273],"study_design_scores_gemma":[0.00000445229,0.0000044638073,0.00010358792,0.0000183226,0.0000037469522,0.000032710508,0.00004919954,0.0004713994,0.000038080078,0.9921442,0.007126382,0.0000033396711],"about_ca_topic_score_codex":0.003023649,"about_ca_topic_score_gemma":0.0023675144,"teacher_disagreement_score":0.014203892,"about_ca_system_score_codex":0.0037846663,"about_ca_system_score_gemma":0.0010602388,"threshold_uncertainty_score":0.047516823},"labels":[],"label_agreement":null},{"id":"W145416232","doi":"10.1093/oso/9780199268535.003.0022","title":"Tense Interpretation in the Context of Narrative","year":2005,"lang":"en","type":"book-chapter","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":43,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Narrative; Interpretation (philosophy); Context (archaeology); Past tense; Linguistics; Present tense; History; Psychology; Cognitive science; Computer science; Epistemology; Artificial intelligence; Philosophy; Verb; Archaeology","score_opus":0.012848247673894185,"score_gpt":0.26663712488564245,"score_spread":0.25378887721174825,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W145416232","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.031174995,0.021149643,0.103665784,0.0072288816,0.00096986257,0.000058111204,0.00027083547,0.00018969168,0.83529216],"genre_scores_gemma":[0.92443967,0.007237177,0.025624286,0.0007824015,0.0009185099,0.00008993585,0.00035452715,0.00015785678,0.040395565],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.999086,0.00054197677,0.000046263587,0.0001336945,0.00013871012,0.00005331904],"domain_scores_gemma":[0.9989371,0.0007229573,0.00007674059,0.000086427375,0.00012900047,0.000047913654],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011239568,0.00062676886,0.00042330835,0.0011062608,0.0019266099,0.0042742677,0.0008431644,0.0011290519,0.005718642],"category_scores_gemma":[0.00283475,0.0005251937,0.00049296103,0.0012360122,0.005339818,0.0077822222,0.001680929,0.0029754117,0.0006861685],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000011290262,0.0000046402447,0.00004518806,0.000044204764,0.00000330534,0.00011418865,0.002469396,0.00033703892,0.00013993168,0.99056447,0.002095773,0.0041706897],"study_design_scores_gemma":[0.000007525542,0.0000068370377,0.00011066268,0.000096419484,0.0000066231974,0.000101004174,0.00063094386,0.0013951529,0.00016035508,0.95216155,0.04531504,0.000007897471],"about_ca_topic_score_codex":0.0014190638,"about_ca_topic_score_gemma":0.0014885496,"teacher_disagreement_score":0.005718642,"about_ca_system_score_codex":0.0023416993,"about_ca_system_score_gemma":0.00071095687,"threshold_uncertainty_score":0.019130766},"labels":[],"label_agreement":null},{"id":"W146018178","doi":"","title":"Simple training of dependency parsers via structured boosting","year":2007,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Boosting (machine learning); Dependency grammar; Structured prediction; Artificial intelligence; Machine learning; Conditional random field; Classifier (UML); Parsing; Dependency (UML); Markov chain; Margin (machine learning); Logistic regression","score_opus":0.020589201879145452,"score_gpt":0.2883813870853417,"score_spread":0.2677921852061963,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W146018178","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007865616,0.00013899899,0.9875503,0.00010477286,0.000034517157,0.00007803192,0.00015870444,0.0028789355,0.0011901562],"genre_scores_gemma":[0.22195171,0.0002512147,0.7717666,0.00027115757,0.00010518017,0.0004246955,0.0019343188,0.00042608535,0.0028690577],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99829966,0.0008576891,0.00008568144,0.0003746132,0.0002708235,0.00011150289],"domain_scores_gemma":[0.99634165,0.0019355889,0.00019820077,0.0007897152,0.0006328033,0.000101950995],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030712732,0.0012198251,0.0014404538,0.0011251331,0.0006379489,0.00075282896,0.0020664386,0.0012255633,0.004655817],"category_scores_gemma":[0.008062777,0.0011280252,0.0012438223,0.00127524,0.000711766,0.0024042376,0.0015004505,0.001961569,0.003837017],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00030722833,0.00033476108,0.0035962383,0.00029509616,0.00018280176,0.00027011996,0.00020785708,0.3391164,0.01386093,0.020606644,0.019434962,0.60178703],"study_design_scores_gemma":[0.00002319496,0.00005490495,0.0003689202,0.000016573194,0.000020536838,0.00006520049,0.0000110078945,0.97414,0.0043442673,0.018390825,0.0025507729,0.000013890939],"about_ca_topic_score_codex":0.0011778146,"about_ca_topic_score_gemma":0.0023655533,"teacher_disagreement_score":0.004655817,"about_ca_system_score_codex":0.00051695673,"about_ca_system_score_gemma":0.0012656227,"threshold_uncertainty_score":0.016242683},"labels":[],"label_agreement":null},{"id":"W1480908060","doi":"10.21236/ada458703","title":"Chinese-English Semantic Resource Construction","year":2000,"lang":"en","type":"report","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Defense Advanced Research Projects Agency; U.S. Department of Defense","keywords":"Computer science; Resource (disambiguation); Natural language processing; Linguistics; Philosophy","score_opus":0.010168377633719401,"score_gpt":0.2819575029610326,"score_spread":0.2717891253273132,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1480908060","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.056370404,0.0008461747,0.78497165,0.0015154268,0.00039549128,0.0025511882,0.02644192,0.011631536,0.11527619],"genre_scores_gemma":[0.19616629,0.0007731743,0.72614,0.00044453406,0.00012827774,0.0022902708,0.05137388,0.0015519352,0.021131504],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9990428,0.00019047334,0.00013948277,0.0002782979,0.0002445647,0.000104296254],"domain_scores_gemma":[0.9989736,0.00023912395,0.00006621855,0.00027870253,0.0003812781,0.000061017116],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013263191,0.0012049209,0.0008791001,0.00653789,0.0024337845,0.0021553442,0.0011439414,0.00042564166,0.016477045],"category_scores_gemma":[0.0025044447,0.0007805028,0.0013360998,0.006441425,0.0013493758,0.004615273,0.0037245227,0.0013480843,0.0055933907],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018149664,0.00026668687,0.0044564283,0.001267542,0.00007884076,0.0012782178,0.004184447,0.0046156724,0.024850136,0.5163704,0.081387065,0.36106318],"study_design_scores_gemma":[0.000090012014,0.000106221436,0.005817266,0.00021056028,0.00021541752,0.0010552218,0.0027779303,0.042982806,0.040483423,0.13268918,0.7734293,0.00014264254],"about_ca_topic_score_codex":0.017104218,"about_ca_topic_score_gemma":0.024619402,"teacher_disagreement_score":0.017104218,"about_ca_system_score_codex":0.0024564315,"about_ca_system_score_gemma":0.005393577,"threshold_uncertainty_score":0.055121243},"labels":[],"label_agreement":null},{"id":"W1481674030","doi":"","title":"Bidirectional segmentation for English-Korean machine translation","year":2012,"lang":"en","type":"dissertation","venue":"Summit (Simon Fraser University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Translation (biology); Computer science; Machine translation; Segmentation; Natural language processing; Artificial intelligence; Biology","score_opus":0.013215665925589367,"score_gpt":0.24382989913365954,"score_spread":0.23061423320807017,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1481674030","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020727985,0.0049530435,0.8704137,0.0010919749,0.0010156974,0.00052938855,0.010424921,0.041591097,0.04925213],"genre_scores_gemma":[0.2705691,0.003200919,0.61445254,0.00039119433,0.00031216152,0.0005698022,0.04062383,0.0046076947,0.06527281],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991333,0.000266405,0.000083091974,0.00030184243,0.00011475269,0.00010070692],"domain_scores_gemma":[0.99936026,0.00014425452,0.000041274587,0.0002363236,0.00018817728,0.000029746609],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00092321535,0.0012988626,0.00072810275,0.0013919325,0.0008985804,0.0020189025,0.0010055318,0.0009946711,0.061614964],"category_scores_gemma":[0.0021522688,0.0005944706,0.0008793368,0.0024327126,0.0005013274,0.002172482,0.0021017112,0.0010227769,0.039876137],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003649326,0.00009061421,0.0006063235,0.0005430453,0.00009699603,0.00034077736,0.00019924475,0.009273781,0.032700825,0.025175322,0.08846871,0.8421394],"study_design_scores_gemma":[0.00011403841,0.00026588485,0.0035450284,0.00031771278,0.00016856325,0.0007632124,0.000584947,0.5636828,0.079802215,0.107105754,0.24352963,0.00012024553],"about_ca_topic_score_codex":0.00696365,"about_ca_topic_score_gemma":0.009592897,"teacher_disagreement_score":0.061614964,"about_ca_system_score_codex":0.00060622446,"about_ca_system_score_gemma":0.0014697243,"threshold_uncertainty_score":0.20612258},"labels":[],"label_agreement":null},{"id":"W1484029852","doi":"10.1007/11424918_43","title":"English to Chinese Translation of Prepositions","year":2005,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Machine translation; Natural language processing; WordNet; Artificial intelligence; Focus (optics); Task (project management); Set (abstract data type); Process (computing); Semantic interpretation; Example-based machine translation; Translation (biology); Programming language","score_opus":0.00973105556912057,"score_gpt":0.26345571797641587,"score_spread":0.2537246624072953,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1484029852","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13319959,0.0066774543,0.2588696,0.0042487048,0.007437122,0.0006639611,0.021267524,0.012827229,0.5548088],"genre_scores_gemma":[0.5201037,0.006868937,0.24754612,0.0011765113,0.00095617224,0.00034590013,0.02602884,0.0051392857,0.19183452],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9996593,0.00007155267,0.000051992167,0.000082156505,0.000092764945,0.000042299333],"domain_scores_gemma":[0.999395,0.00012652788,0.000027161252,0.00012713255,0.0002995725,0.000024481134],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004049576,0.00096302613,0.0004773278,0.000666358,0.00063500146,0.001294516,0.0005005174,0.00037882783,0.03340487],"category_scores_gemma":[0.0012430447,0.00035068425,0.00039623925,0.0013159749,0.00060388434,0.0012613018,0.0008132569,0.0008144819,0.017126061],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00086865673,0.00017854378,0.001249376,0.002504375,0.000054297103,0.00350912,0.0042761895,0.002293769,0.08162609,0.1792282,0.18845683,0.53575456],"study_design_scores_gemma":[0.000108697874,0.00020949786,0.0023256617,0.00025354876,0.0001188476,0.0019740232,0.0011123011,0.008367653,0.08175719,0.02151418,0.88220716,0.000051308332],"about_ca_topic_score_codex":0.0034771452,"about_ca_topic_score_gemma":0.0033769798,"teacher_disagreement_score":0.03340487,"about_ca_system_score_codex":0.0006967706,"about_ca_system_score_gemma":0.0015481338,"threshold_uncertainty_score":0.111750364},"labels":[],"label_agreement":null},{"id":"W1485211753","doi":"","title":"Continuum companion to systemic functional linguistics","year":2009,"lang":"en","type":"book","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":281,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Context (archaeology); Index (typography); Sociology; Applied linguistics; Library science; Linguistics; Art history; History; Philosophy; Archaeology; Computer science","score_opus":0.014789268307401257,"score_gpt":0.25234025908257357,"score_spread":0.2375509907751723,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1485211753","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0005844922,0.08788698,0.056429055,0.021999568,0.020722838,0.00015985973,0.0037529191,0.002621012,0.8058433],"genre_scores_gemma":[0.015945457,0.05916836,0.02354996,0.0055295536,0.011269982,0.0003931197,0.0058489284,0.0020037424,0.87629086],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99925345,0.00024130108,0.00008181913,0.00015005676,0.00022389073,0.000049415652],"domain_scores_gemma":[0.9982614,0.00089846423,0.00005079667,0.00019327654,0.00047222723,0.00012383485],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011318,0.0010701821,0.0010756396,0.0037392674,0.0018558592,0.0044465256,0.0009801679,0.0015319707,0.15347406],"category_scores_gemma":[0.0033855783,0.0005665084,0.00072811183,0.004923536,0.0021083502,0.005222548,0.0018902556,0.0027413578,0.09172797],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000008826185,0.000011986166,0.00007087864,0.00026708838,0.0000046847226,0.000058982005,0.00027698447,0.00011599749,0.00012923277,0.07108809,0.8479085,0.08005889],"study_design_scores_gemma":[0.0000020845494,0.0000027265155,0.0000968936,0.00013068411,0.0000012304039,0.00008459758,0.000055777273,0.00008118463,0.00002625351,0.015349822,0.98416567,0.000003163432],"about_ca_topic_score_codex":0.0039962432,"about_ca_topic_score_gemma":0.005784364,"teacher_disagreement_score":0.15347406,"about_ca_system_score_codex":0.0033783428,"about_ca_system_score_gemma":0.0025982093,"threshold_uncertainty_score":0.5134219},"labels":[],"label_agreement":null},{"id":"W1485394606","doi":"","title":"Towards an Information-Based Procedural Grammar for Natural Language Understanding","year":2003,"lang":"en","type":"article","venue":"DOAJ (DOAJ: Directory of Open Access Journals)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"","keywords":"Computer science; Grammar; Syntax; Generative grammar; Linguistics; Emergent grammar; Affix grammar; Natural language; Natural language processing; Artificial intelligence; Natural (archaeology); History; Philosophy","score_opus":0.17076591193674176,"score_gpt":0.5137270190358065,"score_spread":0.3429611070990648,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1485394606","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004028553,0.00014014164,0.9887033,0.0011026096,0.00003474221,0.00008215552,0.00012265793,0.00043901874,0.0053467317],"genre_scores_gemma":[0.17252934,0.00041180998,0.8218822,0.0004833759,0.00010604485,0.0004323017,0.00052284345,0.00037933592,0.0032528052],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99784625,0.0010589403,0.00016753956,0.0003214326,0.0004770633,0.0001287999],"domain_scores_gemma":[0.99750346,0.0012386338,0.0002073368,0.0005275353,0.00039355518,0.00012935109],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0040369025,0.00062874175,0.0008771799,0.0018034119,0.0011955843,0.004533202,0.0022389488,0.0021730196,0.0037329402],"category_scores_gemma":[0.0062973755,0.0007922438,0.002410762,0.0015178865,0.0072370586,0.0067146295,0.0035510762,0.003212089,0.0011183624],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000065488125,0.000008573007,0.00008549611,0.000040191546,0.000007882759,0.000050742572,0.0006278247,0.0053803194,0.00044473272,0.9859839,0.0005266288,0.0068370965],"study_design_scores_gemma":[0.000009826359,0.000009416633,0.000049914794,0.000042921536,0.000010123722,0.000045066918,0.000116499614,0.040653635,0.0003723972,0.95057935,0.008100383,0.0000103801995],"about_ca_topic_score_codex":0.0044854498,"about_ca_topic_score_gemma":0.0039834334,"teacher_disagreement_score":0.004533202,"about_ca_system_score_codex":0.0023392087,"about_ca_system_score_gemma":0.002915448,"threshold_uncertainty_score":0.02134943},"labels":[],"label_agreement":null},{"id":"W1490722843","doi":"10.1007/11812128_31","title":"Parsing Computer Languages with an Automaton Compiled from a Single Regular Expression","year":2006,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Parsing; Regular expression; Automaton; Programming language; Expression (computer science); Natural language processing; Artificial intelligence","score_opus":0.010131355331903085,"score_gpt":0.24302249351589006,"score_spread":0.23289113818398696,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1490722843","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06346225,0.00013226517,0.8866833,0.00023280169,0.00019788244,0.00017693802,0.0006918141,0.043483414,0.0049394458],"genre_scores_gemma":[0.36175475,0.00019108907,0.6231351,0.00018072514,0.0000802575,0.0001901416,0.0019004951,0.006151286,0.0064162575],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99924767,0.00015215446,0.000085040214,0.000286776,0.00014205801,0.00008627469],"domain_scores_gemma":[0.9982169,0.0009375139,0.00009313813,0.0005287929,0.00018700185,0.000036589154],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006308516,0.00066867075,0.0009336598,0.0008712631,0.0008137721,0.0014990945,0.0011580036,0.000896991,0.0037684592],"category_scores_gemma":[0.0028366242,0.0011378326,0.0017552217,0.00097026065,0.0011418913,0.0024129567,0.0013587453,0.0015424276,0.0022130474],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014136571,0.00053951715,0.007893508,0.0017886532,0.00040539572,0.0025162941,0.003859289,0.08824848,0.16880484,0.20307876,0.02478346,0.49666816],"study_design_scores_gemma":[0.0001422281,0.0004368251,0.001929625,0.0002626593,0.00063988456,0.0009772001,0.00059328665,0.46824273,0.2480192,0.23042504,0.048123028,0.00020829329],"about_ca_topic_score_codex":0.0014453558,"about_ca_topic_score_gemma":0.0021021322,"teacher_disagreement_score":0.0037684592,"about_ca_system_score_codex":0.0005513794,"about_ca_system_score_gemma":0.0011168038,"threshold_uncertainty_score":0.0126068},"labels":[],"label_agreement":null},{"id":"W1493557574","doi":"10.4324/9781315749129","title":"Routledge Encyclopedia of Translation Technology","year":2014,"lang":"en","type":"book","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":202,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Encyclopedia; Translation (biology); History; Library science; Computer science; Biology","score_opus":0.008834097611382362,"score_gpt":0.2447141565008683,"score_spread":0.23588005888948593,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1493557574","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0005931752,0.09674465,0.0066026757,0.0115267625,0.007877338,0.00015446475,0.0057758745,0.0015962513,0.8691288],"genre_scores_gemma":[0.010214487,0.10115694,0.00507023,0.0031670614,0.0019073822,0.00035823937,0.0062581575,0.0015419179,0.87032557],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9984906,0.0003159812,0.00022944425,0.00022086324,0.000603882,0.0001392101],"domain_scores_gemma":[0.99791354,0.00074568996,0.00013717842,0.00045445125,0.00062515575,0.00012406375],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0010603691,0.001330218,0.0014413546,0.0043990663,0.0018655814,0.0108464835,0.0020167418,0.0027032103,0.30314597],"category_scores_gemma":[0.00546538,0.0006905882,0.00070176035,0.012146534,0.0020922688,0.009790854,0.0035624567,0.0031296099,0.22234441],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000025827763,0.000019752437,0.0001308198,0.0014576142,0.000011055287,0.0002607007,0.0014764307,0.00021913438,0.00035471696,0.08030535,0.61781,0.2979287],"study_design_scores_gemma":[0.000001687734,0.000002569063,0.00007445336,0.00037821947,0.0000016052351,0.00007896615,0.00022126654,0.000031819338,0.000040341965,0.004128258,0.9950362,0.000004621347],"about_ca_topic_score_codex":0.007887505,"about_ca_topic_score_gemma":0.0066850665,"teacher_disagreement_score":0.30314597,"about_ca_system_score_codex":0.003752448,"about_ca_system_score_gemma":0.007038189,"threshold_uncertainty_score":0.99397767},"labels":[],"label_agreement":null},{"id":"W1493931908","doi":"10.1007/978-3-642-15754-7_74","title":"Unsupervised Morphological Analysis by Formal Analogy","year":2010,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Analogy; Morpheme; Lexicon; Artificial intelligence; Natural language processing; Relation (database); Formal grammar; Formal description; Reading (process); Formal language; Linguistics; Rule-based machine translation; Algorithm; Programming language; Data mining","score_opus":0.010955837824513343,"score_gpt":0.2534533791715864,"score_spread":0.24249754134707305,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1493931908","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004791414,0.00015988197,0.98785573,0.00008987778,0.00006407314,0.000042885098,0.00009784719,0.0024251272,0.004473082],"genre_scores_gemma":[0.12596065,0.0004516583,0.86016643,0.0001233768,0.00013748485,0.00011446171,0.0011978035,0.0013954735,0.01045275],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99914825,0.00018969504,0.00006925274,0.00027633389,0.00024407273,0.000072407005],"domain_scores_gemma":[0.9989059,0.00039198148,0.000076857286,0.00039051566,0.00020557486,0.000029281497],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006484604,0.0007196115,0.00086947967,0.0027067119,0.0010166623,0.002007992,0.0015396073,0.00088357145,0.014769703],"category_scores_gemma":[0.0021752547,0.00065274094,0.0017574547,0.0024705764,0.0013850479,0.003075282,0.0023586035,0.0013775148,0.009686171],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000116412746,0.00006539648,0.0005348247,0.0003390576,0.00007301985,0.0003034764,0.00024002118,0.009191265,0.048753873,0.10350998,0.006639128,0.8302335],"study_design_scores_gemma":[0.000052112588,0.00012912673,0.0019774693,0.00011268789,0.00010766514,0.0018221104,0.00040235918,0.39968014,0.04937289,0.4856062,0.060641985,0.00009521771],"about_ca_topic_score_codex":0.00042965324,"about_ca_topic_score_gemma":0.0007861021,"teacher_disagreement_score":0.014769703,"about_ca_system_score_codex":0.00034966203,"about_ca_system_score_gemma":0.0007461829,"threshold_uncertainty_score":0.04940957},"labels":[],"label_agreement":null},{"id":"W1496822098","doi":"10.1023/a:1013136005350","title":"Towards a Lexicographic Approach to Lexical Transfer in Machine Translation (Illustrated by the German–Russian Language Pair)","year":2001,"lang":"en","type":"article","venue":"Machine Translation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":26,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computational linguistics; Computer science; Natural language processing; Artificial intelligence; Linguistics; Philosophy","score_opus":0.021196146565790542,"score_gpt":0.2848283274894695,"score_spread":0.26363218092367896,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1496822098","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0043443246,0.0027398174,0.9649716,0.0034729894,0.00043836617,0.000072174524,0.00010125069,0.0006847415,0.023174746],"genre_scores_gemma":[0.10910444,0.0044842935,0.8728415,0.000817676,0.00056346913,0.00022759322,0.0003536938,0.000497307,0.011110085],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9984131,0.0010262058,0.000103958206,0.00016018304,0.00022586608,0.00007068352],"domain_scores_gemma":[0.9986364,0.0008390363,0.000066511646,0.00021425082,0.00021260034,0.000031116888],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002319196,0.0006491313,0.0006879511,0.0029088156,0.001662549,0.0050003123,0.0011265568,0.0017069458,0.008440513],"category_scores_gemma":[0.0041960785,0.0008869142,0.0009805453,0.003269722,0.005242516,0.0074267555,0.0028242012,0.0019847532,0.004532853],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000080899576,0.00005814488,0.00022612422,0.00038412938,0.000025038085,0.00029458356,0.0010962683,0.0039110747,0.005376738,0.84786254,0.006335361,0.1343491],"study_design_scores_gemma":[0.000033933626,0.000053738564,0.00023715495,0.00013005036,0.000023458937,0.00044066226,0.000533872,0.028648457,0.00721441,0.9127493,0.049895022,0.00003984054],"about_ca_topic_score_codex":0.0014851665,"about_ca_topic_score_gemma":0.0018909231,"teacher_disagreement_score":0.008440513,"about_ca_system_score_codex":0.0010222436,"about_ca_system_score_gemma":0.001510561,"threshold_uncertainty_score":0.02823639},"labels":[],"label_agreement":null},{"id":"W1498426425","doi":"10.3917/rfla.131.0097","title":"Ressources lexicales, terminologiques et ontologiques : une analyse comparative dans le domaine de l'informatique","year":2008,"lang":"fr","type":"article","venue":"Revue française de linguistique appliquée","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Humanities; Philosophy; Political science","score_opus":0.029028557463253737,"score_gpt":0.31342194878258484,"score_spread":0.2843933913193311,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1498426425","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5950373,0.012365267,0.3214254,0.0021748624,0.0002349557,0.00049600075,0.0020855754,0.0013906892,0.06478998],"genre_scores_gemma":[0.8415596,0.0058336277,0.13023414,0.00028184798,0.00011661187,0.0004023606,0.002673184,0.0006038052,0.018294837],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.996463,0.0010190189,0.00030312,0.0003867883,0.0017117048,0.00011634996],"domain_scores_gemma":[0.9854848,0.010038612,0.0007893239,0.0007884384,0.0027541365,0.00014470359],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033030973,0.0004854098,0.00043785022,0.0054608695,0.001272092,0.0046269065,0.00034363873,0.00081652997,0.0030866885],"category_scores_gemma":[0.010465545,0.00037139305,0.000613721,0.005673673,0.001953655,0.0046555605,0.0013859617,0.00095332734,0.0010102133],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00077979185,0.00023186172,0.049787514,0.003176044,0.00022789184,0.0028243386,0.07207433,0.0037900128,0.0965546,0.08598126,0.0043754354,0.68019694],"study_design_scores_gemma":[0.00006655238,0.00037719883,0.23330341,0.0021437167,0.0007341355,0.008252262,0.079490654,0.025034424,0.09073868,0.062049646,0.49748176,0.00032755634],"about_ca_topic_score_codex":0.0056477725,"about_ca_topic_score_gemma":0.0050185993,"teacher_disagreement_score":0.0056477725,"about_ca_system_score_codex":0.0018204014,"about_ca_system_score_gemma":0.002050562,"threshold_uncertainty_score":0.017468631},"labels":[],"label_agreement":null},{"id":"W1499417086","doi":"10.46430/phen0010","title":"Keywords in Context (Using n-grams) with Python","year":2012,"lang":"en","type":"article","venue":"The Programming Historian","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Python (programming language); Computer science; Programming language; Mathematics","score_opus":0.01831539175796133,"score_gpt":0.2614360124664428,"score_spread":0.2431206207084815,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1499417086","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0038869607,0.00016502268,0.71339256,0.0014574053,0.000565678,0.00014251033,0.012732681,0.2439582,0.023699082],"genre_scores_gemma":[0.06311396,0.00046300513,0.849654,0.0011931775,0.0002370682,0.00043224246,0.010070514,0.045074426,0.029761579],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99891055,0.00019288648,0.00013771796,0.00027897826,0.0003978049,0.00008201452],"domain_scores_gemma":[0.9978739,0.0010670441,0.00011738375,0.0005727736,0.0002752544,0.00009357676],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011635759,0.001142474,0.00065072597,0.0015601335,0.00080431433,0.002744013,0.00207312,0.0008134623,0.045997888],"category_scores_gemma":[0.008881054,0.0008867106,0.0012198624,0.0019008052,0.0006987579,0.0067220945,0.0029004302,0.0030075326,0.028145831],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004548262,0.000180585,0.0034641651,0.0011648593,0.00011031238,0.0004221058,0.0021053967,0.0040523335,0.009917213,0.08091616,0.2996589,0.5975532],"study_design_scores_gemma":[0.000082128594,0.00005864638,0.0020661252,0.00042076054,0.000067204106,0.00055579445,0.00048056032,0.05257644,0.029868208,0.21216287,0.70151323,0.00014817093],"about_ca_topic_score_codex":0.0028916248,"about_ca_topic_score_gemma":0.004421316,"teacher_disagreement_score":0.045997888,"about_ca_system_score_codex":0.0005978574,"about_ca_system_score_gemma":0.0015450639,"threshold_uncertainty_score":0.15387827},"labels":[],"label_agreement":null},{"id":"W1499594845","doi":"10.1002/9780470253441.ch4","title":"Advances in Hidden Markov Models for Sequence Annotation","year":2007,"lang":"en","type":"other","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Hidden Markov model; Viterbi algorithm; Annotation; Sequence (biology); Computer science; Maximum-entropy Markov model; Sequence labeling; Forward algorithm; Markov model; Markov chain; Artificial intelligence; Viterbi decoder; Decoding methods; Variable-order Markov model; Machine learning; Algorithm; Engineering; Biology; Genetics","score_opus":0.021459203736847187,"score_gpt":0.3228139653249338,"score_spread":0.3013547615880866,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1499594845","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00046366005,0.003876798,0.9896511,0.00066374853,0.00046035793,0.000030863863,0.00043740173,0.0017385909,0.0026774502],"genre_scores_gemma":[0.026004136,0.014053963,0.9381505,0.0006741702,0.0010116893,0.00032885224,0.003306744,0.0016779895,0.01479198],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9975491,0.0009537083,0.0001678168,0.00058555615,0.00065697666,0.00008681894],"domain_scores_gemma":[0.99187356,0.0057758573,0.00021299033,0.001204697,0.0008515982,0.000081414706],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0040127416,0.0017201786,0.002171373,0.0019370025,0.0006844727,0.0027043717,0.003299727,0.002247132,0.009346596],"category_scores_gemma":[0.0143295,0.0015396094,0.0022442683,0.0034791746,0.0011514379,0.0057695718,0.0020251079,0.005882435,0.008606235],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010548719,0.000095979325,0.0006455481,0.0008388581,0.0001926948,0.00019111705,0.00036674654,0.10334413,0.003701505,0.28071478,0.036911007,0.5728921],"study_design_scores_gemma":[0.000024025168,0.000027787624,0.00035337606,0.00023825113,0.00005919208,0.00009960835,0.00004820876,0.5749813,0.0031360066,0.34758875,0.073374815,0.000068588095],"about_ca_topic_score_codex":0.008308005,"about_ca_topic_score_gemma":0.0071892766,"teacher_disagreement_score":0.009346596,"about_ca_system_score_codex":0.0019508703,"about_ca_system_score_gemma":0.0022192725,"threshold_uncertainty_score":0.031267524},"labels":[],"label_agreement":null},{"id":"W1499801486","doi":"10.4324/9781315044811","title":"Intelligent Language Tutors","year":2013,"lang":"it","type":"book","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":94,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Linguistics; Natural language processing; Philosophy","score_opus":0.014736274244100067,"score_gpt":0.2717856103662662,"score_spread":0.2570493361221661,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1499801486","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03386292,0.012283345,0.14725907,0.033945877,0.0095826555,0.0005786026,0.0009797195,0.016696094,0.74481165],"genre_scores_gemma":[0.25200567,0.0049324897,0.07401984,0.012010339,0.0016147487,0.00052617205,0.0017810529,0.0008803862,0.6522294],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9980996,0.0007221746,0.000129027,0.00036464923,0.0004798678,0.00020476583],"domain_scores_gemma":[0.9975406,0.00076008704,0.00014472715,0.00030418465,0.00066428434,0.00058614335],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017988647,0.0006773391,0.0007226814,0.0007517106,0.0013451845,0.0045454474,0.001604054,0.002678619,0.08794896],"category_scores_gemma":[0.009140792,0.00030783872,0.00041937956,0.000599605,0.0007280025,0.0044162874,0.0031945035,0.0016634872,0.04432664],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021632819,0.00036533125,0.0024013785,0.00034131267,0.00002315081,0.00047075434,0.0029014742,0.0017179874,0.0020429797,0.0812846,0.32339793,0.5848368],"study_design_scores_gemma":[0.00007876901,0.00012985352,0.0005069238,0.00013577317,0.000019004648,0.0006317752,0.0009208121,0.0053865924,0.0012179133,0.031961348,0.9589851,0.00002606929],"about_ca_topic_score_codex":0.00034100752,"about_ca_topic_score_gemma":0.00057595456,"teacher_disagreement_score":0.08794896,"about_ca_system_score_codex":0.0010620144,"about_ca_system_score_gemma":0.0011715312,"threshold_uncertainty_score":0.2942186},"labels":[],"label_agreement":null},{"id":"W1500096219","doi":"10.1007/978-94-017-2535-4_19","title":"Evaluation of parallel text alignment systems","year":2000,"lang":"en","type":"book-chapter","venue":"Text, speech and language technology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":59,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Sentence; Spotting; Computer science; Natural language processing; Word (group theory); Task (project management); Artificial intelligence; Track (disk drive); Translation (biology); Speech recognition; Linguistics; Engineering","score_opus":0.016106226271843473,"score_gpt":0.2729720834832077,"score_spread":0.2568658572113642,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1500096219","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8440456,0.006944253,0.09669073,0.0008759784,0.0012051639,0.001247492,0.0074750977,0.023186816,0.018328745],"genre_scores_gemma":[0.7778062,0.0022892365,0.15598567,0.00029604812,0.00040845753,0.0006855097,0.04701161,0.0016723026,0.013845002],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9932856,0.0028116636,0.0008673299,0.0012852058,0.0014578985,0.0002922559],"domain_scores_gemma":[0.9821951,0.009922982,0.0006820073,0.00183728,0.0048185675,0.00054404937],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004999169,0.0019941858,0.0017511344,0.0025745463,0.0013031308,0.002436953,0.0026934263,0.0016651406,0.010527487],"category_scores_gemma":[0.018833654,0.00069000234,0.00072746334,0.003713729,0.000684885,0.0033571497,0.0022927239,0.0009870619,0.0047191707],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.012891811,0.0031529698,0.006836112,0.002436546,0.00088587473,0.0007317816,0.0007212489,0.068462275,0.05441408,0.0027720719,0.032825496,0.8138697],"study_design_scores_gemma":[0.0033745193,0.007126301,0.015692383,0.00014679128,0.0011371817,0.001128515,0.0014572918,0.7759889,0.16226421,0.006946959,0.02452906,0.00020794458],"about_ca_topic_score_codex":0.0076749194,"about_ca_topic_score_gemma":0.005474527,"teacher_disagreement_score":0.010527487,"about_ca_system_score_codex":0.001264258,"about_ca_system_score_gemma":0.0016028824,"threshold_uncertainty_score":0.03521794},"labels":[],"label_agreement":null},{"id":"W1500526171","doi":"10.1007/3-540-45153-6_24","title":"User Interface Aspects of a Translation Typing System","year":2001,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Translation (biology); User interface; Interface (matter); Translation system; Human–computer interaction; User interface design; Natural language processing; Programming language; Operating system","score_opus":0.017240486991445394,"score_gpt":0.26492466269055936,"score_spread":0.24768417569911397,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1500526171","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05135551,0.0003736706,0.8681514,0.0008491084,0.0003863863,0.00022177094,0.0004356187,0.044780612,0.033445895],"genre_scores_gemma":[0.59293634,0.0005688994,0.3259173,0.00077151257,0.00036890354,0.00024910015,0.0009658961,0.013949343,0.06427276],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99847466,0.00058761553,0.00015834549,0.00013879086,0.00052097696,0.00011961941],"domain_scores_gemma":[0.9964269,0.0020043878,0.000091402115,0.0008362542,0.00049880525,0.0001423957],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015231726,0.000779424,0.0008061262,0.00052547513,0.0009830114,0.0046890983,0.0013762678,0.0020467902,0.037665613],"category_scores_gemma":[0.0062884535,0.00082196057,0.0008002064,0.0008087792,0.0007419017,0.0036522457,0.0015806494,0.0012187092,0.008778225],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0043745423,0.00041358703,0.0052445475,0.0010106049,0.00015317657,0.005294252,0.009660882,0.007176504,0.18270789,0.12030079,0.058150712,0.60551244],"study_design_scores_gemma":[0.0005351308,0.001126407,0.0051239077,0.00042792707,0.00092599005,0.0075022057,0.0018430233,0.3996489,0.2126434,0.082778156,0.28709948,0.00034545962],"about_ca_topic_score_codex":0.0012989406,"about_ca_topic_score_gemma":0.0011479739,"teacher_disagreement_score":0.037665613,"about_ca_system_score_codex":0.0003189268,"about_ca_system_score_gemma":0.0004196265,"threshold_uncertainty_score":0.12600398},"labels":[],"label_agreement":null},{"id":"W1501838785","doi":"10.1002/9780470996591.ch73","title":"Topicalization in Asian Languages","year":2006,"lang":"en","type":"other","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":80,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Topicalization; Subject (documents); Linguistics; Computer science; Element (criminal law); World Wide Web; Political science; Philosophy; Law","score_opus":0.004956804844468982,"score_gpt":0.2644940214643192,"score_spread":0.25953721661985024,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1501838785","genre_codex":"empirical","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.53341186,0.0060386546,0.09859329,0.0031467394,0.00035848835,0.00011043129,0.0023733554,0.0023760528,0.35359114],"genre_scores_gemma":[0.94871044,0.0018885647,0.014389053,0.00017323454,0.00019641111,0.000031471332,0.0016054361,0.0004471168,0.03255825],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9996867,0.00010296793,0.000028988514,0.00007102861,0.00006111532,0.00004922683],"domain_scores_gemma":[0.99909866,0.00019895346,0.00015249549,0.00016370429,0.00028499882,0.00010126933],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00058137113,0.00019472121,0.000289111,0.0013407985,0.0011167799,0.002269798,0.00025032274,0.00015563917,0.009400557],"category_scores_gemma":[0.001478537,0.00015911432,0.00026589746,0.0028228571,0.00072564377,0.0031970355,0.0010845229,0.0008509426,0.0021931983],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00039440693,0.0002174886,0.038109627,0.0005389064,0.000074955395,0.0008608176,0.013428473,0.0023994914,0.028328361,0.40470216,0.045514014,0.46543136],"study_design_scores_gemma":[0.000086525244,0.00021633756,0.095201015,0.00033784792,0.0003472412,0.0033914386,0.023165163,0.03251902,0.037511837,0.3667019,0.44038498,0.00013679455],"about_ca_topic_score_codex":0.0055475463,"about_ca_topic_score_gemma":0.006680398,"teacher_disagreement_score":0.009400557,"about_ca_system_score_codex":0.00073872734,"about_ca_system_score_gemma":0.0010725555,"threshold_uncertainty_score":0.031447947},"labels":[],"label_agreement":null},{"id":"W1504003456","doi":"10.1007/978-3-540-30194-3_22","title":"The Contribution of End-Users to the TransType2 Project","year":2004,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; End user; World Wide Web; Human–computer interaction","score_opus":0.01125939724898427,"score_gpt":0.2647825362597617,"score_spread":0.25352313901077744,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1504003456","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.026613107,0.0030244275,0.6935897,0.009131839,0.0022646915,0.0002430971,0.0024785758,0.035554823,0.2270998],"genre_scores_gemma":[0.14551711,0.0046400935,0.480741,0.0029489787,0.0012897782,0.00040721527,0.012028286,0.03826225,0.31416535],"study_design_codex":"design_other","study_design_gemma":"qualitative","domain_scores_codex":[0.99520373,0.0020222005,0.00016964441,0.00061389885,0.0015356052,0.00045487584],"domain_scores_gemma":[0.98548466,0.004085349,0.00026594635,0.00478943,0.0033419633,0.002032683],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006308796,0.00097219466,0.00070795766,0.0011041799,0.0016294173,0.0062879617,0.0022453514,0.0016120586,0.03630606],"category_scores_gemma":[0.012634772,0.00065073866,0.00080222887,0.0018613657,0.0011062314,0.008781408,0.005409545,0.0030601348,0.02315062],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009012353,0.00044057998,0.0024720277,0.000398245,0.00004212453,0.0004561495,0.0033300999,0.00096685713,0.017814072,0.1371471,0.20937106,0.6266604],"study_design_scores_gemma":[0.000050190534,0.000115798284,0.0009938243,0.00014503016,0.00003098218,0.00096740504,0.0010035357,0.007291661,0.016687077,0.054054078,0.9185928,0.00006759074],"about_ca_topic_score_codex":0.0027500833,"about_ca_topic_score_gemma":0.0020631596,"teacher_disagreement_score":0.03630606,"about_ca_system_score_codex":0.0008896544,"about_ca_system_score_gemma":0.0027570047,"threshold_uncertainty_score":0.12145585},"labels":[],"label_agreement":null},{"id":"W1504430306","doi":"10.1007/3-540-44886-1_22","title":"Summarizing Web Sites Automatically","year":2003,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Automatic summarization; Computer science; Directory; Information retrieval; Task (project management); Process (computing); Web site; World Wide Web; Natural language processing; Artificial intelligence; Programming language; The Internet; Engineering","score_opus":0.014036065489801235,"score_gpt":0.25737254834573176,"score_spread":0.2433364828559305,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1504430306","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.096123256,0.00412062,0.77075225,0.0011896251,0.0006245138,0.0010212263,0.040635902,0.06288926,0.022643302],"genre_scores_gemma":[0.2264911,0.0021981057,0.61709523,0.00022604853,0.00044954842,0.00041548052,0.12732272,0.0031036655,0.022698188],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9989492,0.0001479408,0.00009317216,0.0003263716,0.0003896158,0.000093723],"domain_scores_gemma":[0.99812716,0.0006253173,0.00014932534,0.0004953909,0.0005224268,0.000080280115],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00054920925,0.0014582847,0.0014411344,0.009552336,0.00081223925,0.0026538824,0.0012552518,0.0009380977,0.012025817],"category_scores_gemma":[0.0038461047,0.00084547943,0.0013013093,0.00806092,0.00033894763,0.0033460474,0.001124547,0.0012059117,0.008539812],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003643359,0.00041852897,0.0081834,0.0010142114,0.00031478694,0.0004135694,0.00030286508,0.01073703,0.03078254,0.010566371,0.10305259,0.83384967],"study_design_scores_gemma":[0.00021899004,0.0004415767,0.023313375,0.0004705246,0.0013773325,0.0020570024,0.00108502,0.59593594,0.07828256,0.10090727,0.1957413,0.00016910028],"about_ca_topic_score_codex":0.0038824547,"about_ca_topic_score_gemma":0.0141850235,"teacher_disagreement_score":0.012025817,"about_ca_system_score_codex":0.00060236733,"about_ca_system_score_gemma":0.0011015832,"threshold_uncertainty_score":0.040230334},"labels":[],"label_agreement":null},{"id":"W1504725342","doi":"10.1007/11424550_7","title":"MultiText Experiments for INEX 2004","year":2005,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Task (project management); Information retrieval; Advice (programming); World Wide Web; Programming language; Engineering","score_opus":0.020118180309467035,"score_gpt":0.29629477259204245,"score_spread":0.2761765922825754,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1504725342","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7488821,0.0030718555,0.05324939,0.0040165256,0.0025672922,0.0011868344,0.060006984,0.022743018,0.10427593],"genre_scores_gemma":[0.66992277,0.00070808193,0.12587154,0.00098382,0.0006868563,0.0021403665,0.111941926,0.005077631,0.082667],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9936784,0.0040703346,0.00039775303,0.0005156725,0.0009915563,0.00034630985],"domain_scores_gemma":[0.9778276,0.014504352,0.00048081583,0.0046451697,0.0018754065,0.0006666274],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0058881342,0.0012254991,0.0013045432,0.001249566,0.002275549,0.0014986743,0.0020376309,0.0027457348,0.026830794],"category_scores_gemma":[0.021460025,0.0006124945,0.0009128381,0.0018626862,0.00089732476,0.0031811723,0.0021584323,0.0018567867,0.00949496],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.023391008,0.010077503,0.0088752555,0.00217388,0.00069236744,0.0017150457,0.0018361778,0.07605557,0.012756113,0.011941217,0.52425855,0.3262273],"study_design_scores_gemma":[0.01117796,0.012024154,0.049564272,0.0005101164,0.0009531793,0.002549651,0.0030229497,0.41299644,0.09776624,0.048385374,0.3602977,0.00075192214],"about_ca_topic_score_codex":0.005894331,"about_ca_topic_score_gemma":0.008542376,"teacher_disagreement_score":0.026830794,"about_ca_system_score_codex":0.0009356952,"about_ca_system_score_gemma":0.0010640798,"threshold_uncertainty_score":0.08975792},"labels":[],"label_agreement":null},{"id":"W1504747088","doi":"10.12681/eadd/23002","title":"Μοντελοποίηση γλώσσας για συστήματα μηχανικής μετάφρασης με μονόγλωσσο σώμα κειμένων","year":2009,"lang":"el","type":"dissertation","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Pareto principle; Metis; Computer science; Mathematical optimization; Mathematics; World Wide Web","score_opus":0.010036965880431056,"score_gpt":0.29932073552610383,"score_spread":0.28928376964567276,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1504747088","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.036446203,0.0023167187,0.06788852,0.015567791,0.0018644122,0.00016197468,0.0006932839,0.0010413439,0.87401974],"genre_scores_gemma":[0.42553836,0.003585224,0.03353548,0.0041708564,0.00067310774,0.00036067175,0.0009193341,0.0010078443,0.5302091],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9983784,0.00023726556,0.00004781211,0.00035986802,0.0007446505,0.00023191657],"domain_scores_gemma":[0.99773276,0.00062746933,0.00016174177,0.00030435377,0.0008046562,0.0003689872],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011754569,0.00061445293,0.00037120972,0.00086045865,0.002081353,0.0049723135,0.0010453565,0.0019298889,0.09333058],"category_scores_gemma":[0.0050485916,0.00037112247,0.00037634396,0.0008391125,0.0025261668,0.0042845607,0.0027608895,0.0022924175,0.029444259],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00030568425,0.00020838053,0.002353131,0.0008156771,0.000045381104,0.00077436643,0.01726828,0.0012326261,0.028604195,0.67953885,0.06519276,0.20366074],"study_design_scores_gemma":[0.000038290993,0.000066680506,0.003523368,0.00028686028,0.000032273965,0.00031563346,0.008646823,0.0007469256,0.007964733,0.13636838,0.8419511,0.000058969952],"about_ca_topic_score_codex":0.004288607,"about_ca_topic_score_gemma":0.0027614264,"teacher_disagreement_score":0.09333058,"about_ca_system_score_codex":0.002165219,"about_ca_system_score_gemma":0.0029189636,"threshold_uncertainty_score":0.31222188},"labels":[],"label_agreement":null},{"id":"W1509148430","doi":"10.1007/978-3-642-00672-2_2","title":"Towards Multi-modal Extraction and Summarization of Conversations","year":2009,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Automatic summarization; Computer science; Modal; Extraction (chemistry); Artificial intelligence; Natural language processing; Information retrieval","score_opus":0.016167516616110152,"score_gpt":0.27943726872154373,"score_spread":0.2632697521054336,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1509148430","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009323706,0.0010660004,0.97001845,0.0005110019,0.00018251647,0.0002556957,0.005496458,0.009765598,0.0033804711],"genre_scores_gemma":[0.0706796,0.00093811046,0.8912251,0.00017172896,0.0002896317,0.00036715763,0.028132645,0.0011821685,0.0070138727],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99779546,0.0006655671,0.00024717205,0.0006376946,0.00045881994,0.00019531579],"domain_scores_gemma":[0.9951302,0.0021330423,0.00039034383,0.0007497797,0.0014331498,0.00016350906],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017269982,0.0017862618,0.0014585054,0.004743241,0.0015551997,0.003828366,0.0021851792,0.0014432913,0.00842386],"category_scores_gemma":[0.005876476,0.00090261956,0.0022204015,0.0034842105,0.00054155535,0.004014844,0.00308378,0.002539242,0.008942336],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004957831,0.00016909861,0.00096787635,0.0014511473,0.00018332634,0.00034807032,0.0019261335,0.005534572,0.06282947,0.016395206,0.037731804,0.8719674],"study_design_scores_gemma":[0.00011695816,0.0003517709,0.0062669055,0.00061931845,0.0007307909,0.0010832237,0.0038469802,0.53423715,0.1454222,0.12635425,0.18073025,0.00024019137],"about_ca_topic_score_codex":0.0024010136,"about_ca_topic_score_gemma":0.004505905,"teacher_disagreement_score":0.00842386,"about_ca_system_score_codex":0.0008742969,"about_ca_system_score_gemma":0.0015668028,"threshold_uncertainty_score":0.028180659},"labels":[],"label_agreement":null},{"id":"W1510410500","doi":"10.21427/d7mg78","title":"The Extent of Clientelism in Irish Politics: Evidence from Classifying Dáil Questions on a Local-National Dimension","year":2010,"lang":"en","type":"article","venue":"Arrow@dit (Dublin Institute of Technology)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Science Foundation Ireland; Canada Millennium Scholarship Foundation","keywords":"Irish; Clientelism; Politics; Dimension (graph theory); Representation (politics); Proportional representation; Political science; Law; Linguistics","score_opus":0.024194409625848196,"score_gpt":0.30571282092678065,"score_spread":0.28151841130093247,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1510410500","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.97685164,0.00022010665,0.0006884518,0.0003891144,0.000014898551,0.00003671265,0.0005692631,0.000019866582,0.021210037],"genre_scores_gemma":[0.9955955,0.00007507769,0.0005312711,0.00012564374,0.000020684323,0.000049067745,0.0008752482,0.000022200715,0.002705327],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99591017,0.001867879,0.0003234793,0.00064764876,0.0008093217,0.00044151975],"domain_scores_gemma":[0.9649414,0.020508647,0.006360112,0.0022906864,0.004760375,0.0011387012],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005276524,0.00015590801,0.00032081598,0.0032777789,0.0015542256,0.0021911166,0.0006381892,0.00077854283,0.004242],"category_scores_gemma":[0.025247611,0.0002865911,0.00021924108,0.0030232035,0.0027106146,0.0021957345,0.002761982,0.0014069834,0.001352056],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00085046917,0.0003177588,0.75945425,0.00050848385,0.000059993432,0.00069028046,0.0912667,0.000582805,0.008199391,0.0044185533,0.004764171,0.12888709],"study_design_scores_gemma":[0.000016072468,0.00010306584,0.9359201,0.00010161349,0.000020420304,0.00038517866,0.04142487,0.0011597052,0.0022555832,0.000995376,0.0175774,0.000040587973],"about_ca_topic_score_codex":0.020340243,"about_ca_topic_score_gemma":0.043648496,"teacher_disagreement_score":0.020340243,"about_ca_system_score_codex":0.0018918107,"about_ca_system_score_gemma":0.0010128254,"threshold_uncertainty_score":0.04044366},"labels":[],"label_agreement":null},{"id":"W1510578025","doi":"10.1007/3-540-45153-6_32","title":"The Design and Implementation of an Electronic Lexical Knowledge Base","year":2001,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":26,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Knowledge base; Base (topology); Natural language processing; Artificial intelligence; Programming language; Software engineering","score_opus":0.015804501335173363,"score_gpt":0.3045251686871657,"score_spread":0.28872066735199237,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1510578025","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012364979,0.00028334506,0.9698038,0.00037351358,0.00013196084,0.00043661892,0.0003875101,0.009866408,0.0063517927],"genre_scores_gemma":[0.079189695,0.00030009384,0.9117735,0.00028604994,0.000042452783,0.00039266245,0.0011846529,0.0006060057,0.006224889],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99879587,0.00023609892,0.00021502773,0.00024302083,0.00039178136,0.00011827737],"domain_scores_gemma":[0.99745804,0.0011116241,0.00011626561,0.0006079054,0.0005718762,0.00013430457],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025273303,0.00035657288,0.0007841886,0.0017654444,0.0012223485,0.0053431676,0.0045827935,0.0017254838,0.008313683],"category_scores_gemma":[0.008375715,0.0012305949,0.00060582854,0.0017514344,0.001196994,0.0055108,0.0027433399,0.0013666787,0.0036295594],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00071194314,0.00053227274,0.0018766974,0.00083896285,0.00015673359,0.0009337099,0.0012217867,0.019973163,0.03902261,0.17392355,0.01815612,0.7426526],"study_design_scores_gemma":[0.0005502239,0.0004676615,0.0015410027,0.00049210025,0.00075052964,0.0013039887,0.001000982,0.43649822,0.1600093,0.16593122,0.23124625,0.00020857312],"about_ca_topic_score_codex":0.0036519624,"about_ca_topic_score_gemma":0.00524901,"teacher_disagreement_score":0.008313683,"about_ca_system_score_codex":0.0008356188,"about_ca_system_score_gemma":0.002144551,"threshold_uncertainty_score":0.027812064},"labels":[],"label_agreement":null},{"id":"W1512617802","doi":"10.1007/3-540-44399-1_34","title":"Summary Generation and Evaluation in SumUM","year":2000,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Automatic summarization; Computer science; Natural language processing; Task (project management); Categorization; Information retrieval; Text generation; Artificial intelligence; Quality (philosophy); Text categorization; Systems engineering","score_opus":0.022935132657252628,"score_gpt":0.28130164934070984,"score_spread":0.2583665166834572,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1512617802","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05565504,0.0046458184,0.784592,0.0018041129,0.0012781724,0.0011823884,0.011776547,0.09117818,0.04788776],"genre_scores_gemma":[0.35152826,0.000959369,0.58329165,0.0005170684,0.00039357503,0.00078524265,0.019115046,0.00599453,0.037415307],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9945868,0.002472018,0.000624055,0.00065273285,0.0014359596,0.00022841756],"domain_scores_gemma":[0.99118644,0.004518779,0.00037427247,0.0016943053,0.0019474871,0.00027866242],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0047905724,0.0012330689,0.0016880906,0.0031819225,0.0008728012,0.003789738,0.0021058808,0.001092551,0.035030603],"category_scores_gemma":[0.020830054,0.00067557977,0.0009799261,0.0020769285,0.00056249136,0.0044836313,0.0032085108,0.0008351014,0.011293722],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014749027,0.00019010007,0.0017624486,0.00083334785,0.00011311722,0.00020891812,0.00044195948,0.008258572,0.0057647107,0.020690626,0.08443163,0.87582964],"study_design_scores_gemma":[0.00089970994,0.00095855317,0.005380002,0.00053187454,0.00059064856,0.00080124,0.001025702,0.5261112,0.09617414,0.12561254,0.24171932,0.00019511451],"about_ca_topic_score_codex":0.0025625632,"about_ca_topic_score_gemma":0.0043370086,"teacher_disagreement_score":0.035030603,"about_ca_system_score_codex":0.0011456868,"about_ca_system_score_gemma":0.0013118553,"threshold_uncertainty_score":0.11718905},"labels":[],"label_agreement":null},{"id":"W1512661636","doi":"10.1007/978-3-642-20095-3_23","title":"Unsupervised and Open Ontology-Based Semantic Analysis","year":2011,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; Simon Fraser University","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Parsing; Grammar; Ontology; Grammar induction; Inference; Rule-based machine translation; Linguistics","score_opus":0.023439337734706356,"score_gpt":0.2741637432299051,"score_spread":0.25072440549519875,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1512661636","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00780361,0.00025970142,0.9837473,0.00022761915,0.00005978792,0.00009583969,0.0012554557,0.0016504236,0.004900251],"genre_scores_gemma":[0.13933101,0.00067221565,0.84314865,0.00015769042,0.00007124168,0.00028252538,0.009115839,0.0007168211,0.006504107],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99689543,0.00060726854,0.000284813,0.0007166107,0.0012971436,0.00019871058],"domain_scores_gemma":[0.9970471,0.0011465257,0.00017731955,0.0007721747,0.0007616284,0.00009522948],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018475782,0.0007035761,0.00085108745,0.0047730203,0.0016511935,0.0039581438,0.0019864275,0.0008601877,0.0035412752],"category_scores_gemma":[0.0061054737,0.00048157532,0.002413556,0.005229164,0.0015828371,0.007001455,0.0042113145,0.0021938519,0.0018434454],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021595381,0.0004407687,0.003498964,0.0005544942,0.00029383475,0.00031444296,0.0013258025,0.017016035,0.017446421,0.26560286,0.017821023,0.67546946],"study_design_scores_gemma":[0.00002602135,0.000046689966,0.0032522073,0.00020117706,0.00017896421,0.000493106,0.0010549282,0.30035138,0.021903275,0.6099902,0.06242461,0.000077474324],"about_ca_topic_score_codex":0.0054391613,"about_ca_topic_score_gemma":0.010463312,"teacher_disagreement_score":0.0054391613,"about_ca_system_score_codex":0.001584286,"about_ca_system_score_gemma":0.0034236282,"threshold_uncertainty_score":0.011846721},"labels":[],"label_agreement":null},{"id":"W1513008738","doi":"10.1007/978-3-642-01818-3_15","title":"Training Global Linear Models for Chinese Word Segmentation","year":2009,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Margin (machine learning); Segmentation; Word (group theory); Sentence; Artificial intelligence; Perceptron; Pattern recognition (psychology); Text segmentation; Linear model; Natural language processing; Speech recognition; Artificial neural network; Machine learning; Mathematics","score_opus":0.026751314275890333,"score_gpt":0.3085997165788049,"score_spread":0.2818484023029146,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1513008738","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1073308,0.0051829373,0.8550294,0.00078752986,0.00056365,0.00017158144,0.0022414352,0.024123281,0.004569479],"genre_scores_gemma":[0.53311133,0.0018257668,0.41674113,0.0008678692,0.0005053816,0.00063155504,0.016729124,0.002863343,0.026724473],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99928385,0.00017971332,0.00004559826,0.00031234956,0.000058548205,0.00012001632],"domain_scores_gemma":[0.99809974,0.0013376386,0.00006159583,0.00017957947,0.00023771047,0.00008368226],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013419994,0.0030840966,0.0021230732,0.0015765182,0.0009597551,0.0013536692,0.0021573931,0.0022072627,0.009756326],"category_scores_gemma":[0.0029550423,0.0015348976,0.0018255458,0.0025782995,0.0007066163,0.0025782688,0.0017750129,0.0035877593,0.0056452774],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008101602,0.00020838546,0.0015485841,0.00026818571,0.0002390915,0.0001886763,0.00016197671,0.18370166,0.00873025,0.002625253,0.020480424,0.78103733],"study_design_scores_gemma":[0.00003128276,0.00007170111,0.00033119973,0.000012318454,0.000046337227,0.000023134622,0.000050220693,0.99271095,0.0025780692,0.0030543094,0.0010773697,0.000013119694],"about_ca_topic_score_codex":0.02004873,"about_ca_topic_score_gemma":0.03426552,"teacher_disagreement_score":0.02004873,"about_ca_system_score_codex":0.0011566635,"about_ca_system_score_gemma":0.0018438192,"threshold_uncertainty_score":0.039864063},"labels":[],"label_agreement":null},{"id":"W1515253087","doi":"","title":"Using Statistical Word Associations for the Retrieval of Strongly-Textual Cases.","year":2003,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Natural language processing; Word (group theory); Similarity (geometry); Artificial intelligence; Process (computing); Information retrieval; Translation (biology); Machine translation; Statistical model; Statistical analysis; Linguistics; Mathematics; Statistics","score_opus":0.05522528577050666,"score_gpt":0.3498113241993679,"score_spread":0.2945860384288612,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1515253087","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0657221,0.001065278,0.92395157,0.0006925902,0.000080141115,0.00037558182,0.0005274459,0.0031103361,0.004474852],"genre_scores_gemma":[0.46235305,0.00069677975,0.5324028,0.00019125552,0.00010963771,0.00042326708,0.0016708551,0.00024786513,0.0019044885],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9955259,0.0019144039,0.00032808064,0.000486395,0.0016147508,0.00013045441],"domain_scores_gemma":[0.9777518,0.016483642,0.002543065,0.0017490945,0.0011847941,0.00028764587],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0045910035,0.00054908387,0.0006094466,0.008880987,0.0009865932,0.0024648295,0.0012436426,0.0010478068,0.003403772],"category_scores_gemma":[0.049168807,0.00044455746,0.0010340263,0.007512554,0.0012136133,0.006436995,0.0014154857,0.000915329,0.0028806578],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000580619,0.0005797525,0.022421949,0.0012660254,0.00034915865,0.00084580446,0.0024043608,0.03164973,0.029360577,0.06412228,0.010098583,0.8363211],"study_design_scores_gemma":[0.00014607335,0.0005335793,0.011542494,0.00023845078,0.00026923275,0.0032475933,0.001140801,0.8071817,0.022222986,0.12610912,0.027182149,0.0001857551],"about_ca_topic_score_codex":0.0027745373,"about_ca_topic_score_gemma":0.00471826,"teacher_disagreement_score":0.008880987,"about_ca_system_score_codex":0.0010396255,"about_ca_system_score_gemma":0.0015011245,"threshold_uncertainty_score":0.024279833},"labels":[],"label_agreement":null},{"id":"W1515435652","doi":"10.1163/9789004333901_011","title":"Proper Name Extraction from Non-Journalistic Texts","year":2001,"lang":"en","type":"book-chapter","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":84,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Object (grammar); Linguistics; Verb; Computer science; Natural language processing; Philosophy","score_opus":0.017714699480991452,"score_gpt":0.27036570117265835,"score_spread":0.2526510016916669,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1515435652","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4890362,0.020324087,0.30092087,0.0042625335,0.0019123626,0.0006671025,0.04160101,0.007364776,0.13391116],"genre_scores_gemma":[0.62349826,0.01049464,0.21323799,0.0004986546,0.00088855217,0.0003994294,0.09805177,0.002289091,0.050641663],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99868184,0.00034239786,0.00017808196,0.00026919338,0.00046990666,0.000058602956],"domain_scores_gemma":[0.9915222,0.0060315356,0.0005705277,0.0006034386,0.0011665011,0.000105744235],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012435658,0.00054279563,0.0007329377,0.0048369095,0.001217734,0.0024172405,0.00075885805,0.00064365065,0.009317295],"category_scores_gemma":[0.007105391,0.00046162723,0.00030849184,0.0081847,0.00065633183,0.0040744133,0.0011678081,0.00079696265,0.0043445285],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005939205,0.00022328101,0.00721897,0.0064692744,0.00009376713,0.0026921825,0.010641296,0.0011162085,0.10693963,0.033110023,0.05077446,0.7801271],"study_design_scores_gemma":[0.000091808935,0.00012797883,0.033492558,0.00041616475,0.00017507687,0.003020757,0.005041442,0.018592888,0.12717499,0.017578444,0.7941539,0.00013409535],"about_ca_topic_score_codex":0.0011719558,"about_ca_topic_score_gemma":0.0034166798,"teacher_disagreement_score":0.009317295,"about_ca_system_score_codex":0.0008748569,"about_ca_system_score_gemma":0.0010816396,"threshold_uncertainty_score":0.031169474},"labels":[],"label_agreement":null},{"id":"W1515576489","doi":"","title":"Word-for-word glossing with contextually similar words","year":2000,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Word (group theory); Machine translation; Set (abstract data type); Speech recognition; Linguistics","score_opus":0.011895499964433462,"score_gpt":0.2619694554042695,"score_spread":0.25007395543983607,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1515576489","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01609783,0.0010975937,0.9727015,0.00013955137,0.00016142518,0.00039781074,0.00054499705,0.005022609,0.0038366315],"genre_scores_gemma":[0.05729706,0.00068674586,0.93633467,0.00011811231,0.000103261555,0.00025616097,0.0020719,0.0011467125,0.0019854703],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99836224,0.00041094655,0.00023682791,0.00050966354,0.00041154266,0.00006876899],"domain_scores_gemma":[0.99642086,0.0010874113,0.0003582306,0.0012787171,0.00077894976,0.00007584737],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021083602,0.0015089709,0.0015436691,0.0033045919,0.0014382189,0.002247537,0.001320487,0.00095422607,0.0074837105],"category_scores_gemma":[0.010563808,0.0006898104,0.0010697178,0.0038625982,0.0010642766,0.004119625,0.0028877615,0.0014165094,0.0052459035],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023883219,0.00012554799,0.0020188526,0.00072440784,0.00017737588,0.00032130477,0.00057005964,0.018121472,0.043219984,0.018192144,0.012303359,0.9039867],"study_design_scores_gemma":[0.00026724074,0.0006273056,0.007184368,0.0005426662,0.00065707404,0.0026189757,0.0015823384,0.56693727,0.1580414,0.11359895,0.14760976,0.00033271275],"about_ca_topic_score_codex":0.002177648,"about_ca_topic_score_gemma":0.0045731193,"teacher_disagreement_score":0.0074837105,"about_ca_system_score_codex":0.0004420733,"about_ca_system_score_gemma":0.001151013,"threshold_uncertainty_score":0.02503556},"labels":[],"label_agreement":null},{"id":"W151579683","doi":"","title":"Framework for Abstractive Summarization using Text-to-Text Generation","year":2011,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":123,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Automatic summarization; Computer science; Text graph; Natural language processing; Sentence; Multi-document summarization; Representation (politics); Domain (mathematical analysis); Information retrieval; Artificial intelligence; Source text; Text generation; Element (criminal law); Mathematics","score_opus":0.08061998104264394,"score_gpt":0.3292945165487601,"score_spread":0.24867453550611612,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W151579683","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.000597764,0.00019098744,0.9958527,0.00015982118,0.00005530177,0.0001351326,0.00016700661,0.0020430537,0.00079819886],"genre_scores_gemma":[0.022156231,0.0002779944,0.9732537,0.0001429844,0.00013583088,0.00038013587,0.0011475376,0.00034468606,0.0021608053],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.998002,0.0007170361,0.0002238651,0.00041068232,0.0005640859,0.0000823443],"domain_scores_gemma":[0.9970988,0.0011273653,0.00030343607,0.00062445767,0.0007379192,0.00010803005],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002759442,0.0015419114,0.0011832329,0.0029068193,0.0010274696,0.0028064009,0.0027031822,0.0013098607,0.0049292184],"category_scores_gemma":[0.005926219,0.00049485674,0.0014449946,0.0019160404,0.0012458587,0.0027425208,0.0021446254,0.002114701,0.0036017133],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021723486,0.00027940233,0.0007077753,0.0011580043,0.00023925754,0.000626159,0.0013819045,0.050445944,0.04198208,0.14731376,0.03210335,0.72354513],"study_design_scores_gemma":[0.00016512857,0.00035090756,0.0005865188,0.00018779289,0.0002642827,0.0006342162,0.00031610954,0.62281835,0.037810475,0.20173228,0.13496917,0.0001648301],"about_ca_topic_score_codex":0.0019548936,"about_ca_topic_score_gemma":0.0025356913,"teacher_disagreement_score":0.0049292184,"about_ca_system_score_codex":0.00084175635,"about_ca_system_score_gemma":0.0013378488,"threshold_uncertainty_score":0.016489863},"labels":[],"label_agreement":null},{"id":"W1516034496","doi":"10.1007/978-3-642-01818-3_6","title":"Enhancing the Bilingual Concordancer TransSearch with Word-Level Alignment","year":2009,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Machine translation; Word (group theory); Natural language processing; Sentence; Translation (biology); Artificial intelligence; Bilingual dictionary; Linguistics","score_opus":0.021117568354432677,"score_gpt":0.28170173870557486,"score_spread":0.2605841703511422,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1516034496","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.057162005,0.0013018656,0.89910805,0.00040020188,0.00059793936,0.0002708042,0.0012258895,0.018027583,0.021905635],"genre_scores_gemma":[0.1737221,0.0006726464,0.797176,0.00035032514,0.00024126084,0.0001861443,0.005546333,0.0046294434,0.017475635],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99627614,0.0012886026,0.0003338566,0.0008928154,0.00095346826,0.00025518896],"domain_scores_gemma":[0.9934696,0.0021723832,0.0002496998,0.001448069,0.00244261,0.00021767267],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002876333,0.0014011352,0.00197247,0.0037281767,0.0012612201,0.0033085933,0.0022832963,0.0018314308,0.020156551],"category_scores_gemma":[0.009553388,0.0008443953,0.0009930019,0.0046717036,0.0005267808,0.004477417,0.0037437913,0.0016081142,0.022752842],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00070684357,0.00035172963,0.001877213,0.00078931666,0.00016823193,0.0007203951,0.0007102167,0.0048862034,0.10474184,0.013548015,0.025465248,0.84603477],"study_design_scores_gemma":[0.00035470855,0.00071572646,0.003890627,0.00022874575,0.00050764444,0.0042837076,0.00173373,0.51078165,0.31500778,0.05217656,0.11001052,0.00030854286],"about_ca_topic_score_codex":0.00198288,"about_ca_topic_score_gemma":0.004856115,"teacher_disagreement_score":0.020156551,"about_ca_system_score_codex":0.00044929143,"about_ca_system_score_gemma":0.0018159454,"threshold_uncertainty_score":0.06743044},"labels":[],"label_agreement":null},{"id":"W1517708896","doi":"10.1109/nlpke.2005.1598738","title":"Improved Estimation for Unsupervised Part-of-Speech Tagging","year":2006,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Hidden Markov model; Computer science; Word (group theory); Speech recognition; Artificial intelligence; Simple (philosophy); Maximum-entropy Markov model; Pattern recognition (psychology); Markov model; Natural language processing; Machine learning; Markov chain; Mathematics; Variable-order Markov model","score_opus":0.011916864409370547,"score_gpt":0.2657707391658634,"score_spread":0.25385387475649285,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1517708896","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004852135,0.00004875906,0.99331045,0.00003358558,0.00002654336,0.000011373146,0.00005423864,0.0013093465,0.00035361055],"genre_scores_gemma":[0.18657362,0.00017228078,0.8078691,0.0001419483,0.00007935363,0.00013229132,0.0010517382,0.00077816204,0.003201479],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99850816,0.0006126124,0.00007106485,0.00038129368,0.0003183852,0.000108443885],"domain_scores_gemma":[0.9953687,0.0025674088,0.00021128597,0.0011187542,0.0006626061,0.00007120997],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019511108,0.001107533,0.0010564196,0.000579176,0.0005116274,0.0009957985,0.0016447119,0.0011770681,0.0020744987],"category_scores_gemma":[0.009100508,0.00063025963,0.0007246341,0.0009166822,0.00060227717,0.002042552,0.0013310639,0.001514697,0.004136621],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023075091,0.00021489056,0.0038954653,0.000155243,0.00015067993,0.00019701515,0.00031067873,0.48600745,0.044075754,0.018972019,0.0064228824,0.43936723],"study_design_scores_gemma":[0.0000057337693,0.000015685117,0.00043462386,0.000005531097,0.00001248384,0.00005339277,0.000010038626,0.9822037,0.008501667,0.00709017,0.0016473194,0.00001962643],"about_ca_topic_score_codex":0.0035873486,"about_ca_topic_score_gemma":0.006664454,"teacher_disagreement_score":0.0035873486,"about_ca_system_score_codex":0.00046437304,"about_ca_system_score_gemma":0.0008953345,"threshold_uncertainty_score":0.010318577},"labels":[],"label_agreement":null},{"id":"W1518500625","doi":"10.5772/5192","title":"\"From Saying to Doing\" - Natural Language Interaction with Artificial Agents and Robots","year":2007,"lang":"en","type":"book-chapter","venue":"Human-Robot Interaction","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Natural (archaeology); Robot; Computer science; Human–computer interaction; Communication; Artificial intelligence; Psychology; History; Archaeology","score_opus":0.054164945437780054,"score_gpt":0.35452485545454854,"score_spread":0.3003599100167685,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1518500625","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008313402,0.00442725,0.88548857,0.00882027,0.000759905,0.00019801653,0.00026262112,0.0016423803,0.09008755],"genre_scores_gemma":[0.34134835,0.0030983717,0.6133794,0.0038969337,0.00043048413,0.0007996137,0.0007745382,0.00068636093,0.035585873],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99167013,0.00584979,0.00038522974,0.0008361839,0.0009764673,0.00028211437],"domain_scores_gemma":[0.99665797,0.0021344984,0.00022095798,0.00043768503,0.00033321528,0.00021555602],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004959851,0.00097627315,0.00069754967,0.001706832,0.0018973706,0.0063882996,0.0018830316,0.0028447164,0.0070798364],"category_scores_gemma":[0.008547353,0.00065153546,0.0012631738,0.0013710159,0.010786668,0.012359152,0.005386471,0.0029350163,0.0016263415],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00002988399,0.000021847525,0.00024527256,0.0002499368,0.000024372186,0.0002663389,0.008989144,0.0013505878,0.0011997704,0.95202386,0.005442254,0.030156571],"study_design_scores_gemma":[0.00002230119,0.000042617095,0.00038026684,0.00030675065,0.000032216412,0.0005580112,0.003947607,0.0113817565,0.0014736513,0.7109591,0.27082717,0.00006860569],"about_ca_topic_score_codex":0.0040473817,"about_ca_topic_score_gemma":0.0024194608,"teacher_disagreement_score":0.0070798364,"about_ca_system_score_codex":0.001994367,"about_ca_system_score_gemma":0.0017141812,"threshold_uncertainty_score":0.026230514},"labels":[],"label_agreement":null},{"id":"W1518614801","doi":"","title":"Identification of Cognates and Recurrent Sound Correspondences in Word Lists","year":2009,"lang":"fr","type":"article","venue":"Trait. Autom. des Langues","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":34,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Word (group theory); Natural language processing; Computer science; Similarity (geometry); Artificial intelligence; Identification (biology); Recall; Cognate; Linguistics; Speech recognition","score_opus":0.028557800583898243,"score_gpt":0.31655776305123756,"score_spread":0.28799996246733933,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1518614801","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.52813256,0.0008171435,0.45326582,0.00026528098,0.00007970418,0.0002919598,0.0022358824,0.005850439,0.009061164],"genre_scores_gemma":[0.75018495,0.00030959872,0.2421008,0.00005462574,0.0000703918,0.0001279323,0.0035923412,0.00026153575,0.0032977911],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.998594,0.0002574674,0.00020556126,0.0005279325,0.00029819272,0.00011683521],"domain_scores_gemma":[0.99447614,0.0022989914,0.0009596906,0.000996942,0.0010948426,0.00017342926],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001071981,0.0005038293,0.0006254069,0.009206038,0.0012383614,0.0022546828,0.00093844545,0.0010841308,0.0039583202],"category_scores_gemma":[0.0082218,0.00046073983,0.0005348256,0.0032322726,0.00088268454,0.0044033374,0.0014754701,0.00058485445,0.0024685536],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00067880127,0.00027713002,0.05760707,0.000590859,0.00016962961,0.00083067326,0.0029558646,0.004502563,0.070733294,0.018102191,0.004507136,0.8390448],"study_design_scores_gemma":[0.00019834719,0.00085989037,0.14993793,0.0003743322,0.00070649496,0.0071375784,0.008993874,0.46360525,0.19276838,0.12510243,0.049849123,0.00046632648],"about_ca_topic_score_codex":0.002875328,"about_ca_topic_score_gemma":0.005124371,"teacher_disagreement_score":0.009206038,"about_ca_system_score_codex":0.0005344763,"about_ca_system_score_gemma":0.0008953072,"threshold_uncertainty_score":0.013241887},"labels":[],"label_agreement":null},{"id":"W1519642919","doi":"10.1007/978-3-540-30211-7_58","title":"A Nearest-Neighbor Method for Resolving PP-Attachment Ambiguity","year":2005,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":30,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Pointwise; Pointwise mutual information; Cosine similarity; k-nearest neighbors algorithm; Ambiguity; Similarity (geometry); Computer science; Phrase; Artificial intelligence; Task (project management); Trigonometric functions; Word (group theory); Similarity measure; Pattern recognition (psychology); Natural language processing; Algorithm; Mathematics; Mutual information; Image (mathematics); Engineering","score_opus":0.021031008482158116,"score_gpt":0.3217057140259904,"score_spread":0.3006747055438323,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1519642919","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008211653,0.00046610375,0.98217195,0.00013950453,0.00028033718,0.00013195143,0.00042446048,0.0030366867,0.005137422],"genre_scores_gemma":[0.066847675,0.00029154995,0.9186531,0.00014875414,0.00014411286,0.00011387476,0.0015490513,0.00067202456,0.011579888],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9981206,0.00031774,0.0001264009,0.0004462794,0.0008447027,0.00014437115],"domain_scores_gemma":[0.9977914,0.0005636287,0.000094007744,0.00061169476,0.0008633413,0.0000759911],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015406719,0.0009579175,0.0013334887,0.002634365,0.0018766408,0.0019398385,0.003114798,0.0020228627,0.013926234],"category_scores_gemma":[0.006155824,0.00070327404,0.0011994067,0.003351637,0.00059190916,0.0032146669,0.0021001927,0.0015626127,0.0073628454],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022068869,0.00015789595,0.00085440726,0.00014836784,0.00005795348,0.00018824657,0.0002114883,0.009751451,0.009175845,0.017811542,0.018575907,0.9428462],"study_design_scores_gemma":[0.00008389096,0.00009249919,0.0014762346,0.000060783364,0.00014727814,0.0008752204,0.00041348475,0.86134094,0.019641241,0.07146075,0.04429098,0.000116677074],"about_ca_topic_score_codex":0.0066668564,"about_ca_topic_score_gemma":0.013130506,"teacher_disagreement_score":0.013926234,"about_ca_system_score_codex":0.00055429817,"about_ca_system_score_gemma":0.0013894935,"threshold_uncertainty_score":0.046587884},"labels":[],"label_agreement":null},{"id":"W1520498874","doi":"10.1007/978-3-540-92673-3_12","title":"Ontology and the Lexicon","year":2009,"lang":"en","type":"book-chapter","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":174,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Lexicon; Ontology; Linguistics; Computer science; Domain (mathematical analysis); Object (grammar); Natural language processing; Artificial intelligence; Word (group theory); Limit (mathematics); Philosophy; Mathematics; Epistemology","score_opus":0.011559035183462112,"score_gpt":0.24397502902000892,"score_spread":0.23241599383654682,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1520498874","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004077213,0.045906488,0.18572292,0.014164642,0.001913177,0.00008394601,0.0007430913,0.00073921937,0.74664927],"genre_scores_gemma":[0.2316677,0.0536175,0.19296256,0.0064883046,0.0030870936,0.00046359096,0.004176079,0.0011946838,0.50634253],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9995253,0.00015444127,0.00003768578,0.000102903505,0.00014225407,0.00003739896],"domain_scores_gemma":[0.9996897,0.0001437103,0.000018884193,0.00007314603,0.000053169213,0.000021449447],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005475568,0.0006546439,0.0006145809,0.0019590259,0.0014016426,0.005376713,0.0009828265,0.001164372,0.012662541],"category_scores_gemma":[0.0018657607,0.0005541753,0.00056867074,0.00238124,0.0054574837,0.011105134,0.0018215637,0.002668289,0.005313278],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000057525094,0.000006656108,0.000037684396,0.00005964207,0.0000043406108,0.000032166783,0.00024779295,0.00017663273,0.00017533726,0.9380789,0.02042143,0.040753577],"study_design_scores_gemma":[0.0000027948802,0.0000022037202,0.00005352493,0.000053949876,0.000004428805,0.00008267854,0.000093729686,0.00047780242,0.00012573412,0.796276,0.20282236,0.0000048774036],"about_ca_topic_score_codex":0.004311299,"about_ca_topic_score_gemma":0.0049641384,"teacher_disagreement_score":0.012662541,"about_ca_system_score_codex":0.0021357413,"about_ca_system_score_gemma":0.0020520994,"threshold_uncertainty_score":0.042360365},"labels":[],"label_agreement":null},{"id":"W1520647843","doi":"10.1007/3-540-44886-1_48","title":"Not as Easy as It Seems: Automating the Construction of Lexical Chains Using Roget’s Thesaurus","year":2003,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":53,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Thesaurus; Computer science; Automatic summarization; Natural language processing; Information retrieval; Artificial intelligence; WordNet","score_opus":0.022232095732569082,"score_gpt":0.28414608764356314,"score_spread":0.26191399191099407,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1520647843","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04799782,0.001622873,0.91319627,0.0014016081,0.0004048758,0.00013427057,0.0006006724,0.01880571,0.015835892],"genre_scores_gemma":[0.08320861,0.0007977258,0.9012378,0.00034647874,0.00005800344,0.00008161469,0.0016755674,0.0025833673,0.010010811],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991873,0.00027960882,0.000084636384,0.00023166505,0.00016008374,0.0000566169],"domain_scores_gemma":[0.99737954,0.0011746113,0.00012107901,0.00078215206,0.00047037916,0.00007232799],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016012486,0.00082487107,0.0011337693,0.0022262067,0.0012815723,0.0037561345,0.001833572,0.0014737184,0.011750834],"category_scores_gemma":[0.006872654,0.0011378945,0.0011701275,0.0026668087,0.0011822883,0.009709723,0.002731937,0.0015008886,0.0113266045],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014688823,0.0000705238,0.0012846082,0.0005161706,0.00008657484,0.00037613927,0.002299301,0.0014863607,0.023164267,0.048777465,0.025614116,0.89617753],"study_design_scores_gemma":[0.00016326149,0.0002770682,0.0034370874,0.0007787433,0.00046188367,0.0035920271,0.005196965,0.07628109,0.09380662,0.2843126,0.5313358,0.0003568912],"about_ca_topic_score_codex":0.0029587355,"about_ca_topic_score_gemma":0.005161305,"teacher_disagreement_score":0.011750834,"about_ca_system_score_codex":0.0004473444,"about_ca_system_score_gemma":0.0013229714,"threshold_uncertainty_score":0.039310515},"labels":[],"label_agreement":null},{"id":"W1521154370","doi":"10.1016/s0020-7063(02)00127-9","title":"Significant differences in GAAP in Canada, Chile, Mexico, and the United States: An analysis of accounting pronouncements as of January 2001","year":2002,"lang":"en","type":"article","venue":"The International Journal of Accounting","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Accounting; Political science; Business","score_opus":0.017289464110396325,"score_gpt":0.25921410823414354,"score_spread":0.2419246441237472,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1521154370","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99274373,0.00026732776,0.00004110291,0.00040486112,0.0000118604985,0.000007768163,0.0020485749,0.000007824298,0.00446686],"genre_scores_gemma":[0.9967765,0.00020822008,0.00008046639,0.000043416265,0.000004082863,0.0000058867854,0.00118995,0.000007123948,0.0016845139],"study_design_codex":"observational","study_design_gemma":"qualitative","domain_scores_codex":[0.99956566,0.00005651882,0.000033838845,0.000048474223,0.0001400201,0.000155505],"domain_scores_gemma":[0.99582386,0.00093831413,0.0008192011,0.00010645022,0.0019937123,0.00031839684],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00069785345,0.00011229523,0.0002505668,0.0021704733,0.001826134,0.0016562766,0.0006240871,0.00036552796,0.002076069],"category_scores_gemma":[0.0064929524,0.00014430528,0.00012176543,0.00909826,0.0006161941,0.0006631615,0.00086205936,0.0007547019,0.00017504217],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002214383,0.000051277028,0.96010494,0.000073328745,0.00008568627,0.00040239378,0.012158618,0.00057131203,0.000477807,0.0028378626,0.00537965,0.017635684],"study_design_scores_gemma":[0.0000022041534,0.0000055413684,0.98713034,0.0000131838815,0.000013766058,0.000025472942,0.008852866,0.00021676898,0.00012057534,0.000110853696,0.003501802,0.00000669332],"about_ca_topic_score_codex":0.901738,"about_ca_topic_score_gemma":0.9581671,"teacher_disagreement_score":0.09826201,"about_ca_system_score_codex":0.0067387694,"about_ca_system_score_gemma":0.006695873,"threshold_uncertainty_score":0.19768137},"labels":[],"label_agreement":null},{"id":"W1521635597","doi":"","title":"The Implementation of Arabic Subject Markers in the LKB System","year":2009,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Computer science; Subject (documents); Syntax; Arabic; Natural language processing; Phrase; Artificial intelligence; Linguistics; Grammar; Interface (matter); Morphology (biology); Head (geology); Philosophy; Biology","score_opus":0.007590918212678164,"score_gpt":0.28928339353929694,"score_spread":0.2816924753266188,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1521635597","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08645344,0.0005140019,0.7226002,0.0014350096,0.0004929361,0.0005998921,0.002395367,0.12484801,0.060661085],"genre_scores_gemma":[0.43306577,0.0003155614,0.52905875,0.00093252823,0.00014873363,0.00048218196,0.0028887661,0.008511132,0.0245966],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99871624,0.00036705565,0.00014538899,0.00030699256,0.00034468208,0.000119734876],"domain_scores_gemma":[0.9983753,0.00034295002,0.00009915073,0.00054053497,0.0005505163,0.00009151964],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014148529,0.00069482543,0.0006790688,0.0010897705,0.000770346,0.003674556,0.0014702945,0.00091644685,0.017297842],"category_scores_gemma":[0.0043134857,0.00083558174,0.00039413932,0.00069939153,0.0011060765,0.002895664,0.0030148504,0.0010800868,0.022714263],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014698827,0.0002866281,0.008093477,0.0014230283,0.00008738013,0.0015055452,0.009094002,0.0052232076,0.16169414,0.13348998,0.06360306,0.6140297],"study_design_scores_gemma":[0.0004198405,0.00042315072,0.004409523,0.0004959729,0.00017406578,0.0017363993,0.0015464518,0.1247224,0.22380462,0.039787233,0.6021471,0.00033315743],"about_ca_topic_score_codex":0.0023742432,"about_ca_topic_score_gemma":0.0014295882,"teacher_disagreement_score":0.017297842,"about_ca_system_score_codex":0.0011730694,"about_ca_system_score_gemma":0.001291799,"threshold_uncertainty_score":0.05786699},"labels":[],"label_agreement":null},{"id":"W1524343434","doi":"10.5860/lrts.52n2.29","title":"Subject Access Tools in English for Canadian Topics","year":2008,"lang":"en","type":"article","venue":"Library Resources and Technical Services","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"Government of Canada","keywords":"Subject (documents); Cataloging; Subject access; Dewey Decimal Classification; Terminology; Library of congress; Computer science; Library science; Library of Congress Classification; Library classification; World Wide Web; Controlled vocabulary; Library catalog; Linguistics","score_opus":0.01748919794725137,"score_gpt":0.24736673005071264,"score_spread":0.22987753210346126,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1524343434","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018089548,0.005396687,0.19130746,0.018142093,0.0015768462,0.0015658139,0.042996384,0.03428146,0.6866438],"genre_scores_gemma":[0.11350692,0.0131614115,0.46546105,0.004672012,0.0009658359,0.0010994567,0.056870367,0.009622638,0.3346403],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9954913,0.00065663294,0.0006831306,0.00041512455,0.0020678947,0.0006860195],"domain_scores_gemma":[0.9769959,0.0040319106,0.00084906584,0.0033465296,0.013651425,0.0011252206],"candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.0065375804,0.00079172687,0.00071825157,0.014491671,0.007346522,0.007986997,0.0018879178,0.0009053162,0.056436267],"category_scores_gemma":[0.024923908,0.00064874254,0.0008729188,0.021978185,0.0026551208,0.008312279,0.0052414862,0.0015947341,0.02157451],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012268964,0.000044318356,0.0031653577,0.0009047745,0.000019482295,0.00044066898,0.011886662,0.0007155496,0.003937922,0.1743417,0.40550038,0.39892054],"study_design_scores_gemma":[0.000008635733,0.0000053196,0.0015338899,0.0002096101,0.000009779672,0.00008109987,0.0011840978,0.00024908248,0.0005789767,0.0035850096,0.99251443,0.000040028524],"about_ca_topic_score_codex":0.8454502,"about_ca_topic_score_gemma":0.90783876,"teacher_disagreement_score":0.992013,"about_ca_system_score_codex":0.026072033,"about_ca_system_score_gemma":0.07661798,"threshold_uncertainty_score":0.31091988},"labels":[],"label_agreement":null},{"id":"W1524444324","doi":"10.1007/978-3-642-01818-3_9","title":"Machine Translation of Legal Information and Its Evaluation","year":2009,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Machine translation; Certification; Translation (biology); Natural language processing; Context (archaeology); Machine translation system; Machine translation software usability; Artificial intelligence; Example-based machine translation; Law; Political science; History","score_opus":0.015865212163329694,"score_gpt":0.2749952949071796,"score_spread":0.2591300827438499,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1524444324","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.52029324,0.01437846,0.2870349,0.0033498243,0.0022027595,0.0020691005,0.02968916,0.04952566,0.09145685],"genre_scores_gemma":[0.71030253,0.002604277,0.19621989,0.00045501292,0.00044404806,0.00070852,0.069423184,0.0024329063,0.01740973],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99033713,0.0051907636,0.00073182903,0.00086303713,0.0024572467,0.00042001766],"domain_scores_gemma":[0.9791965,0.01270434,0.00046240023,0.0027729478,0.00440701,0.0004568708],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0061362945,0.0013225419,0.0015556434,0.005175395,0.0018058746,0.0033853354,0.0021437868,0.0027053142,0.01605479],"category_scores_gemma":[0.023842096,0.0004779133,0.0010013636,0.0048702797,0.0010618105,0.0036573107,0.0023467587,0.0014386907,0.007458564],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0040991777,0.0019298935,0.0055426448,0.0023221052,0.00051336543,0.0010392325,0.0007338208,0.06287394,0.01744688,0.020636538,0.09419616,0.7886663],"study_design_scores_gemma":[0.00094685284,0.0014798343,0.0067309444,0.000297976,0.00045902148,0.0012004619,0.0010594833,0.8467659,0.06879104,0.030787311,0.041338388,0.00014281305],"about_ca_topic_score_codex":0.0072332635,"about_ca_topic_score_gemma":0.006078971,"teacher_disagreement_score":0.01605479,"about_ca_system_score_codex":0.0015452683,"about_ca_system_score_gemma":0.0023068783,"threshold_uncertainty_score":0.053708613},"labels":[],"label_agreement":null},{"id":"W1526749895","doi":"10.1109/icassp.1995.479660","title":"Searching with a transcription graph","year":2002,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Institut National de la Recherche Scientifique","funders":"","keywords":"Computer science; Vocabulary; Transcription (linguistics); Graph; Lexicon; Artificial intelligence; Natural language processing; Theoretical computer science; Linguistics","score_opus":0.018251009031378457,"score_gpt":0.23935842710464364,"score_spread":0.22110741807326517,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1526749895","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01417676,0.0003117387,0.9769394,0.00086607627,0.00009536187,0.000057909776,0.0011604921,0.0024615258,0.0039307717],"genre_scores_gemma":[0.23667817,0.00089454994,0.7360682,0.0007203475,0.0001675391,0.00021403312,0.01195182,0.0013434563,0.011961849],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.998679,0.00046712128,0.00006494846,0.00045681506,0.00024433903,0.00008780423],"domain_scores_gemma":[0.9976718,0.0013513829,0.00012497528,0.0005179906,0.00025347242,0.00008044193],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006822596,0.0007153616,0.0011920705,0.0014261337,0.0010082566,0.0017893082,0.0014524672,0.001670759,0.011981447],"category_scores_gemma":[0.005922478,0.0006092324,0.0011592181,0.002018527,0.0010712893,0.004528593,0.0018860212,0.0012960908,0.00449188],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00047284274,0.00026775035,0.0019867017,0.0007624323,0.00011543342,0.0012749874,0.0007842457,0.14390644,0.028300557,0.18979728,0.056627717,0.57570356],"study_design_scores_gemma":[0.00008855896,0.0001240759,0.000415095,0.000068337904,0.0000763785,0.0005188188,0.00040749324,0.56513125,0.008759812,0.39661813,0.027717652,0.00007444416],"about_ca_topic_score_codex":0.0046176305,"about_ca_topic_score_gemma":0.005765626,"teacher_disagreement_score":0.011981447,"about_ca_system_score_codex":0.0006354143,"about_ca_system_score_gemma":0.0010474814,"threshold_uncertainty_score":0.040081978},"labels":[],"label_agreement":null},{"id":"W1528797807","doi":"10.1007/11799573_40","title":"Semantic Property Grammars for Knowledge Extraction from Biomedical Text","year":2006,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Rule-based machine translation; Parsing; Natural language processing; Property (philosophy); Artificial intelligence; Constraint (computer-aided design); Relationship extraction; L-attributed grammar; Information extraction; Context-free grammar; Mathematics","score_opus":0.01732528293889576,"score_gpt":0.2769781635403407,"score_spread":0.259652880601445,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1528797807","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00436767,0.00057255913,0.9823942,0.00037106857,0.00007049595,0.0002197735,0.0023835006,0.007049314,0.002571402],"genre_scores_gemma":[0.07798937,0.0011142015,0.90482295,0.00022522034,0.00007286642,0.0004927172,0.011264096,0.001322831,0.002695801],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99896634,0.0002906618,0.00018955093,0.00019679937,0.00030018142,0.00005643561],"domain_scores_gemma":[0.99715304,0.0019863609,0.00010665534,0.00044098313,0.0002754383,0.000037593967],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014189597,0.00070434186,0.0010449876,0.0020720174,0.00083634653,0.0019219778,0.0016315426,0.0009976351,0.0047954335],"category_scores_gemma":[0.005251194,0.000846304,0.0020341547,0.0024805842,0.0013696854,0.0037832449,0.0020309254,0.0017674075,0.0029002053],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001875678,0.0001454331,0.0010163165,0.0015545477,0.0001863598,0.0011712462,0.0012824823,0.030093053,0.011174255,0.30191055,0.034498047,0.6167801],"study_design_scores_gemma":[0.000072069015,0.000048576905,0.0005612481,0.0003349353,0.00016714254,0.0006500473,0.0003435315,0.16117613,0.017159225,0.74785227,0.071567684,0.000067025074],"about_ca_topic_score_codex":0.002345626,"about_ca_topic_score_gemma":0.003172063,"teacher_disagreement_score":0.0047954335,"about_ca_system_score_codex":0.00084615924,"about_ca_system_score_gemma":0.0019733375,"threshold_uncertainty_score":0.016042292},"labels":[],"label_agreement":null},{"id":"W1531113454","doi":"10.1007/978-3-642-01818-3_13","title":"Automatic Frame Extraction from Sentences","year":2009,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; FrameNet; Natural language processing; Artificial intelligence; Automatic summarization; Frame (networking); Dependency (UML); Task (project management); Novelty; Information extraction; Textual entailment; Parsing; Logical consequence","score_opus":0.0127621946886624,"score_gpt":0.2709686067026411,"score_spread":0.2582064120139787,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1531113454","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02931733,0.00398382,0.90096796,0.00068138615,0.0009909228,0.00058755634,0.01414251,0.030385548,0.018942993],"genre_scores_gemma":[0.110816486,0.0026962725,0.8178538,0.00025910407,0.00037859118,0.00048150762,0.047031187,0.0022137198,0.018269323],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99953294,0.00005972639,0.00004909118,0.00016996155,0.0001211218,0.00006716308],"domain_scores_gemma":[0.9994041,0.00020463769,0.000040295472,0.00008565025,0.00024012597,0.000025222012],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00039914632,0.0020484058,0.0014023274,0.0032791814,0.0011105647,0.0014383834,0.0011382402,0.00096237165,0.013502366],"category_scores_gemma":[0.001418839,0.0007629574,0.0013611544,0.0027549213,0.00036796345,0.0017749784,0.001265795,0.0010526909,0.011569803],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00033777076,0.00006347127,0.0004518271,0.0008759524,0.00006614315,0.00061994046,0.0003108155,0.0010683151,0.10767728,0.008105057,0.055058062,0.82536525],"study_design_scores_gemma":[0.00018714713,0.00038568265,0.007353729,0.0007408091,0.0010325055,0.0026133342,0.0011486269,0.15384889,0.3980651,0.05245175,0.3819506,0.00022182622],"about_ca_topic_score_codex":0.0037531557,"about_ca_topic_score_gemma":0.0058021634,"teacher_disagreement_score":0.013502366,"about_ca_system_score_codex":0.0006095977,"about_ca_system_score_gemma":0.0011223637,"threshold_uncertainty_score":0.04516989},"labels":[],"label_agreement":null},{"id":"W1531868668","doi":"10.1007/978-3-540-73351-5_2","title":"An Efficient Denotational Semantics for Natural Language Database Queries","year":2007,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Windsor","funders":"","keywords":"Denotational semantics; Computer science; Programming language; Action semantics; Negation; Semantics (computer science); Operational semantics; Well-founded semantics; Computational semantics; Denotational semantics of the Actor model; Natural language processing; Artificial intelligence","score_opus":0.016433020641701566,"score_gpt":0.30543840489623825,"score_spread":0.2890053842545367,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1531868668","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008973018,0.0002344716,0.9830341,0.00042348803,0.0001237842,0.000121271725,0.00032636436,0.0031500626,0.003613501],"genre_scores_gemma":[0.22222103,0.0003571872,0.7688282,0.00039098153,0.00021287847,0.0004073478,0.0010736055,0.0011815112,0.005327213],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99514574,0.001121305,0.00067898736,0.000705049,0.0019075461,0.0004412452],"domain_scores_gemma":[0.9951348,0.0022781389,0.00021120631,0.0013617239,0.00087530137,0.00013900924],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003006186,0.0009841655,0.0017264549,0.0016792942,0.001608167,0.006223524,0.0028854897,0.0015082038,0.006498168],"category_scores_gemma":[0.008258828,0.0014718658,0.0019109952,0.002499885,0.003213715,0.012474759,0.0051109716,0.0031428824,0.0019724474],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029136514,0.00017484366,0.0004218402,0.00031167408,0.000042307165,0.00016284948,0.0008249025,0.0100285085,0.008513089,0.8581468,0.0075896597,0.11349209],"study_design_scores_gemma":[0.00012008058,0.00009575813,0.00020559448,0.00007103114,0.00011245316,0.00028654424,0.00022261991,0.13434222,0.0111507,0.8229519,0.03035806,0.0000830817],"about_ca_topic_score_codex":0.0021380913,"about_ca_topic_score_gemma":0.002516081,"teacher_disagreement_score":0.006498168,"about_ca_system_score_codex":0.0021673485,"about_ca_system_score_gemma":0.0020567332,"threshold_uncertainty_score":0.02173853},"labels":[],"label_agreement":null},{"id":"W1532156013","doi":"10.1007/978-3-642-12116-6_24","title":"Lexical Chains Using Distributional Measures of Concept Distance","year":2010,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence","score_opus":0.028210738497016326,"score_gpt":0.28011033873276103,"score_spread":0.2518996002357447,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1532156013","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.057068963,0.0010030327,0.93035996,0.00020747149,0.00014809766,0.000115839575,0.0006152033,0.0007771012,0.00970434],"genre_scores_gemma":[0.48861971,0.001319392,0.49589637,0.00009182546,0.0003266657,0.0003910512,0.003091498,0.0005831815,0.009680286],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99737394,0.0008155452,0.0003303533,0.00053129456,0.00080849126,0.00014032371],"domain_scores_gemma":[0.9882776,0.00794912,0.000677531,0.0012446594,0.0015051522,0.00034587164],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022932745,0.0006031616,0.0009737279,0.0075361235,0.0013610098,0.0045958767,0.0011441256,0.0011142333,0.010516798],"category_scores_gemma":[0.01541067,0.0007575077,0.0008219735,0.008054776,0.0012948595,0.012253932,0.0031633347,0.0013258493,0.0025165346],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004518001,0.00021609696,0.005905391,0.0004552626,0.00014241245,0.00028646522,0.001005367,0.020533059,0.010307827,0.45947814,0.004787228,0.49643096],"study_design_scores_gemma":[0.000052519405,0.00013354569,0.0018186356,0.00017848093,0.00007672783,0.00032144896,0.00054784934,0.23469058,0.0043809144,0.7474202,0.010309883,0.00006908643],"about_ca_topic_score_codex":0.0010586719,"about_ca_topic_score_gemma":0.0018517007,"teacher_disagreement_score":0.010516798,"about_ca_system_score_codex":0.00067491404,"about_ca_system_score_gemma":0.00082303234,"threshold_uncertainty_score":0.035182178},"labels":[],"label_agreement":null},{"id":"W1533819011","doi":"10.1007/11736790_19","title":"Partial Predicate Argument Structure Matching for Entailment Determination","year":2006,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Textual entailment; WordNet; Logical consequence; Computer science; Predicate (mathematical logic); Natural language processing; Argument (complex analysis); Artificial intelligence; Programming language","score_opus":0.010378289575054097,"score_gpt":0.26317349269816404,"score_spread":0.25279520312310994,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1533819011","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008754089,0.00040044656,0.9742628,0.00028210963,0.00010329612,0.00018997525,0.00058451993,0.0047622332,0.010660545],"genre_scores_gemma":[0.16651233,0.00037685974,0.81895274,0.00022522669,0.00015544619,0.00023418484,0.002764925,0.0017711472,0.009007251],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9963791,0.0009561179,0.00034799532,0.000798789,0.0012081593,0.00030980346],"domain_scores_gemma":[0.99545515,0.0021510085,0.00015364903,0.0015589618,0.0006147632,0.0000663312],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027508435,0.0007564058,0.0016537255,0.002457605,0.0014207357,0.0030371693,0.0032920141,0.0022415442,0.02298578],"category_scores_gemma":[0.010116908,0.0014667581,0.0022208486,0.0033992734,0.001442153,0.009981468,0.003640315,0.0026146115,0.007337552],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00043758584,0.000234374,0.0005961747,0.0006142461,0.00011983577,0.00023658907,0.00046489976,0.007366052,0.011177275,0.39826107,0.020015297,0.56047654],"study_design_scores_gemma":[0.000073703195,0.000073663,0.0003469542,0.00013719707,0.00014391998,0.0003018821,0.0001460855,0.14296305,0.028513126,0.7965069,0.030730814,0.000062739164],"about_ca_topic_score_codex":0.0014705712,"about_ca_topic_score_gemma":0.0020086286,"teacher_disagreement_score":0.02298578,"about_ca_system_score_codex":0.0011734617,"about_ca_system_score_gemma":0.0015232111,"threshold_uncertainty_score":0.07689512},"labels":[],"label_agreement":null},{"id":"W1535378484","doi":"","title":"Xarxa Panllatina de Terminologia (Realiter) (2012). Lèxic panllatí de l'energia eòlica [recurs en línia]. Gatineaux: Bureau de la Traduction du Gouvernement du Canada, 505 p.","year":2012,"lang":"ca","type":"article","venue":"Estudis Romànics (Institut d'Estudis Catalans)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Humanities; Political science; Philosophy","score_opus":0.008465757921344751,"score_gpt":0.2552731767169326,"score_spread":0.24680741879558787,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1535378484","genre_codex":"review","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0036504657,0.49812758,0.11057116,0.029822478,0.010752143,0.00023810656,0.009714048,0.008227939,0.32889605],"genre_scores_gemma":[0.045942076,0.28017372,0.20690206,0.0071457527,0.009537525,0.0009893165,0.033282273,0.009339184,0.40668806],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99771225,0.000759841,0.0003020459,0.00044661184,0.0006371224,0.00014208394],"domain_scores_gemma":[0.9975356,0.0009592001,0.00023552614,0.00032944247,0.0007642408,0.00017591173],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0040406156,0.0011666685,0.0010954634,0.00453425,0.0018636537,0.007826779,0.0013076414,0.002129823,0.043352064],"category_scores_gemma":[0.004341491,0.0010241434,0.00065472344,0.005984274,0.0018727838,0.009434583,0.0028904204,0.004257422,0.03400587],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000120616685,0.00003648846,0.00050352275,0.0012583766,0.000016229116,0.000095045834,0.0015361625,0.00022361903,0.0014153263,0.1333629,0.53864324,0.3227884],"study_design_scores_gemma":[0.000008998809,0.000006447543,0.0006520395,0.00039398743,0.000005953672,0.00011963883,0.00019372739,0.000115381656,0.00038300746,0.0059685954,0.99214363,0.000008546773],"about_ca_topic_score_codex":0.013590183,"about_ca_topic_score_gemma":0.014277626,"teacher_disagreement_score":0.043352064,"about_ca_system_score_codex":0.003536229,"about_ca_system_score_gemma":0.0042527593,"threshold_uncertainty_score":0.1450271},"labels":[],"label_agreement":null},{"id":"W1535384510","doi":"10.16995/dscn.138","title":"Corpus Linguistics beyond Google: the WebCorp Linguist’s Search Engine","year":2009,"lang":"en","type":"article","venue":"Digital Studies / Le champ numérique","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Disk formatting; Computer science; World Wide Web; Information retrieval; Search engine; Web standards; Web search engine; Web page; Web search query","score_opus":0.01986126175897783,"score_gpt":0.2839510981054043,"score_spread":0.26408983634642647,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1535384510","genre_codex":"other","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020417657,0.023905613,0.0881675,0.35277718,0.008232077,0.00041521998,0.03748738,0.03296776,0.43562958],"genre_scores_gemma":[0.26819295,0.021745961,0.16816844,0.03199978,0.008036609,0.00086254923,0.060228143,0.030426443,0.41033912],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9970957,0.0010476448,0.00023652815,0.00025055362,0.0011938924,0.00017571932],"domain_scores_gemma":[0.9808476,0.0071282843,0.00036967828,0.003462536,0.006215891,0.0019759499],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004661338,0.00041019107,0.0007961015,0.0048739347,0.002291593,0.0068048076,0.0016653436,0.0021149525,0.04202892],"category_scores_gemma":[0.032785628,0.0006692822,0.00040830483,0.0053504654,0.0015513225,0.015010051,0.0031019358,0.0019458613,0.030556055],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015359576,0.00004058263,0.001205395,0.0004214482,0.000032041164,0.00018502676,0.00085457653,0.00016918671,0.0014985118,0.045401577,0.78040135,0.16963671],"study_design_scores_gemma":[0.000025242853,0.000013732457,0.000944179,0.00030822612,0.000020478908,0.00018156975,0.0005244451,0.001641367,0.0016845252,0.022813013,0.97180444,0.000038921153],"about_ca_topic_score_codex":0.022334255,"about_ca_topic_score_gemma":0.055702563,"teacher_disagreement_score":0.04202892,"about_ca_system_score_codex":0.0023969337,"about_ca_system_score_gemma":0.00909312,"threshold_uncertainty_score":0.1406008},"labels":[],"label_agreement":null},{"id":"W1536307206","doi":"10.1007/978-3-540-30228-5_11","title":"Developing a Minimalist Parser for Free Word Order Languages with Discontinuous Constituency","year":2004,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Parsing; Word order; Natural language processing; Artificial intelligence; Sentence; Top-down parsing; Parser combinator; Sketch; Phrase; Word (group theory); Linguistics; Algorithm","score_opus":0.015222180761953892,"score_gpt":0.2695471241693817,"score_spread":0.2543249434074278,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1536307206","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015618175,0.00015778025,0.9460693,0.00067769154,0.00015533481,0.00021976636,0.0014127498,0.030162102,0.005527121],"genre_scores_gemma":[0.13421112,0.0002240534,0.84243333,0.0004003733,0.00013554384,0.00028953396,0.0066638454,0.00699076,0.0086514875],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9988416,0.00024334552,0.00012887662,0.0003745339,0.00027149665,0.00014019302],"domain_scores_gemma":[0.9959526,0.0025826192,0.00013427783,0.00045624364,0.0007783491,0.00009592552],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015580835,0.0010708123,0.0016238891,0.0012578286,0.0015142534,0.0037100455,0.002940488,0.0018538206,0.014104348],"category_scores_gemma":[0.0053995405,0.0027948243,0.0019496434,0.0014386347,0.0012514335,0.006744322,0.004075838,0.0038496086,0.007884741],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003938516,0.0003265213,0.0028678095,0.0013212402,0.00021348082,0.0015142142,0.0028435227,0.03192169,0.05067798,0.23578034,0.08014835,0.59199095],"study_design_scores_gemma":[0.00018808259,0.00017366502,0.0011021327,0.00018163159,0.0003397188,0.0010366556,0.0012307537,0.52669275,0.07564668,0.30913565,0.08406915,0.00020307563],"about_ca_topic_score_codex":0.004076831,"about_ca_topic_score_gemma":0.010537322,"teacher_disagreement_score":0.014104348,"about_ca_system_score_codex":0.0014108329,"about_ca_system_score_gemma":0.003376858,"threshold_uncertainty_score":0.047183692},"labels":[],"label_agreement":null},{"id":"W1536638548","doi":"10.1007/978-3-642-14770-8_36","title":"Using Comparable Corpora to Improve the Effectiveness of Cross-Language Information Retrieval","year":2010,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Computer science; Natural language processing; Cross-language information retrieval; Terminology; Artificial intelligence; Information retrieval; Focus (optics); Selection (genetic algorithm); Machine translation; Point (geometry); Ambiguity; Text corpus; Linguistics","score_opus":0.014605610077631628,"score_gpt":0.2992259525697489,"score_spread":0.28462034249211726,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1536638548","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.23987865,0.037025854,0.45597744,0.0063441144,0.010281141,0.0045362287,0.054418497,0.056686748,0.13485125],"genre_scores_gemma":[0.31157723,0.008110849,0.49462655,0.0020564662,0.0019069513,0.0018685521,0.15464473,0.0062661613,0.018942572],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9867322,0.0063477755,0.001960056,0.0020312055,0.0024586332,0.0004700939],"domain_scores_gemma":[0.9545796,0.024000287,0.0009660532,0.0091423225,0.010720533,0.000591227],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.014052716,0.0027207823,0.0030742895,0.01607631,0.003261031,0.006320552,0.0031770093,0.0023913144,0.028205547],"category_scores_gemma":[0.066834,0.001290571,0.001677514,0.013095838,0.0011878693,0.013311733,0.006175163,0.0026717135,0.017875474],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012653939,0.0015397252,0.0055519673,0.0059006843,0.0010563505,0.0016630791,0.0018829545,0.0045623705,0.059922516,0.010101665,0.17632216,0.73023105],"study_design_scores_gemma":[0.0024863605,0.0023355323,0.029848449,0.0031864329,0.005821525,0.006944537,0.006099789,0.1566087,0.19159614,0.039101496,0.5549099,0.0010611577],"about_ca_topic_score_codex":0.0074402797,"about_ca_topic_score_gemma":0.011233363,"teacher_disagreement_score":0.028205547,"about_ca_system_score_codex":0.0011822303,"about_ca_system_score_gemma":0.0025431653,"threshold_uncertainty_score":0.094356954},"labels":[],"label_agreement":null},{"id":"W1538134818","doi":"10.13140/2.1.1779.4243","title":"Neural-Symbolic Learning and Reasoning: Contributions and Challenges","year":2015,"lang":"en","type":"article","venue":"City Research Online (City University London)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":127,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Acadia University","funders":"","keywords":"Connectionism; Computer science; Artificial intelligence; Artificial neural network; Representation (politics); Computation; Knowledge representation and reasoning; Symbolic-numeric computation; Key (lock); Models of neural computation; The Symbolic; Cognitive science; Machine learning; Programming language; Psychology","score_opus":0.09420937839203838,"score_gpt":0.3544807468344663,"score_spread":0.26027136844242793,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1538134818","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010781324,0.33715993,0.38439763,0.2094474,0.004195156,0.00008148872,0.00049747963,0.00087867887,0.05256102],"genre_scores_gemma":[0.290219,0.36107865,0.2892987,0.011655437,0.021564092,0.00029250083,0.0013697249,0.0006176879,0.02390423],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9958664,0.0016367912,0.00021956004,0.0006725338,0.0013957078,0.00020895056],"domain_scores_gemma":[0.9760094,0.017631378,0.00046246639,0.0025693914,0.0025644954,0.00076286803],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00831303,0.0012367943,0.0017789326,0.0024461614,0.001678698,0.008310923,0.003609952,0.0051628305,0.006198953],"category_scores_gemma":[0.018336091,0.0009232915,0.0008075751,0.004242578,0.009151514,0.02656297,0.006511308,0.009301102,0.0030268172],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000801172,0.00006848508,0.0003638629,0.00095603266,0.000050308485,0.00006929956,0.0004393076,0.0071680634,0.00025004544,0.7691047,0.025121653,0.19632813],"study_design_scores_gemma":[0.000006754426,0.00000963334,0.000098090604,0.00019745294,0.000007405958,0.00005235421,0.00020899427,0.013709723,0.00023955075,0.94041795,0.0450342,0.000017936483],"about_ca_topic_score_codex":0.0035858045,"about_ca_topic_score_gemma":0.0021280402,"teacher_disagreement_score":0.00831303,"about_ca_system_score_codex":0.003850936,"about_ca_system_score_gemma":0.003438702,"threshold_uncertainty_score":0.04396403},"labels":[],"label_agreement":null},{"id":"W1538415244","doi":"10.46430/phen0016","title":"Output Keywords in Context in an HTML File with Python","year":2012,"lang":"en","type":"article","venue":"The Programming Historian","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Python (programming language); Computer science; World Wide Web; Window (computing); Information retrieval; Context (archaeology); The Internet; Programming language; History","score_opus":0.016764025499693656,"score_gpt":0.2582049743164332,"score_spread":0.24144094881673953,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1538415244","genre_codex":"software","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0038207169,0.00011091696,0.19420896,0.0007763094,0.0004452692,0.00028139373,0.065953135,0.7079783,0.026425004],"genre_scores_gemma":[0.10171813,0.0006716607,0.33429167,0.0024343615,0.00034657287,0.001792105,0.12596193,0.34402433,0.08875924],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99948657,0.000047456248,0.000049893148,0.00014269764,0.00019408116,0.00007918797],"domain_scores_gemma":[0.9987373,0.00048623313,0.00008783867,0.00023626503,0.00034245782,0.00010998352],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007191244,0.0018122083,0.0008687257,0.0009474209,0.0005434119,0.0015039358,0.0018116109,0.0010808278,0.17344709],"category_scores_gemma":[0.005075212,0.00085552357,0.0012397971,0.0010322045,0.00045331684,0.002851467,0.0030950238,0.0017712681,0.09706583],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001047547,0.00020611002,0.002587646,0.0016152118,0.00009775593,0.0007285498,0.0004817505,0.0023319486,0.009779706,0.009042093,0.8103951,0.16168657],"study_design_scores_gemma":[0.0005976924,0.0001910314,0.005561845,0.00052422186,0.000106540174,0.00079654815,0.0004003596,0.04204975,0.07323878,0.057731606,0.8185299,0.00027177384],"about_ca_topic_score_codex":0.0017418573,"about_ca_topic_score_gemma":0.001888777,"teacher_disagreement_score":0.17344709,"about_ca_system_score_codex":0.000651624,"about_ca_system_score_gemma":0.0011868366,"threshold_uncertainty_score":0.58023834},"labels":[],"label_agreement":null},{"id":"W1539683106","doi":"10.21236/ada458694","title":"Construction of Chinese-English Semantic Hierarchy for Information Retrieval","year":2000,"lang":"en","type":"report","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba","funders":"Natural Sciences and Engineering Research Council of Canada; Defense Advanced Research Projects Agency; U.S. Department of Defense","keywords":"Hierarchy; Computer science; Information retrieval; Natural language processing; Artificial intelligence; Political science","score_opus":0.010654080821200325,"score_gpt":0.28214864997745015,"score_spread":0.2714945691562498,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1539683106","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.028216586,0.0007760345,0.9198294,0.00086963567,0.00013935803,0.00089384685,0.0049694683,0.005196758,0.039108936],"genre_scores_gemma":[0.14524058,0.0005174962,0.8381857,0.00021759569,0.0000448216,0.00051236595,0.008653625,0.0005256987,0.0061020954],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99907434,0.00022305766,0.00015105034,0.00021359546,0.00025509746,0.00008281865],"domain_scores_gemma":[0.9991339,0.00019791849,0.00006166801,0.000198627,0.00032399697,0.000083879124],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016036081,0.0004920536,0.0004752308,0.00600363,0.0021814476,0.0017710658,0.0007517504,0.00035998758,0.0057388376],"category_scores_gemma":[0.0027572708,0.00049331074,0.00094442104,0.005171623,0.0010376391,0.0048936564,0.002083292,0.0008419413,0.0021427118],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010960513,0.00010499978,0.0033123535,0.0007129513,0.00006467461,0.00043186796,0.0058328514,0.0031038234,0.02033188,0.64171505,0.032947592,0.29133242],"study_design_scores_gemma":[0.00007761288,0.00011601809,0.0057940683,0.00044152088,0.00026247735,0.0005886112,0.0035366942,0.07752926,0.025287248,0.28266212,0.60357255,0.00013190629],"about_ca_topic_score_codex":0.017227141,"about_ca_topic_score_gemma":0.026908828,"teacher_disagreement_score":0.017227141,"about_ca_system_score_codex":0.0022294396,"about_ca_system_score_gemma":0.0037268482,"threshold_uncertainty_score":0.034253716},"labels":[],"label_agreement":null},{"id":"W1539959748","doi":"","title":"Proceedings of CoNLL-2003, Edmonton, Canada","year":2003,"lang":"ca","type":"article","venue":"Data Archiving and Networked Services (DANS)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science","score_opus":0.009445589851074403,"score_gpt":0.2265567037773208,"score_spread":0.21711111392624638,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1539959748","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.040619716,0.02312805,0.29763263,0.02577277,0.020420123,0.0027272527,0.036436196,0.048704635,0.50455874],"genre_scores_gemma":[0.026306722,0.005431621,0.048122596,0.0014176292,0.00055436615,0.00031597726,0.028671425,0.004670743,0.8845089],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99828243,0.00027510434,0.00007594922,0.00036552543,0.0006771082,0.0003238484],"domain_scores_gemma":[0.99662614,0.00046221478,0.000058138106,0.0005110095,0.0017798535,0.0005627549],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037347975,0.0025279892,0.0022670547,0.0014456387,0.0033658594,0.007338851,0.0035221244,0.0017915188,0.122647025],"category_scores_gemma":[0.0032036845,0.0010567583,0.0011888625,0.0022745815,0.001700657,0.002918421,0.0018507827,0.0026407572,0.06430496],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00058726716,0.000393711,0.0009850627,0.00026798123,0.000074022726,0.00035751265,0.00030276508,0.0026795808,0.005250804,0.005011719,0.8334864,0.1506032],"study_design_scores_gemma":[0.0002075447,0.00014595353,0.0031283717,0.00023044967,0.0001517063,0.00034454727,0.0008976767,0.01678898,0.008006993,0.006183154,0.96383345,0.00008116468],"about_ca_topic_score_codex":0.5053858,"about_ca_topic_score_gemma":0.6465671,"teacher_disagreement_score":0.49461418,"about_ca_system_score_codex":0.009511751,"about_ca_system_score_gemma":0.015471396,"threshold_uncertainty_score":0.9950541},"labels":[],"label_agreement":null},{"id":"W1540125288","doi":"10.1002/0470018860.s00081","title":"Natural Language Processing, Disambiguation in","year":2005,"lang":"en","type":"other","venue":"Encyclopedia of Cognitive Science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Natural language processing; Computer science; Natural (archaeology); Artificial intelligence; Linguistics; History; Philosophy","score_opus":0.005620601048807358,"score_gpt":0.2898411098513931,"score_spread":0.28422050880258576,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1540125288","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019852433,0.08491095,0.4250221,0.021636745,0.0026118583,0.00033597508,0.0012427464,0.0029896025,0.44139752],"genre_scores_gemma":[0.4731173,0.048430894,0.32865342,0.0069650454,0.003100294,0.00066972635,0.0034529066,0.0010006625,0.13460977],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9984189,0.0005467866,0.000119064396,0.0004422362,0.00039703352,0.000075996846],"domain_scores_gemma":[0.9983277,0.00096096477,0.0001284869,0.00027083035,0.00023530304,0.000076718134],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001836546,0.0008096239,0.0005796164,0.0024874085,0.0020102924,0.008761255,0.0011531964,0.0016382431,0.015738217],"category_scores_gemma":[0.00644064,0.00046074245,0.00053630245,0.00390635,0.0049084253,0.0105639575,0.0037342953,0.0018686686,0.006199432],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010537955,0.00007517947,0.0007767745,0.0005613924,0.000035791545,0.00045222385,0.001383644,0.0024301165,0.0013239665,0.7317265,0.058760416,0.2023686],"study_design_scores_gemma":[0.000016094442,0.000008837557,0.0005071122,0.00017090808,0.000019837778,0.00027528493,0.00036002876,0.0056777736,0.0008929198,0.8671872,0.12486446,0.000019630366],"about_ca_topic_score_codex":0.004805988,"about_ca_topic_score_gemma":0.0038300373,"teacher_disagreement_score":0.015738217,"about_ca_system_score_codex":0.0018194998,"about_ca_system_score_gemma":0.002426614,"threshold_uncertainty_score":0.052649617},"labels":[],"label_agreement":null},{"id":"W1540222736","doi":"10.7551/mitpress/2708.003.0004","title":"A Geometric Approach to Mapping Bitext Correspondence","year":2001,"lang":"en","type":"book-chapter","venue":"The MIT Press eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":80,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Cartography; Geography","score_opus":0.05071402189334313,"score_gpt":0.25662295241757144,"score_spread":0.20590893052422832,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1540222736","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0008372723,0.0003176674,0.9843658,0.0002006515,0.00018718532,0.00004752175,0.00017270751,0.0015791557,0.012292056],"genre_scores_gemma":[0.030504096,0.0009973926,0.94299996,0.00025411265,0.00024146227,0.00025361733,0.001102519,0.0014449486,0.0222018],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99772197,0.00048309774,0.00015936524,0.0007526435,0.0007873582,0.00009552932],"domain_scores_gemma":[0.9982021,0.0004491009,0.00012160953,0.0007896543,0.00038813808,0.00004930913],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013266786,0.0018154527,0.0010565142,0.0055306335,0.001756279,0.0046002325,0.003286958,0.002190221,0.028902255],"category_scores_gemma":[0.0058104102,0.0012517936,0.0019126048,0.005783207,0.003378611,0.0063761133,0.0050143576,0.0031537153,0.016243419],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000057386824,0.000052532945,0.0004283809,0.00038706028,0.00006397245,0.00030904668,0.0006811849,0.019091545,0.0078227185,0.41324714,0.022591135,0.53526795],"study_design_scores_gemma":[0.000022144188,0.00012098384,0.00078753807,0.00013397669,0.000047660054,0.0018390709,0.00053590647,0.10348884,0.014111391,0.5900029,0.28880608,0.000103476836],"about_ca_topic_score_codex":0.0016212618,"about_ca_topic_score_gemma":0.001586579,"teacher_disagreement_score":0.028902255,"about_ca_system_score_codex":0.00097357953,"about_ca_system_score_gemma":0.0010598964,"threshold_uncertainty_score":0.096687675},"labels":[],"label_agreement":null},{"id":"W1541090817","doi":"10.21236/ada458682","title":"Large-Scale Construction of a Chinese-English Semantic Hierarchy","year":2000,"lang":"en","type":"report","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Defense Advanced Research Projects Agency; U.S. Department of Defense","keywords":"Computer science; Hierarchy; Natural language processing; Lexicon; Machine translation; Artificial intelligence; Thematic map; Scale (ratio); Linguistics; Information retrieval; Geography","score_opus":0.0087973133222273,"score_gpt":0.2759066546152737,"score_spread":0.2671093412930464,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1541090817","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0821187,0.00096327584,0.8716673,0.0018684296,0.00021404773,0.0006618611,0.0057665026,0.004724349,0.032015525],"genre_scores_gemma":[0.28981405,0.00054796867,0.6888597,0.00024762112,0.00004338339,0.00035958397,0.012510429,0.0006374723,0.00697977],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9990632,0.00024304578,0.0001024015,0.0002264641,0.0002845873,0.00008027773],"domain_scores_gemma":[0.9984211,0.00043590853,0.000101413796,0.00038696206,0.0005719348,0.000082721024],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018808023,0.0004866521,0.00042950746,0.0031116444,0.002172577,0.0015461905,0.0008793318,0.0003387969,0.005032421],"category_scores_gemma":[0.003993947,0.00051370426,0.0008583828,0.0027789895,0.0008945402,0.0042560757,0.0022874442,0.0009878131,0.002419994],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023386329,0.00019860009,0.011187664,0.0012723716,0.00012963815,0.0010976932,0.00704082,0.006583082,0.040140543,0.2649213,0.049272068,0.6179223],"study_design_scores_gemma":[0.00016287499,0.00017919786,0.01755305,0.0006670684,0.0004476931,0.0010225594,0.005993382,0.13318878,0.08115663,0.24099286,0.5184499,0.00018595056],"about_ca_topic_score_codex":0.015332327,"about_ca_topic_score_gemma":0.022410732,"teacher_disagreement_score":0.015332327,"about_ca_system_score_codex":0.0014679387,"about_ca_system_score_gemma":0.0045134374,"threshold_uncertainty_score":0.030486166},"labels":[],"label_agreement":null},{"id":"W1541914436","doi":"10.1007/11766247_24","title":"Unsupervised Labeling of Noun Clusters","year":2006,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Noun; Linguistics; Natural language processing; Computer science; Artificial intelligence; Psychology; Philosophy","score_opus":0.011750333482289908,"score_gpt":0.24511323993200274,"score_spread":0.23336290644971283,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1541914436","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.047419436,0.0014887975,0.91338754,0.00071045425,0.00031602272,0.00043944333,0.009447435,0.009752653,0.017038194],"genre_scores_gemma":[0.15377787,0.00087220105,0.78746516,0.0002552151,0.0001776527,0.00040405893,0.0350044,0.0018172112,0.020226283],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9981305,0.0003515143,0.00013003754,0.00076520693,0.00045410293,0.00016865865],"domain_scores_gemma":[0.99684614,0.0010240957,0.00021478979,0.0006696833,0.0010585167,0.00018677604],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009603052,0.0011237138,0.001084336,0.0046832273,0.0024689988,0.0028919824,0.0028494124,0.0014618913,0.008302478],"category_scores_gemma":[0.0041133757,0.0008789808,0.0016250506,0.0053390176,0.0010479849,0.002578897,0.002185933,0.0017618538,0.007143953],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00064598257,0.00031289225,0.006376507,0.0007440399,0.00019349209,0.00042152908,0.00077517354,0.010537233,0.04800297,0.032601938,0.058095977,0.8412922],"study_design_scores_gemma":[0.00016479497,0.00020680153,0.012623809,0.0003908841,0.00032074982,0.001447596,0.0016095311,0.60539305,0.07119524,0.16357896,0.14290395,0.00016469158],"about_ca_topic_score_codex":0.009747641,"about_ca_topic_score_gemma":0.032163903,"teacher_disagreement_score":0.009747641,"about_ca_system_score_codex":0.0015996028,"about_ca_system_score_gemma":0.0032595138,"threshold_uncertainty_score":0.027774572},"labels":[],"label_agreement":null},{"id":"W1542447206","doi":"10.1007/3-540-45399-7_13","title":"Using Information Extraction and Natural Language Generation to Answer E-Mail","year":2001,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Information extraction; Natural language; Natural language generation; Question answering; Natural language processing; Artificial intelligence; World Wide Web; Information retrieval","score_opus":0.01768209204753136,"score_gpt":0.28808717223935154,"score_spread":0.2704050801918202,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1542447206","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025867399,0.00058928237,0.9435841,0.0007201931,0.00023415657,0.0005294952,0.0010914067,0.019759642,0.0076243132],"genre_scores_gemma":[0.12547024,0.00035631337,0.86354506,0.00031270826,0.00014931678,0.00033225617,0.0039612846,0.0006298284,0.0052429345],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9984138,0.00068157125,0.00014377512,0.0003088745,0.0003698867,0.00008198609],"domain_scores_gemma":[0.9955279,0.0032710265,0.00017748857,0.00030175823,0.00067050447,0.0000512793],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017022281,0.0010273497,0.0009035902,0.002588041,0.00082294334,0.0015726314,0.0013866586,0.0013707457,0.0075714453],"category_scores_gemma":[0.0068262434,0.00060721074,0.0011348963,0.0016420628,0.00060561724,0.0026947216,0.0010980159,0.0009967069,0.004267918],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00038739826,0.00044642983,0.0017848365,0.0006850071,0.000096329306,0.0006802002,0.0006859916,0.011006399,0.037924014,0.020022187,0.028666275,0.8976149],"study_design_scores_gemma":[0.0003298819,0.0003834895,0.0018553054,0.0001709498,0.00036520814,0.0012419599,0.0005257194,0.69128954,0.13383953,0.10778744,0.062088333,0.00012269692],"about_ca_topic_score_codex":0.0016163413,"about_ca_topic_score_gemma":0.002112371,"teacher_disagreement_score":0.0075714453,"about_ca_system_score_codex":0.0005674033,"about_ca_system_score_gemma":0.00082345953,"threshold_uncertainty_score":0.025329053},"labels":[],"label_agreement":null},{"id":"W1545521613","doi":"","title":"Résumés de texte en langue maternelle et en langue seconde : Différences dans l'application des macrorègles entre experts et étudiants de différents niveaux universitaires","year":2001,"lang":"en","type":"article","venue":"DOAJ (DOAJ: Directory of Open Access Journals)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Acadia University","funders":"","keywords":"Humanities; Physics; Philosophy","score_opus":0.07712572407597482,"score_gpt":0.47984958709837827,"score_spread":0.4027238630224035,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1545521613","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98046345,0.0012428833,0.0045875693,0.00056661555,0.000064827225,0.00011621513,0.0023191227,0.0005003962,0.0101389],"genre_scores_gemma":[0.9822515,0.00091778423,0.0046820054,0.00008013554,0.000044170378,0.00010621611,0.002891013,0.0001190103,0.008908068],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99773395,0.0012323281,0.00023653447,0.00019970104,0.0005303694,0.00006713356],"domain_scores_gemma":[0.95629466,0.033306938,0.003053931,0.0015932011,0.0050572273,0.00069406856],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002404646,0.00036482612,0.00032703072,0.001808561,0.0005113293,0.0013238146,0.00027176284,0.0003583936,0.00816189],"category_scores_gemma":[0.042636644,0.00013175397,0.00027152122,0.0013360305,0.0004133169,0.0010728887,0.00053608656,0.00033315964,0.0018479589],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020153956,0.00056220713,0.2034212,0.0033553075,0.0002565426,0.0027420176,0.14101803,0.004569463,0.07961582,0.0038342213,0.02115258,0.5374571],"study_design_scores_gemma":[0.00009843299,0.0019224985,0.6720875,0.00095727376,0.00026260523,0.00373506,0.0602854,0.00976065,0.058893926,0.0033739768,0.1883582,0.00026454809],"about_ca_topic_score_codex":0.0042849495,"about_ca_topic_score_gemma":0.0029069064,"teacher_disagreement_score":0.00816189,"about_ca_system_score_codex":0.0005358247,"about_ca_system_score_gemma":0.00040818047,"threshold_uncertainty_score":0.027304292},"labels":[],"label_agreement":null},{"id":"W1546453457","doi":"10.1109/icassp.2000.862072","title":"French large vocabulary recognition with cross-word phonology transducers","year":2002,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Computer Research Institute of Montréal","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Word error rate; Speech recognition; Vocabulary; Pronunciation; Word (group theory); Dictation; Context (archaeology); Language model; Artificial intelligence; Hidden Markov model; Natural language processing; Task (project management); Linguistics","score_opus":0.0174505513817706,"score_gpt":0.2516273653355634,"score_spread":0.2341768139537928,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1546453457","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.23227301,0.0002773436,0.72455823,0.00023560532,0.00013838084,0.000160673,0.0008005517,0.03752409,0.004032054],"genre_scores_gemma":[0.6093617,0.00010861616,0.38350952,0.0001833346,0.000051027033,0.00024878504,0.0017899061,0.0007974207,0.0039497684],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99906844,0.00023395263,0.00009037457,0.00032777185,0.00018951012,0.00008997605],"domain_scores_gemma":[0.9971819,0.0014960208,0.00009697626,0.00052394986,0.00064419827,0.000056962286],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00081572076,0.00065888674,0.00079544785,0.00051511935,0.00039967557,0.0011943892,0.0008768365,0.00076912256,0.0034157839],"category_scores_gemma":[0.004208356,0.00044382663,0.00045717976,0.0005243202,0.00042687188,0.0019289475,0.0007730261,0.0007877444,0.001941754],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008609483,0.00022354566,0.0044996925,0.00025851995,0.00016074356,0.0006384277,0.0006692828,0.048843052,0.31984353,0.006185968,0.006312528,0.6115038],"study_design_scores_gemma":[0.00007545183,0.0003170098,0.0038875595,0.0000144902615,0.00007579144,0.0004437224,0.00012914905,0.74433196,0.24075146,0.003569632,0.0063165925,0.00008720048],"about_ca_topic_score_codex":0.015859473,"about_ca_topic_score_gemma":0.021054382,"teacher_disagreement_score":0.015859473,"about_ca_system_score_codex":0.0006646132,"about_ca_system_score_gemma":0.00081367086,"threshold_uncertainty_score":0.031534374},"labels":[],"label_agreement":null},{"id":"W1551436493","doi":"10.1007/978-3-642-01818-3_45","title":"A Semi-supervised Approach to Bengali-English Phrase-Based Statistical Machine Translation","year":2009,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Bengali; Computer science; Phrase; Machine translation; Natural language processing; Artificial intelligence; Machine translation software usability; Evaluation of machine translation; Example-based machine translation; Translation (biology)","score_opus":0.017684208910393368,"score_gpt":0.2575843355221754,"score_spread":0.23990012661178203,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1551436493","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008364458,0.00075060467,0.9760139,0.00030154563,0.00025799967,0.00019789825,0.0010185088,0.009102589,0.003992595],"genre_scores_gemma":[0.13538675,0.0006994605,0.84339803,0.00034308297,0.00032679032,0.0005505932,0.0075024916,0.0019426837,0.009850146],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99597794,0.0019018016,0.0003718917,0.000753731,0.0007709686,0.00022377551],"domain_scores_gemma":[0.9937727,0.002385584,0.00032838323,0.0011041106,0.0022744501,0.00013484417],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002281078,0.0012871082,0.0019543672,0.0021128512,0.0015474285,0.0028279554,0.0026805676,0.0011353294,0.005066681],"category_scores_gemma":[0.007397032,0.00079226535,0.0014335717,0.0035626495,0.0007441995,0.002111169,0.002554765,0.0023644564,0.008143486],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005302603,0.0004824262,0.0009665506,0.0007064706,0.0004051539,0.00040048506,0.0006902161,0.01949404,0.05244903,0.017133094,0.035884876,0.8708574],"study_design_scores_gemma":[0.00009637908,0.00033948186,0.0022720932,0.00008169366,0.00027193822,0.000790651,0.00046753717,0.85535485,0.06724207,0.03712856,0.03578685,0.00016789921],"about_ca_topic_score_codex":0.0053489255,"about_ca_topic_score_gemma":0.0096146045,"teacher_disagreement_score":0.0053489255,"about_ca_system_score_codex":0.0008819199,"about_ca_system_score_gemma":0.002484411,"threshold_uncertainty_score":0.016949773},"labels":[],"label_agreement":null},{"id":"W1551794154","doi":"","title":"Transliteration Generation and Mining with Limited Training Resources","year":2010,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":39,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Transliteration; Computer science; Discriminative model; Artificial intelligence; Natural language processing; Joint (building); Training set; Sequence (biology); n-gram; Language model; Machine learning; Engineering","score_opus":0.019454530742981983,"score_gpt":0.24201050284886888,"score_spread":0.2225559721058869,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1551794154","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.032256074,0.0005582831,0.9114462,0.00046241164,0.00019359271,0.00019863056,0.0022300743,0.048826877,0.0038279323],"genre_scores_gemma":[0.18875986,0.00030548064,0.78794277,0.00028389247,0.00012621871,0.0003069405,0.010921457,0.0021398298,0.009213679],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9987935,0.00031034692,0.00007709131,0.00047040224,0.00024247738,0.00010618055],"domain_scores_gemma":[0.99635893,0.001301948,0.00027157448,0.0014492186,0.0004923426,0.00012601788],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010672702,0.0016456586,0.0014867563,0.0011875683,0.0007779317,0.0012110174,0.0033170478,0.0012416088,0.010328582],"category_scores_gemma":[0.00613895,0.0007549986,0.0008591441,0.0016287806,0.0005444535,0.003194245,0.00239187,0.0016887392,0.013166813],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005464653,0.00032803632,0.0019475148,0.00066150114,0.00010580446,0.0005383462,0.0002246435,0.0409385,0.039769083,0.0039169462,0.037336253,0.8736869],"study_design_scores_gemma":[0.00011131445,0.00023170783,0.00073033595,0.000042601805,0.000054289518,0.0006078911,0.00010417857,0.90575486,0.06402391,0.010403853,0.01788556,0.000049465496],"about_ca_topic_score_codex":0.0022512793,"about_ca_topic_score_gemma":0.0064012515,"teacher_disagreement_score":0.010328582,"about_ca_system_score_codex":0.00063458463,"about_ca_system_score_gemma":0.0013867558,"threshold_uncertainty_score":0.034552574},"labels":[],"label_agreement":null},{"id":"W1553669113","doi":"10.1007/978-3-642-12116-6_43","title":"Identification of Translationese: A Machine Learning Approach","year":2010,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":74,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Artificial intelligence; Support vector machine; Machine learning; Classifier (UML)","score_opus":0.012428459942297453,"score_gpt":0.2552222638959913,"score_spread":0.24279380395369382,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1553669113","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.106261924,0.0021066507,0.85216594,0.00068087556,0.00056884944,0.00038465345,0.0040090447,0.011989659,0.021832354],"genre_scores_gemma":[0.40295547,0.0015155177,0.5500119,0.00026847454,0.0003521413,0.0002822657,0.014105677,0.002277721,0.028230939],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991968,0.00013821089,0.000061753344,0.00030293703,0.00018287056,0.000117484014],"domain_scores_gemma":[0.99796474,0.0005582834,0.00017870631,0.0003773703,0.0008235085,0.0000974382],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00084346853,0.0015748378,0.0013164512,0.0036323934,0.0014715377,0.0028925445,0.0012563965,0.0015822357,0.011402053],"category_scores_gemma":[0.0033485196,0.0004914077,0.0013104597,0.0035583575,0.00067952974,0.00303828,0.0014429622,0.0018739955,0.011513445],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005992684,0.0003253564,0.007199232,0.00049758714,0.00011923894,0.0010198833,0.0004520936,0.0069169463,0.057332303,0.014969712,0.016789868,0.89377856],"study_design_scores_gemma":[0.00014550649,0.00063979736,0.012974314,0.00022503498,0.0004674654,0.0035015983,0.0016601529,0.7239512,0.11923259,0.061588567,0.075415894,0.00019785615],"about_ca_topic_score_codex":0.0027018776,"about_ca_topic_score_gemma":0.00322848,"teacher_disagreement_score":0.011402053,"about_ca_system_score_codex":0.00049327855,"about_ca_system_score_gemma":0.0014601753,"threshold_uncertainty_score":0.038143635},"labels":[],"label_agreement":null},{"id":"W1553770895","doi":"10.1007/11878773_83","title":"Using Various Indexing Schemes and Multiple Translations in the CL-SR Task at CLEF 2005","year":2006,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Clef; Computer science; Search engine indexing; Information retrieval; Weighting; Task (project management); Natural language processing; Scheme (mathematics); Document retrieval; Artificial intelligence; Mathematics","score_opus":0.021290792088598617,"score_gpt":0.26971849834721756,"score_spread":0.24842770625861893,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1553770895","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4308045,0.0090496745,0.33076635,0.010409498,0.0032057515,0.0021277668,0.04676786,0.106219746,0.06064889],"genre_scores_gemma":[0.35510418,0.0012915869,0.49224398,0.0015732385,0.0006554332,0.000633928,0.11648709,0.009230588,0.022780068],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9925055,0.003293485,0.0008494948,0.0017279105,0.0011108851,0.00051273586],"domain_scores_gemma":[0.9869572,0.0068805036,0.0004112174,0.0027537567,0.0025400121,0.00045729004],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0072219414,0.0025390005,0.003125165,0.00311743,0.002992651,0.0035993163,0.0031151164,0.004039215,0.029046088],"category_scores_gemma":[0.01856536,0.001071027,0.0017997752,0.0044309413,0.00096448464,0.008324843,0.0028885715,0.00312278,0.014576523],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002940103,0.0010813343,0.0019743342,0.0025606689,0.00023585962,0.00073080015,0.0009742594,0.007413278,0.030760562,0.010150874,0.29767248,0.6435055],"study_design_scores_gemma":[0.006037851,0.0037181377,0.008187696,0.0007127994,0.0010944656,0.004048013,0.0057684383,0.45695254,0.18642999,0.08375112,0.24231341,0.0009854932],"about_ca_topic_score_codex":0.007726159,"about_ca_topic_score_gemma":0.014959005,"teacher_disagreement_score":0.029046088,"about_ca_system_score_codex":0.0011590343,"about_ca_system_score_gemma":0.0035378914,"threshold_uncertainty_score":0.09716886},"labels":[],"label_agreement":null},{"id":"W1556481571","doi":"","title":"Improved Natural Language Learning via Variance-Regularization Support Vector Machines","year":2010,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Support vector machine; Regularization (linguistics); Computer science; Machine learning; Artificial intelligence; Ranking SVM; Variance (accounting); Pattern recognition (psychology)","score_opus":0.0032146007211074127,"score_gpt":0.24009704032381177,"score_spread":0.23688243960270436,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1556481571","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008047382,0.00023716546,0.9891566,0.00016694936,0.000049782848,0.000016108652,0.000044396344,0.0016734094,0.000608136],"genre_scores_gemma":[0.25493705,0.00027928944,0.7388115,0.0003855719,0.00024749027,0.00015172065,0.0007563531,0.00058896886,0.0038420632],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9969988,0.0011816047,0.00018005908,0.0004714793,0.0009871257,0.00018092782],"domain_scores_gemma":[0.9967193,0.0015717677,0.00021779566,0.00061393465,0.0008073706,0.00006977944],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029163447,0.0009432839,0.0013709089,0.0011546782,0.00042501243,0.0012703454,0.0020210366,0.0013256769,0.002183963],"category_scores_gemma":[0.011138971,0.00054282875,0.0009260998,0.0012395858,0.00075121864,0.00300454,0.0018366263,0.002579125,0.0017781181],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002944447,0.0002545542,0.001300393,0.00019132768,0.00015702714,0.00015143956,0.00016907875,0.2993719,0.018057277,0.039918426,0.011515834,0.6286183],"study_design_scores_gemma":[0.000013614161,0.000027100126,0.00013545192,0.000004517002,0.000007018885,0.000024152983,0.0000042012043,0.9851436,0.001794865,0.011911714,0.0009240368,0.000009700583],"about_ca_topic_score_codex":0.001410427,"about_ca_topic_score_gemma":0.0021038183,"teacher_disagreement_score":0.0029163447,"about_ca_system_score_codex":0.0005496431,"about_ca_system_score_gemma":0.0008831944,"threshold_uncertainty_score":0.015423298},"labels":[],"label_agreement":null},{"id":"W1556778233","doi":"","title":"METISII: Example-based Machine Translation Using Monolingual CorporaSystem Description","year":2005,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Metis; Machine translation; Computer science; Natural language processing; Artificial intelligence; Translation (biology); Machine translation software usability; Computer-assisted translation; Example-based machine translation; Programming language; World Wide Web","score_opus":0.0743452629739193,"score_gpt":0.2932830911342762,"score_spread":0.2189378281603569,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1556778233","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00947458,0.0027741813,0.6544055,0.0019362771,0.0011177857,0.0020081133,0.038302347,0.21203126,0.07794994],"genre_scores_gemma":[0.050835688,0.0018682242,0.6982753,0.00075947016,0.0005811592,0.0028522685,0.14238003,0.020359872,0.08208793],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9978504,0.0005059662,0.0001792118,0.0004425633,0.00087072107,0.00015111035],"domain_scores_gemma":[0.9989262,0.00016171194,0.000077419194,0.00032394135,0.00039864777,0.0001119693],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016503447,0.0022711202,0.001458889,0.0027764034,0.0011158516,0.0038111885,0.0037043267,0.0014993243,0.08692234],"category_scores_gemma":[0.0035099385,0.001539847,0.0009117159,0.0030865506,0.0007769039,0.0043349457,0.0041017947,0.0025330537,0.06427912],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000926805,0.00027861723,0.0012017098,0.0025830178,0.00012253203,0.0009786735,0.00065140385,0.0067448462,0.03838975,0.056026302,0.5451845,0.34691188],"study_design_scores_gemma":[0.00035186717,0.0002555198,0.002092742,0.0002574072,0.00006339534,0.0020941694,0.00020389132,0.07348735,0.04326827,0.026519902,0.8512322,0.0001733601],"about_ca_topic_score_codex":0.0021629783,"about_ca_topic_score_gemma":0.0022909793,"teacher_disagreement_score":0.08692234,"about_ca_system_score_codex":0.0011522274,"about_ca_system_score_gemma":0.0014427287,"threshold_uncertainty_score":0.29078424},"labels":[],"label_agreement":null},{"id":"W1557217172","doi":"10.1007/978-3-540-74628-7_21","title":"Word Distribution Based Methods for Minimizing Segment Overlaps","year":2007,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Guelph","funders":"","keywords":"Cohesion (chemistry); Computer science; Natural language processing; Text segmentation; Word (group theory); Segmentation; Artificial intelligence; Task (project management); Sequence (biology); Linguistics; Engineering","score_opus":0.030367122043345688,"score_gpt":0.34208663133721773,"score_spread":0.31171950929387204,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1557217172","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010772478,0.00082130946,0.9845659,0.00008299655,0.00008688646,0.0000922589,0.00029807707,0.002331243,0.0009487819],"genre_scores_gemma":[0.10308762,0.0008706648,0.87862337,0.0001918266,0.00023470653,0.0004504539,0.0040708864,0.0021180685,0.010352346],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99795866,0.00037122174,0.00017769235,0.0005454242,0.00075220666,0.00019476067],"domain_scores_gemma":[0.99568594,0.0026106527,0.0001720266,0.00043923,0.0009807068,0.00011157576],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020063224,0.0020844373,0.0027062215,0.004299731,0.0012165642,0.0017977769,0.002992321,0.0023987347,0.008062995],"category_scores_gemma":[0.0056845327,0.0011279543,0.001627088,0.005348971,0.0010171991,0.0030835413,0.0023919484,0.0018074629,0.004427964],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006968411,0.00018926727,0.0012082467,0.00037454913,0.00015853708,0.00017000517,0.0002195813,0.061028242,0.023110166,0.008560121,0.009452296,0.8948322],"study_design_scores_gemma":[0.0000897848,0.00015632113,0.0012599183,0.000044964658,0.00014340674,0.00027127995,0.0001641575,0.9440269,0.020039028,0.024791608,0.00897041,0.000042283387],"about_ca_topic_score_codex":0.0052023027,"about_ca_topic_score_gemma":0.0116072595,"teacher_disagreement_score":0.008062995,"about_ca_system_score_codex":0.00086588366,"about_ca_system_score_gemma":0.0017898013,"threshold_uncertainty_score":0.026973367},"labels":[],"label_agreement":null},{"id":"W1557439938","doi":"","title":"Shallow-Transfer Rule-Based Machine Translation between Icelandic and Swedish. Developing Apertium-is-sv: A Bidirectional Open-Source RBMT Application for Icelandic and Swedish","year":2013,"lang":"en","type":"article","venue":"Skemman","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Alberta-Pacific Forest Industries","keywords":"Icelandic; Translation (biology); Computer science; Open source; Artificial intelligence; Chemistry; Linguistics; Software","score_opus":0.02240229384020454,"score_gpt":0.2746513759470874,"score_spread":0.25224908210688285,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1557439938","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11173948,0.0010406644,0.7387512,0.00096228113,0.00062469085,0.0005323442,0.00920008,0.060654525,0.076494835],"genre_scores_gemma":[0.30222484,0.0005669963,0.6434869,0.0003652106,0.00009211898,0.00030985553,0.019557282,0.004604443,0.02879236],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9992448,0.00022865107,0.00010812543,0.00020975638,0.00015699679,0.000051599825],"domain_scores_gemma":[0.99892634,0.00051875005,0.00008179429,0.00014935997,0.00028782972,0.000035849025],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010628033,0.0007698312,0.00048685068,0.00086919515,0.00069894106,0.0015636345,0.00075688964,0.00065506797,0.009035722],"category_scores_gemma":[0.0026836952,0.00039366435,0.00049740926,0.00059303263,0.0005263361,0.0014821994,0.0012400356,0.0007016072,0.006341724],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004166252,0.00020924989,0.0037277278,0.001346031,0.00012327815,0.0012676402,0.0027698847,0.018455725,0.10199612,0.030242356,0.052756652,0.7866887],"study_design_scores_gemma":[0.00017185543,0.00029167035,0.005982596,0.0004921775,0.00017484748,0.002090208,0.0022111333,0.19655803,0.3065538,0.03417462,0.45112607,0.00017293375],"about_ca_topic_score_codex":0.004674482,"about_ca_topic_score_gemma":0.008340916,"teacher_disagreement_score":0.009035722,"about_ca_system_score_codex":0.0006539412,"about_ca_system_score_gemma":0.0019088845,"threshold_uncertainty_score":0.030227542},"labels":[],"label_agreement":null},{"id":"W1559186764","doi":"10.1007/978-3-540-45091-7_26","title":"Sesei: A CG-Based Filter for Internet Search Engines","year":2003,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Computer science; The Internet; Filter (signal processing); Search engine; World Wide Web; Information retrieval; Artificial intelligence; Speech recognition; Computer vision","score_opus":0.022449494893691795,"score_gpt":0.2802579448231724,"score_spread":0.2578084499294806,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1559186764","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018004114,0.000615344,0.6976673,0.0004286117,0.00028962982,0.00050775707,0.011897829,0.26134518,0.009244228],"genre_scores_gemma":[0.14646408,0.00054111064,0.763673,0.0007651122,0.00029835227,0.00057300733,0.031486433,0.0145227695,0.041676152],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99914896,0.00010314031,0.000080403384,0.00015021244,0.00041419134,0.00010313726],"domain_scores_gemma":[0.9978276,0.00086712383,0.00012121603,0.00052861637,0.0004900474,0.00016535986],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011914723,0.0013788647,0.0015660602,0.004275489,0.00090716017,0.0021477104,0.0017826699,0.0014030366,0.02046576],"category_scores_gemma":[0.0054316837,0.00072797516,0.0011652374,0.0030616012,0.00045836027,0.0029694256,0.0017447078,0.001222447,0.010995127],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00213259,0.0004462131,0.0046278555,0.00088615797,0.00027452325,0.00038295222,0.0003466274,0.009115722,0.035959616,0.021453826,0.23737621,0.68699783],"study_design_scores_gemma":[0.0005840671,0.0005153769,0.0056249355,0.00017299008,0.00036015396,0.00067240524,0.00022853384,0.51546675,0.1073182,0.04495723,0.32388175,0.00021767913],"about_ca_topic_score_codex":0.008464719,"about_ca_topic_score_gemma":0.015502673,"teacher_disagreement_score":0.02046576,"about_ca_system_score_codex":0.0008382837,"about_ca_system_score_gemma":0.0016110683,"threshold_uncertainty_score":0.068464816},"labels":[],"label_agreement":null},{"id":"W1560117764","doi":"10.1007/s11390-011-9410-0","title":"A New Multiword Expression Metric and Its Applications","year":2011,"lang":"en","type":"article","venue":"Journal of Computer Science and Technology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Metric (unit); Artificial intelligence; Natural language processing; Theory of computation; Question answering; Semantics (computer science); Expression (computer science); Natural language; Programming language","score_opus":0.015518475762255969,"score_gpt":0.26688788907247896,"score_spread":0.251369413310223,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1560117764","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03449352,0.0036240031,0.95158803,0.00082296965,0.00059444754,0.00014082923,0.0011798191,0.0014382869,0.0061180606],"genre_scores_gemma":[0.23490167,0.00191569,0.750461,0.00032257318,0.0008217136,0.0004960664,0.0025688435,0.00078542053,0.007727074],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9951244,0.001424139,0.00070758513,0.00091662636,0.001685348,0.00014189188],"domain_scores_gemma":[0.9924469,0.00282379,0.000518426,0.0008641094,0.0028743662,0.00047242342],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003454671,0.0010480133,0.0014819553,0.0054815905,0.0011513787,0.0027752973,0.0017367831,0.0015383404,0.004754843],"category_scores_gemma":[0.01506962,0.0003586694,0.0010214535,0.0064531695,0.0010095976,0.0059488188,0.0023722108,0.0015013238,0.0024244955],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00048859156,0.0002246875,0.0055597248,0.00058862317,0.00014565693,0.00030205568,0.00046466442,0.010020393,0.023359805,0.10112293,0.01123189,0.846491],"study_design_scores_gemma":[0.00009742097,0.0010077859,0.01000823,0.0001939997,0.0002700729,0.002448668,0.0006664782,0.63364094,0.024861429,0.2569485,0.06959409,0.00026245904],"about_ca_topic_score_codex":0.0012770589,"about_ca_topic_score_gemma":0.0010516209,"teacher_disagreement_score":0.0054815905,"about_ca_system_score_codex":0.000993265,"about_ca_system_score_gemma":0.0010575162,"threshold_uncertainty_score":0.018270254},"labels":[],"label_agreement":null},{"id":"W1560635010","doi":"10.1023/a:1007653929870","title":"Stochastic Grammatical Inference of Text Database Structure","year":2000,"lang":"en","type":"article","venue":"Machine Learning","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":41,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada; University of Waterloo","keywords":"Grammar induction; Computer science; Nesting (process); Markup language; Inference; Construct (python library); Semantics (computer science); Grammar; Artificial intelligence; Natural language processing; Theoretical computer science; Programming language; XML; Rule-based machine translation; Linguistics","score_opus":0.008191670648523592,"score_gpt":0.2716978941340115,"score_spread":0.2635062234854879,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1560635010","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07095329,0.00023420618,0.9199224,0.0020547227,0.00012592926,0.00006998386,0.0012928292,0.0023185639,0.0030280843],"genre_scores_gemma":[0.7457427,0.00030303063,0.24674068,0.00048635385,0.00026991018,0.0001161711,0.002746617,0.0005805786,0.003013992],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99717975,0.0012086852,0.00023519795,0.0005819648,0.0006152524,0.00017909153],"domain_scores_gemma":[0.9683925,0.02529387,0.0010535449,0.002208408,0.0027408751,0.00031075737],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034547923,0.00035659346,0.00088788994,0.002170717,0.0010639314,0.0023053861,0.002358176,0.0014764927,0.004576388],"category_scores_gemma":[0.033008408,0.0009271363,0.0013843726,0.0017061105,0.0013311197,0.006336168,0.0018660484,0.0020239642,0.0008856315],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00046884175,0.00027175085,0.009204257,0.00056694215,0.00017535617,0.0004959423,0.00096945255,0.21542104,0.008545945,0.49446398,0.020190269,0.2492263],"study_design_scores_gemma":[0.000026223703,0.000015208555,0.00048062354,0.000019624415,0.000031239637,0.000059812082,0.000058190817,0.7679944,0.0022170409,0.22762126,0.0014619699,0.000014409198],"about_ca_topic_score_codex":0.007129606,"about_ca_topic_score_gemma":0.0127827125,"teacher_disagreement_score":0.007129606,"about_ca_system_score_codex":0.002040903,"about_ca_system_score_gemma":0.0019078419,"threshold_uncertainty_score":0.01827091},"labels":[],"label_agreement":null},{"id":"W1560709708","doi":"10.1007/3-540-45153-6_4","title":"A Statistical Corpus-Based Term Extractor","year":2001,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":107,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Perplexity; Extractor; Term (time); Natural language processing; Artificial intelligence; Lexicography; Language model; Linguistics","score_opus":0.016066060833879844,"score_gpt":0.27804188393684853,"score_spread":0.2619758231029687,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1560709708","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010592877,0.0017740999,0.8878135,0.0003614252,0.00060139835,0.0006038201,0.021011546,0.07293578,0.0043055783],"genre_scores_gemma":[0.028672155,0.0009324369,0.9095166,0.00018206009,0.0003373313,0.0007560346,0.047115292,0.0024196394,0.010068475],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9983918,0.0002030962,0.0002278886,0.0004284651,0.00064835954,0.00010028231],"domain_scores_gemma":[0.9960024,0.0013742084,0.00022929873,0.0005838537,0.0016541108,0.00015615461],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019924613,0.0017678958,0.0021032628,0.009937399,0.0012618888,0.0022138823,0.0020910546,0.0011740196,0.016856384],"category_scores_gemma":[0.006690515,0.0009914383,0.0017332103,0.009513778,0.00049372896,0.0033580528,0.0022505578,0.001983907,0.02526328],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003968746,0.00014190705,0.0012922009,0.0006241666,0.00018291564,0.00022861293,0.00009974767,0.002513322,0.052097507,0.0030598403,0.057555854,0.881807],"study_design_scores_gemma":[0.00050145586,0.0007514829,0.011345171,0.00032604163,0.0010652127,0.002554473,0.0006176247,0.49368042,0.1835856,0.024592534,0.28051835,0.00046154222],"about_ca_topic_score_codex":0.005511184,"about_ca_topic_score_gemma":0.010467046,"teacher_disagreement_score":0.016856384,"about_ca_system_score_codex":0.00055817794,"about_ca_system_score_gemma":0.0031465418,"threshold_uncertainty_score":0.056390166},"labels":[],"label_agreement":null},{"id":"W1561788061","doi":"10.1007/3-540-36456-0_25","title":"Automatic Sense Disambiguation of the Near-Synonyms in a Dictionary Entry","year":2003,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"WordNet; Computer science; Natural language processing; Artificial intelligence; Word-sense disambiguation; Information retrieval","score_opus":0.009501691656527785,"score_gpt":0.24101613695380883,"score_spread":0.23151444529728105,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1561788061","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.38631043,0.016825363,0.5122096,0.0026906887,0.006130012,0.0009630228,0.02850136,0.015913118,0.030456388],"genre_scores_gemma":[0.40718132,0.0037344364,0.5437044,0.0006618622,0.00054911297,0.00021433592,0.033860736,0.0019560964,0.008137676],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9972625,0.00045969448,0.0006184851,0.0007557972,0.000751977,0.00015167358],"domain_scores_gemma":[0.9934875,0.0028489362,0.0006822426,0.00066790124,0.0019491415,0.00036418575],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014618308,0.0014182789,0.001992376,0.0105204955,0.0022418438,0.004066187,0.0014603045,0.0016513254,0.006823438],"category_scores_gemma":[0.008673685,0.0009431703,0.0013519382,0.008533072,0.000981242,0.00568169,0.0041430886,0.0017048481,0.005191961],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019385337,0.00050972146,0.020975634,0.004466492,0.0005543993,0.009238262,0.006383492,0.0028204864,0.17553543,0.038522817,0.08685992,0.6521948],"study_design_scores_gemma":[0.00056945,0.0009342155,0.043818284,0.0024568497,0.0018935137,0.026893517,0.015919164,0.10402392,0.19270499,0.13427685,0.4756446,0.00086469314],"about_ca_topic_score_codex":0.002536158,"about_ca_topic_score_gemma":0.0063407356,"teacher_disagreement_score":0.0105204955,"about_ca_system_score_codex":0.00071088475,"about_ca_system_score_gemma":0.0024251437,"threshold_uncertainty_score":0.022826731},"labels":[],"label_agreement":null},{"id":"W1564649749","doi":"10.1007/11424918_40","title":"Voting Between Multiple Data Representations for Text Chunking","year":2005,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":55,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Voting; Computer science; Phrase; Chunking (psychology); Identification (biology); Focus (optics); Artificial intelligence; Natural language processing; Set (abstract data type); Training set; Condorcet method","score_opus":0.051626273531095314,"score_gpt":0.32918940770071725,"score_spread":0.2775631341696219,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1564649749","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03453257,0.00095462566,0.9572948,0.00041292486,0.0003464568,0.00026784465,0.00072291994,0.0039525237,0.0015153204],"genre_scores_gemma":[0.36859682,0.00036571102,0.61971486,0.00027797936,0.00025468928,0.00042248322,0.004138767,0.0005524758,0.005676281],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99597067,0.0012303374,0.00049442786,0.0009672541,0.0009388177,0.000398552],"domain_scores_gemma":[0.99083066,0.004768164,0.00034164876,0.0021508595,0.0016131606,0.00029568945],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0052053714,0.0012047901,0.002270459,0.0027513718,0.000990405,0.0018909825,0.0030866528,0.0022351982,0.0050480547],"category_scores_gemma":[0.012611844,0.0007315093,0.0012922832,0.002581146,0.0007461154,0.0047154366,0.003164405,0.0021488501,0.0023508272],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010187313,0.00022872943,0.0021638824,0.0002510104,0.00018599474,0.00013089296,0.00017238225,0.028207406,0.019379808,0.007153439,0.014717565,0.92639023],"study_design_scores_gemma":[0.00007886834,0.00019634476,0.0009098617,0.00006595116,0.00013198542,0.00013352849,0.00016917213,0.9305814,0.029349877,0.03278021,0.0055635097,0.000039200586],"about_ca_topic_score_codex":0.0020518582,"about_ca_topic_score_gemma":0.0042187776,"teacher_disagreement_score":0.0052053714,"about_ca_system_score_codex":0.00076672446,"about_ca_system_score_gemma":0.0015033116,"threshold_uncertainty_score":0.027529001},"labels":[],"label_agreement":null},{"id":"W1565752952","doi":"10.1007/3-540-45486-1_11","title":"Collocation Discovery for Optimal Bilingual Lexicon Development","year":2000,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Lexicon; Collocation (remote sensing); Natural language processing; Artificial intelligence; Variety (cybernetics); Lexicography; Machine translation; Linguistics; Machine learning","score_opus":0.015497514024833485,"score_gpt":0.27019197902969894,"score_spread":0.2546944650048655,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1565752952","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021421345,0.0007104059,0.9523187,0.0004650359,0.00023168116,0.00028761505,0.0018387738,0.013921234,0.00880524],"genre_scores_gemma":[0.16827138,0.0005220936,0.81292784,0.00019241199,0.00013991157,0.00037798574,0.008730462,0.0023360997,0.0065017478],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99807465,0.00062329945,0.00024682895,0.00048987643,0.00033261089,0.00023261984],"domain_scores_gemma":[0.9960387,0.0018490923,0.0001314955,0.00059424946,0.0012535616,0.00013299695],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016268194,0.0013552022,0.0019003619,0.004508838,0.0017042551,0.0027142484,0.0020177537,0.0013881121,0.03151518],"category_scores_gemma":[0.008414167,0.0013541966,0.0014335548,0.0039912886,0.0008208401,0.005445015,0.0041769627,0.0016256983,0.014757891],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00047789465,0.00013760812,0.000995499,0.0006232984,0.00008586523,0.0004958764,0.00037635938,0.008845214,0.02006585,0.022966087,0.03213138,0.91279906],"study_design_scores_gemma":[0.0004029474,0.00026610473,0.0014081653,0.00021556129,0.00033103878,0.0016237014,0.001633942,0.723914,0.06576204,0.14754955,0.056730423,0.00016248655],"about_ca_topic_score_codex":0.004395818,"about_ca_topic_score_gemma":0.010231691,"teacher_disagreement_score":0.03151518,"about_ca_system_score_codex":0.000879011,"about_ca_system_score_gemma":0.0038156775,"threshold_uncertainty_score":0.105428755},"labels":[],"label_agreement":null},{"id":"W1566430846","doi":"10.1007/3-540-45632-5_19","title":"On Implicit Meanings","year":2002,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Parsing; Symbol (formal); Computation; Natural language processing; Contiguity; Word (group theory); Artificial intelligence; Blocking (statistics); Meaning (existential); State (computer science); Theoretical computer science; Algorithm; Programming language; Linguistics","score_opus":0.013202452917125313,"score_gpt":0.25065263438594015,"score_spread":0.23745018146881483,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1566430846","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017697306,0.004079468,0.47577858,0.009163362,0.0017810704,0.000105449726,0.00084896205,0.0007446219,0.48980123],"genre_scores_gemma":[0.6309874,0.005218803,0.1507547,0.0022425626,0.0023594727,0.0005306281,0.0027467504,0.0016858059,0.20347378],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99772376,0.00078621984,0.0001723453,0.00048152663,0.00062923867,0.00020701111],"domain_scores_gemma":[0.99718696,0.0012838839,0.00011070764,0.0009481883,0.0004033426,0.00006693698],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018592231,0.0011880492,0.0011504649,0.0020412218,0.0032845999,0.004577945,0.001958694,0.0021099357,0.036360856],"category_scores_gemma":[0.008628901,0.0011625866,0.0012155952,0.0027795858,0.0066650817,0.03000979,0.0069327755,0.005874699,0.006037707],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000069844073,0.0000035754806,0.000027792163,0.000026714746,0.0000023163593,0.0000147445635,0.00022531014,0.000081001344,0.00008656982,0.9915658,0.0017031399,0.006256077],"study_design_scores_gemma":[0.0000035868634,0.0000024525264,0.00002987621,0.00001806209,0.0000046984374,0.00002843539,0.000058330832,0.00045678028,0.00012046602,0.9857859,0.01348758,0.000003826949],"about_ca_topic_score_codex":0.00092368387,"about_ca_topic_score_gemma":0.0011124542,"teacher_disagreement_score":0.036360856,"about_ca_system_score_codex":0.0015461986,"about_ca_system_score_gemma":0.000783528,"threshold_uncertainty_score":0.12163919},"labels":[],"label_agreement":null},{"id":"W1567074120","doi":"10.1007/978-3-540-72665-4_43","title":"Rethinking the Semantics of Complex Nominals","year":2007,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Windsor","funders":"","keywords":"Computer science; Adjective; Natural language processing; Noun; Noun phrase; Artificial intelligence; Principle of compositionality; Semantics (computer science); Intersection (aeronautics); Head (geology); Set (abstract data type); Semantic property; Linguistics; Programming language; Philosophy","score_opus":0.04568903296887927,"score_gpt":0.3024490970901756,"score_spread":0.2567600641212963,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1567074120","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.030943992,0.0027796142,0.90100193,0.0070488215,0.0015485918,0.00007581416,0.00046369448,0.002426973,0.05371056],"genre_scores_gemma":[0.53510654,0.002773944,0.4324648,0.0022642906,0.0012971986,0.0001126606,0.0008819322,0.0032765884,0.021822039],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9966949,0.0011950536,0.0003087176,0.0006474853,0.00091750454,0.00023638348],"domain_scores_gemma":[0.99358577,0.0032909454,0.00025994048,0.0017542962,0.0009469759,0.00016202962],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0042680134,0.0010180149,0.0011348617,0.001744986,0.0023914932,0.009070316,0.0042061266,0.0026730301,0.009168665],"category_scores_gemma":[0.011989308,0.0014686112,0.001888067,0.0020024069,0.010575611,0.030085208,0.0070111346,0.0054566343,0.0021791614],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000032247262,0.000009950317,0.00012835771,0.00007847413,0.000010945118,0.00005941121,0.00089834834,0.0010533803,0.0009674923,0.97664136,0.0021827526,0.017937329],"study_design_scores_gemma":[0.000009637068,0.000008576366,0.000065033804,0.00003688461,0.000014582935,0.000068292,0.00021765604,0.004999531,0.0008400544,0.9653863,0.02833481,0.00001877378],"about_ca_topic_score_codex":0.003937159,"about_ca_topic_score_gemma":0.0055203494,"teacher_disagreement_score":0.009168665,"about_ca_system_score_codex":0.0027357605,"about_ca_system_score_gemma":0.0015593434,"threshold_uncertainty_score":0.030672193},"labels":[],"label_agreement":null},{"id":"W1567115194","doi":"10.3233/978-1-58603-891-5-50","title":"Extending the Knowledge Compilation Map: Closure Principles","year":2008,"lang":"en","type":"book-chapter","venue":"Frontiers in artificial intelligence and applications","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Agence Nationale de la Recherche; Public Health Agency of Canada","keywords":"Closure (psychology); Forgetting; Fragment (logic); Computer science; Existentialism; Algorithm; Theoretical computer science; Epistemology; Philosophy; Linguistics; Political science","score_opus":0.05580239362905733,"score_gpt":0.2987716418303622,"score_spread":0.24296924820130486,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1567115194","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018740635,0.00074988167,0.96470237,0.0005900552,0.00011500433,0.00011879854,0.000121575635,0.0008240498,0.014037667],"genre_scores_gemma":[0.31918174,0.0012850715,0.67015636,0.00039486782,0.00038384632,0.00022985446,0.00028198908,0.0005304538,0.0075557088],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9983966,0.00034534518,0.00008746287,0.00031857227,0.0006926221,0.0001594864],"domain_scores_gemma":[0.99640477,0.0020585714,0.00013921053,0.00083099626,0.00044940287,0.00011714166],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021694873,0.00058117165,0.0005546263,0.001350158,0.00073880376,0.0019688238,0.0015801258,0.00063025096,0.0029312773],"category_scores_gemma":[0.0073215477,0.0005412638,0.0012156587,0.0011438843,0.0023897158,0.0059757796,0.0037118471,0.0023391207,0.00063308736],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000138703,0.000121080484,0.00058786164,0.00042133752,0.00005793471,0.00066605984,0.001325695,0.018640481,0.0068651573,0.64672536,0.004546442,0.31990388],"study_design_scores_gemma":[0.00006050365,0.00013343067,0.0006421025,0.00012993044,0.000105572275,0.0010296451,0.0002719829,0.089796916,0.019051706,0.82665896,0.062058233,0.000061027924],"about_ca_topic_score_codex":0.0012928613,"about_ca_topic_score_gemma":0.00073252775,"teacher_disagreement_score":0.0029312773,"about_ca_system_score_codex":0.0007862194,"about_ca_system_score_gemma":0.00085146213,"threshold_uncertainty_score":0.011473477},"labels":[],"label_agreement":null},{"id":"W1568140279","doi":"","title":"Proceedings of the Workshop on Multiword Expressions: Identifying and Exploiting Underlying Properties","year":2006,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Presentation (obstetrics); Lexicon; Focus (optics); Automatic summarization; Principle of compositionality; Question answering; Sentence; FrameNet; Representation (politics); Parsing; Computational linguistics; Linguistics","score_opus":0.06285309301651662,"score_gpt":0.28990418880356605,"score_spread":0.22705109578704943,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1568140279","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025442977,0.05984762,0.57887715,0.062104996,0.07329112,0.0013763795,0.008134135,0.007113834,0.18381183],"genre_scores_gemma":[0.07433349,0.03515638,0.38094303,0.007823139,0.014541897,0.0011245628,0.026387453,0.0068167774,0.45287317],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99632394,0.00116909,0.0004038811,0.00080412364,0.0010743932,0.00022451987],"domain_scores_gemma":[0.99359447,0.0024951447,0.00021724332,0.0008378405,0.0022576672,0.00059768336],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0042117126,0.0011545292,0.0015170954,0.0018294157,0.001567932,0.00838843,0.0020325622,0.0023398264,0.04410403],"category_scores_gemma":[0.008920172,0.00062708586,0.0016634926,0.0023334357,0.0013668454,0.0090126535,0.0035233097,0.0033239725,0.024625659],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00034673788,0.00017120213,0.0007585296,0.0011150964,0.0000864149,0.0004767865,0.0018254169,0.0010493576,0.010541314,0.0274855,0.4935588,0.46258497],"study_design_scores_gemma":[0.00002341748,0.00005568575,0.0008765073,0.00031420786,0.00004607718,0.00043987687,0.0006518226,0.0023045219,0.0032265235,0.019726016,0.9722945,0.00004081566],"about_ca_topic_score_codex":0.0011441534,"about_ca_topic_score_gemma":0.0018431805,"teacher_disagreement_score":0.04410403,"about_ca_system_score_codex":0.0013639332,"about_ca_system_score_gemma":0.0023457645,"threshold_uncertainty_score":0.14754272},"labels":[],"label_agreement":null},{"id":"W1568948840","doi":"10.1007/978-3-642-13059-5_29","title":"Word Reordering Approaches for Bangla-English Statistical Machine Translation","year":2010,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Bengali; Computer science; Natural language processing; Phrase; Machine translation; Artificial intelligence; Word (group theory); Evaluation of machine translation; Set (abstract data type); BLEU; Test set; Translation (biology); Speech recognition; Machine translation software usability; Example-based machine translation; Linguistics; Programming language","score_opus":0.02836534934894377,"score_gpt":0.2629976114513641,"score_spread":0.2346322621024203,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1568948840","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015671855,0.0039173807,0.9609963,0.00062924816,0.000642402,0.00019470058,0.001691721,0.0060984516,0.010157991],"genre_scores_gemma":[0.15997566,0.0036468513,0.807627,0.00036209376,0.00041594682,0.0002705868,0.0076962644,0.002440921,0.017564615],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9984988,0.0006302316,0.00023524486,0.000237139,0.00027979398,0.00011876409],"domain_scores_gemma":[0.99762374,0.0010026764,0.00014381157,0.00049925776,0.0006752857,0.000055281795],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014041502,0.0011163849,0.0009951054,0.0019386671,0.0011527593,0.002524531,0.0011533524,0.00075014634,0.00867811],"category_scores_gemma":[0.0030413407,0.00067428243,0.0011948592,0.0037894282,0.0005006837,0.0021100752,0.0014490804,0.0015638974,0.008367972],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029551474,0.00017360903,0.0009772469,0.00074934226,0.00020238502,0.00048474877,0.00070052576,0.018285293,0.0320001,0.030244103,0.028657679,0.88722944],"study_design_scores_gemma":[0.00015795894,0.00044648076,0.0036443598,0.00025121693,0.00046383525,0.001577646,0.0018683453,0.5857523,0.0886832,0.17066301,0.14628386,0.00020777575],"about_ca_topic_score_codex":0.0038712383,"about_ca_topic_score_gemma":0.010687216,"teacher_disagreement_score":0.00867811,"about_ca_system_score_codex":0.00070987875,"about_ca_system_score_gemma":0.0016693559,"threshold_uncertainty_score":0.029031157},"labels":[],"label_agreement":null},{"id":"W1569248516","doi":"10.1007/978-3-540-30194-3_18","title":"Weather Report Translation Using a Translation Memory","year":2004,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Translation (biology); Machine translation; Landmark; Transfer-based machine translation; Context (archaeology); Natural language processing; Artificial intelligence; Example-based machine translation; History","score_opus":0.029810146427275832,"score_gpt":0.284141609649421,"score_spread":0.25433146322214517,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1569248516","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0354859,0.0011343686,0.71856785,0.0017475039,0.0038972048,0.00058649713,0.01079826,0.085031666,0.14275065],"genre_scores_gemma":[0.32499394,0.0015950314,0.49565282,0.00095353695,0.0013087466,0.00043153338,0.021453416,0.014621443,0.1389896],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99920136,0.000196905,0.00012762196,0.00024322387,0.00016091928,0.00007006804],"domain_scores_gemma":[0.99669373,0.00083664333,0.00015272267,0.0013961693,0.00085492345,0.000065877415],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009527779,0.001040004,0.00079810683,0.0020271835,0.0009390595,0.0029753805,0.0011074581,0.000897103,0.06535399],"category_scores_gemma":[0.004344467,0.00071476505,0.0008475726,0.0027417436,0.0005311797,0.003745077,0.0018186026,0.0010404167,0.036297034],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011190956,0.00023929689,0.0010476218,0.00067562255,0.0001170099,0.0010721495,0.0011973007,0.004856297,0.0316822,0.048459522,0.17242381,0.73711],"study_design_scores_gemma":[0.00032037412,0.0004264034,0.0015880056,0.0002555739,0.00045882852,0.0013407484,0.0012327346,0.08428544,0.13820142,0.06140011,0.7103065,0.00018388942],"about_ca_topic_score_codex":0.0018536751,"about_ca_topic_score_gemma":0.0015030669,"teacher_disagreement_score":0.06535399,"about_ca_system_score_codex":0.00045638072,"about_ca_system_score_gemma":0.000988693,"threshold_uncertainty_score":0.21863085},"labels":[],"label_agreement":null},{"id":"W1569397218","doi":"","title":"Utilizing Extra-Sentential Context for Parsing","year":2010,"lang":"en","type":"article","venue":"Empirical Methods in Natural Language Processing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Treebank; Parsing; Computer science; Natural language processing; Artificial intelligence; Context (archaeology); Consistency (knowledge bases); Conditional random field; Generative grammar; Set (abstract data type); Feature (linguistics); Linguistics; Programming language","score_opus":0.05448447238682705,"score_gpt":0.44656764896122564,"score_spread":0.3920831765743986,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1569397218","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20882133,0.00087209174,0.769417,0.0009804724,0.00014109722,0.00017358697,0.0018626169,0.010209896,0.007521971],"genre_scores_gemma":[0.7508827,0.00026483418,0.24395072,0.00018286992,0.00008836942,0.000090967595,0.0019842044,0.0011454141,0.0014099748],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9971641,0.0012878877,0.00014705538,0.00072590413,0.0005401171,0.00013492424],"domain_scores_gemma":[0.99041325,0.0059827724,0.0005411411,0.002054213,0.0008802782,0.00012838036],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036075686,0.0010105975,0.00087955594,0.0023878135,0.0012524108,0.0025242115,0.0011939825,0.0011745576,0.0029562495],"category_scores_gemma":[0.016475491,0.0010100858,0.00080567476,0.0021186522,0.0011072704,0.005743702,0.0020362972,0.002147874,0.0011275724],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007121082,0.00042036627,0.030850729,0.0006706361,0.00030370595,0.0010927769,0.0021790075,0.105183855,0.073343396,0.06751309,0.017106965,0.70062333],"study_design_scores_gemma":[0.00007630782,0.00016272637,0.013058858,0.000101908794,0.00030141987,0.000717827,0.00034100015,0.78410006,0.05567835,0.1249592,0.02026481,0.00023750907],"about_ca_topic_score_codex":0.004363167,"about_ca_topic_score_gemma":0.013463791,"teacher_disagreement_score":0.004363167,"about_ca_system_score_codex":0.0009030251,"about_ca_system_score_gemma":0.0019991563,"threshold_uncertainty_score":0.01907891},"labels":[],"label_agreement":null},{"id":"W1569927945","doi":"10.1007/3-540-47922-8_27","title":"Extraction of Text Phrases Using Hierarchical Grammar","year":2002,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Bigram; Computer science; Artificial intelligence; Natural language processing; Classifier (UML); Grammar; Pattern recognition (psychology); Linguistics","score_opus":0.024417248599950623,"score_gpt":0.2858280265384344,"score_spread":0.2614107779384838,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1569927945","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025974026,0.0020370889,0.90841603,0.0008763616,0.00044566474,0.0011771498,0.017737523,0.033564847,0.0097712455],"genre_scores_gemma":[0.07091778,0.0013862222,0.8714584,0.00026703114,0.00020835016,0.00055404083,0.044902302,0.0032573738,0.0070484327],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99914324,0.00015399726,0.0001610363,0.00021246434,0.00024062717,0.00008859501],"domain_scores_gemma":[0.9980428,0.0009505818,0.00016736238,0.00019101758,0.00057713845,0.00007102809],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006006835,0.0019638399,0.0019687917,0.005692181,0.0011675623,0.0019961055,0.0015680119,0.0011807582,0.01347843],"category_scores_gemma":[0.002752107,0.001216037,0.002478295,0.0061626486,0.00072888524,0.002916102,0.001926158,0.0016319164,0.014383343],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003043038,0.00014326176,0.0019940855,0.0026693866,0.0001910256,0.0022083274,0.0010863452,0.0042436775,0.096540764,0.018170532,0.061929226,0.81051904],"study_design_scores_gemma":[0.00049030344,0.0007434549,0.01091171,0.001288452,0.0015348726,0.005137413,0.003101804,0.22647637,0.24688414,0.14763492,0.35541022,0.00038638338],"about_ca_topic_score_codex":0.0025464334,"about_ca_topic_score_gemma":0.003642737,"teacher_disagreement_score":0.01347843,"about_ca_system_score_codex":0.00065691414,"about_ca_system_score_gemma":0.0022345018,"threshold_uncertainty_score":0.04508984},"labels":[],"label_agreement":null},{"id":"W157143106","doi":"","title":"A Hierarchical EM Approach to Word Segmentation.","year":2001,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Morpheme; Word (group theory); Artificial intelligence; Computer science; Segmentation; Lexicon; Text segmentation; Natural language processing; Character (mathematics); Speech recognition; Minimum description length; Hidden Markov model; Pattern recognition (psychology); Mathematics","score_opus":0.018210886773809647,"score_gpt":0.28255137740568814,"score_spread":0.2643404906318785,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W157143106","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0012594324,0.00010405087,0.9967199,0.00010613194,0.000019556453,0.00003094474,0.000091823786,0.00072412455,0.00094402494],"genre_scores_gemma":[0.09285357,0.00025729058,0.8971863,0.0003998051,0.000099097524,0.00034419762,0.0013706214,0.0007157007,0.0067734113],"study_design_codex":"simulation_or_modeling","study_design_gemma":"not_applicable","domain_scores_codex":[0.9989064,0.00043224063,0.000060473692,0.00033732716,0.0001868357,0.000076713426],"domain_scores_gemma":[0.9983833,0.0009628677,0.00010732621,0.0002490873,0.00023527554,0.000062108775],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016590883,0.0009923606,0.0011641058,0.0014580223,0.0006662875,0.0014649126,0.0024377538,0.0015152524,0.006090284],"category_scores_gemma":[0.0059259324,0.0010074414,0.0015915296,0.0015199526,0.0010016179,0.00246356,0.001990527,0.0017304065,0.004615468],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001417644,0.00011415546,0.0016521803,0.00030619794,0.00030433215,0.00017975265,0.00044762797,0.46611384,0.008302548,0.08409807,0.01409667,0.42424282],"study_design_scores_gemma":[0.000012156162,0.000016182736,0.00021808392,0.000011052245,0.000014713963,0.000046831392,0.000022574528,0.9524277,0.0011723161,0.042726737,0.0033183692,0.000013326978],"about_ca_topic_score_codex":0.005048309,"about_ca_topic_score_gemma":0.00860523,"teacher_disagreement_score":0.006090284,"about_ca_system_score_codex":0.0011559487,"about_ca_system_score_gemma":0.0015907586,"threshold_uncertainty_score":0.020374},"labels":[],"label_agreement":null},{"id":"W1571803718","doi":"10.1007/11671299_9","title":"Automatically Determining Allowable Combinations of a Class of Flexible Multiword Expressions","year":2006,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Measure (data warehouse); Class (philosophy); Collocation (remote sensing); Natural language processing; Simple (philosophy); Artificial intelligence; Noun; Machine learning; Data mining","score_opus":0.01519254296208463,"score_gpt":0.27374078839029903,"score_spread":0.2585482454282144,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1571803718","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.31558606,0.0009078958,0.6453988,0.0006536074,0.00025474408,0.00048051303,0.005097212,0.014615454,0.017005742],"genre_scores_gemma":[0.51948637,0.00035969884,0.46357092,0.0002190418,0.000113836475,0.0003710081,0.0076303887,0.0033002628,0.004948497],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99773717,0.00035609095,0.00029957053,0.00083744497,0.0005324383,0.00023723487],"domain_scores_gemma":[0.99445254,0.003432993,0.00040689026,0.00062564475,0.00085297256,0.00022902818],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001157923,0.0015585879,0.0012587917,0.0027873327,0.0015571624,0.0021419944,0.001823065,0.0015947432,0.008983289],"category_scores_gemma":[0.0061357445,0.0013825762,0.0017951893,0.0020082921,0.00096528075,0.0053605903,0.002617142,0.0021430005,0.002755495],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015652353,0.00045448038,0.014860729,0.0014187142,0.0003380178,0.0052213413,0.0018022467,0.0105659785,0.17102955,0.06507118,0.02366299,0.7040095],"study_design_scores_gemma":[0.00024670572,0.0007251834,0.012027149,0.000618289,0.0007536103,0.009747488,0.0039775507,0.4159729,0.19367313,0.2690987,0.092737146,0.00042223476],"about_ca_topic_score_codex":0.0011472098,"about_ca_topic_score_gemma":0.002374111,"teacher_disagreement_score":0.008983289,"about_ca_system_score_codex":0.00052001217,"about_ca_system_score_gemma":0.0010876625,"threshold_uncertainty_score":0.030052066},"labels":[],"label_agreement":null},{"id":"W1571972440","doi":"","title":"A Symbolic Summarizer for the Update Task of TAC 2008","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Automatic summarization; Computer science; Relevance (law); Natural language processing; Task (project management); Parsing; Similarity (geometry); Artificial intelligence; Feature (linguistics); Quality (philosophy); Information retrieval; Linguistics","score_opus":0.01563820452331472,"score_gpt":0.25835422809546105,"score_spread":0.24271602357214633,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1571972440","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07566041,0.0044849967,0.4714308,0.0023251383,0.0018827481,0.001510975,0.06986383,0.3434872,0.029353822],"genre_scores_gemma":[0.2015274,0.0010203522,0.5213486,0.0006504401,0.00092056923,0.0013106931,0.23124117,0.0065662973,0.03541444],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989273,0.00030930314,0.00016109347,0.00018135906,0.00034736376,0.00007358514],"domain_scores_gemma":[0.99745864,0.00076992094,0.00018181922,0.0004895813,0.0009699115,0.00013015626],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017770809,0.001108199,0.0010714541,0.0022325092,0.0007764783,0.0013241697,0.0013433549,0.0009578929,0.0150498655],"category_scores_gemma":[0.0061729606,0.00038726325,0.0005002531,0.0015644096,0.00017483167,0.0019600163,0.00085885683,0.0012929499,0.009357317],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00067314226,0.00025831448,0.0011655618,0.0007869639,0.00016741134,0.0003052632,0.00039164862,0.0031510394,0.027159216,0.0018810622,0.40026805,0.56379235],"study_design_scores_gemma":[0.0009437846,0.0019120114,0.012552372,0.00017353323,0.00068516925,0.0010901401,0.0005008438,0.30059704,0.10385202,0.009308865,0.56810856,0.0002757147],"about_ca_topic_score_codex":0.0035973438,"about_ca_topic_score_gemma":0.010213647,"teacher_disagreement_score":0.0150498655,"about_ca_system_score_codex":0.0005910142,"about_ca_system_score_gemma":0.00093570445,"threshold_uncertainty_score":0.05034685},"labels":[],"label_agreement":null},{"id":"W157259314","doi":"","title":"Integrating Joint n-gram Features into a Discriminative Training Framework","year":2010,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":58,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Discriminative model; Transliteration; Computer science; Context (archaeology); Artificial intelligence; Joint (building); Generative grammar; n-gram; Transduction (biophysics); Feature (linguistics); String (physics); Context model; Natural language processing; Speech recognition; Machine learning; Pattern recognition (psychology); Language model; Engineering; Mathematics; Linguistics","score_opus":0.018778259797715628,"score_gpt":0.2974458039104205,"score_spread":0.2786675441127049,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W157259314","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025425447,0.00023242502,0.96507674,0.00012327611,0.000046162284,0.000045699162,0.00032647836,0.0071147494,0.0016089964],"genre_scores_gemma":[0.44174296,0.00025242858,0.54732275,0.00024355175,0.000085622894,0.00017459606,0.0039577032,0.0005154289,0.0057050474],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99920875,0.0002686059,0.00004009748,0.0002582392,0.00013739028,0.00008700487],"domain_scores_gemma":[0.998681,0.00059330894,0.0000799964,0.00036049658,0.00021408146,0.00007115257],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001165253,0.001407695,0.0011283422,0.001068408,0.0005195408,0.0006976294,0.001582828,0.0010614586,0.0031053587],"category_scores_gemma":[0.0033705877,0.00047427163,0.0007772445,0.001371622,0.00045750279,0.0021672258,0.0012407358,0.0015341815,0.0034573511],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00033736,0.00038711834,0.0035099147,0.00013630133,0.0001057083,0.00019501486,0.00009128316,0.13724454,0.02976037,0.0052399817,0.007811803,0.81518054],"study_design_scores_gemma":[0.000011241975,0.00007911683,0.0006125023,0.0000071819272,0.00002285908,0.00010893358,0.000017730912,0.9863637,0.0074212896,0.003653728,0.0016857641,0.00001591007],"about_ca_topic_score_codex":0.005215769,"about_ca_topic_score_gemma":0.013936433,"teacher_disagreement_score":0.005215769,"about_ca_system_score_codex":0.0004548736,"about_ca_system_score_gemma":0.0008956822,"threshold_uncertainty_score":0.010388494},"labels":[],"label_agreement":null},{"id":"W1574548479","doi":"10.1007/3-540-45691-0_11","title":"Using Statistical Translation Models for Bilingual IR","year":2002,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":28,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Machine translation; Natural language processing; Clef; Artificial intelligence; Translation (biology); Set (abstract data type); Machine translation software usability; Example-based machine translation; Quality (philosophy); Information retrieval; Programming language","score_opus":0.06355541360915397,"score_gpt":0.31854023540942916,"score_spread":0.2549848218002752,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1574548479","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005924042,0.00084751105,0.9828607,0.00032163967,0.00019514533,0.00006948221,0.00035384615,0.005076033,0.0043515917],"genre_scores_gemma":[0.1987424,0.001381904,0.78320456,0.0003241936,0.0004309046,0.00042133074,0.0037834626,0.0027174663,0.008993796],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9963785,0.002435336,0.00017471015,0.00044003376,0.00042732715,0.00014412055],"domain_scores_gemma":[0.9942654,0.0037050543,0.0002375969,0.0010354278,0.0006739373,0.0000826121],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0041709812,0.0014167365,0.0014548226,0.0018102507,0.001118656,0.0027172924,0.0014231855,0.0016074766,0.010370682],"category_scores_gemma":[0.011556537,0.0011725834,0.0018637788,0.0027964835,0.0006748289,0.0051723686,0.0020430884,0.002144911,0.008628183],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008913658,0.0003137057,0.00097292266,0.0007701463,0.00043576537,0.00034517076,0.00037994212,0.0863739,0.014390795,0.06742334,0.02893366,0.79876924],"study_design_scores_gemma":[0.00010656147,0.00013656313,0.00033948303,0.00006785343,0.0001313083,0.0002306074,0.00011636334,0.8272452,0.010122163,0.15063697,0.01079776,0.00006919568],"about_ca_topic_score_codex":0.0015675499,"about_ca_topic_score_gemma":0.0033420888,"teacher_disagreement_score":0.010370682,"about_ca_system_score_codex":0.00072550785,"about_ca_system_score_gemma":0.00118966,"threshold_uncertainty_score":0.03469336},"labels":[],"label_agreement":null},{"id":"W1577070617","doi":"10.21236/ada456027","title":"TREC-8 Experiments at Maryland: CLIR, QA and Routing","year":2000,"lang":"en","type":"report","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba","funders":"","keywords":"Computer science; Information retrieval; Parsing; Task (project management); Search engine indexing; Natural language processing; Question answering; Artificial intelligence; Matching (statistics); Routing (electronic design automation)","score_opus":0.030666650228793415,"score_gpt":0.3270000880792282,"score_spread":0.2963334378504348,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1577070617","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7087831,0.025096387,0.024521574,0.009396402,0.0052791033,0.009032661,0.052981,0.028655535,0.13625434],"genre_scores_gemma":[0.7278524,0.0024217265,0.089239866,0.0045531765,0.0010854434,0.0044019464,0.10334552,0.0019781387,0.065121844],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9861508,0.007943354,0.0010932711,0.0020277025,0.0017996617,0.000985104],"domain_scores_gemma":[0.9736984,0.016768005,0.0006723587,0.003403865,0.0034931696,0.0019641763],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.017361814,0.002500996,0.0025182685,0.0025873862,0.0049294387,0.0037330214,0.0035673126,0.0047135213,0.011721754],"category_scores_gemma":[0.030195616,0.0013611452,0.0013719277,0.0034514023,0.0016553507,0.0068517555,0.0030642017,0.004457493,0.0060810745],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.017342225,0.012563499,0.009273373,0.004032305,0.001011602,0.00091394724,0.0021138217,0.03236195,0.023742052,0.008604113,0.63170713,0.25633404],"study_design_scores_gemma":[0.027870784,0.022762503,0.084128514,0.0011016219,0.0018957851,0.0030514232,0.006493526,0.41784796,0.07829726,0.022339758,0.33254325,0.0016675673],"about_ca_topic_score_codex":0.09604375,"about_ca_topic_score_gemma":0.08903878,"teacher_disagreement_score":0.09604375,"about_ca_system_score_codex":0.0045635374,"about_ca_system_score_gemma":0.0036664633,"threshold_uncertainty_score":0.19096947},"labels":[],"label_agreement":null},{"id":"W1577122732","doi":"10.1007/3-540-45486-1_12","title":"The Power of the TSNLP: Lessons from a Diagnostic Evaluation of a Broad-Coverage Parser","year":2000,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Suite; Parsing; Test suite; Natural language processing; Artificial intelligence; Test (biology); Programming language; Test case; Machine learning","score_opus":0.018868433627276485,"score_gpt":0.2853218196749292,"score_spread":0.26645338604765273,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1577122732","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.36915615,0.0039808215,0.4728022,0.036854435,0.0011244363,0.00053828466,0.006170626,0.037469853,0.071903184],"genre_scores_gemma":[0.7659086,0.00082738866,0.21163243,0.0031605025,0.00033744128,0.000115243776,0.0027989342,0.00830852,0.00691086],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9776398,0.011904741,0.0011951526,0.0018180436,0.0067666536,0.00067556533],"domain_scores_gemma":[0.7917615,0.17584802,0.0029799615,0.014256816,0.013691171,0.0014625686],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.019316204,0.0013208123,0.0010419864,0.0029544074,0.0015753516,0.006705353,0.0043596798,0.0035990756,0.008141326],"category_scores_gemma":[0.15913358,0.0010600691,0.00074826967,0.0031884038,0.0047902176,0.017283436,0.0044361884,0.0044805678,0.002743404],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0029224497,0.00087943673,0.017406804,0.0016706827,0.00022984971,0.0028557654,0.012644672,0.032021213,0.027592089,0.14581321,0.10794761,0.6480163],"study_design_scores_gemma":[0.0007950081,0.00089225045,0.0064186854,0.00076838676,0.0005833143,0.0037652447,0.0067740544,0.43788216,0.12902527,0.29659516,0.116058655,0.00044194254],"about_ca_topic_score_codex":0.013456218,"about_ca_topic_score_gemma":0.011145002,"teacher_disagreement_score":0.019316204,"about_ca_system_score_codex":0.0035268608,"about_ca_system_score_gemma":0.0042231972,"threshold_uncertainty_score":0.10215509},"labels":[],"label_agreement":null},{"id":"W1578279521","doi":"10.1007/3-540-45820-4_5","title":"Text Prediction with Fuzzy Alignments","year":2002,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Simple (philosophy); Machine translation; Translation (biology); Fuzzy logic; Artificial intelligence; Relation (database); Productivity; Machine learning; Data mining; Natural language processing","score_opus":0.011356779240361856,"score_gpt":0.23144922996641834,"score_spread":0.2200924507260565,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1578279521","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09589299,0.0017379447,0.86548644,0.00083862373,0.00041476617,0.00033838316,0.004312467,0.016947921,0.014030436],"genre_scores_gemma":[0.42141965,0.00069097587,0.5503241,0.00021830198,0.00030646313,0.0002637399,0.011311548,0.0006459537,0.014819316],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992372,0.00012589482,0.00005940304,0.0003102264,0.00019220484,0.00007503027],"domain_scores_gemma":[0.9981384,0.0008708894,0.00012225084,0.0003304332,0.0004620845,0.000075991295],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009230421,0.0008174727,0.00092804217,0.0026096832,0.00087697135,0.0012248199,0.0011864654,0.0010295523,0.011100751],"category_scores_gemma":[0.003988969,0.00043392714,0.00084324635,0.0024557563,0.00040832948,0.0024057627,0.0011851257,0.0010275807,0.007449719],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00084014673,0.00017227909,0.0032294248,0.00021979718,0.00010196088,0.00019554274,0.00008864523,0.02232752,0.015818058,0.006941752,0.019888898,0.930176],"study_design_scores_gemma":[0.00007637408,0.0001994473,0.0026250102,0.00007281796,0.000109000925,0.00024703212,0.00014127443,0.91337425,0.030293722,0.042716276,0.010109291,0.000035545643],"about_ca_topic_score_codex":0.0032434345,"about_ca_topic_score_gemma":0.004896427,"teacher_disagreement_score":0.011100751,"about_ca_system_score_codex":0.00039761674,"about_ca_system_score_gemma":0.0007775541,"threshold_uncertainty_score":0.03713566},"labels":[],"label_agreement":null},{"id":"W1578310937","doi":"10.1007/978-3-319-10106-4_7","title":"Indeterminate Pronouns: The View from Japanese","year":2017,"lang":"en","type":"book-chapter","venue":"Studies in natural language and linguistic theory","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":443,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Indeterminate; Linguistics; Philosophy; Mathematics; Pure mathematics","score_opus":0.01968714651575379,"score_gpt":0.3135701793963078,"score_spread":0.29388303288055406,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1578310937","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.086954,0.11873348,0.06704908,0.05677607,0.0032895994,0.00004366073,0.000771463,0.0003514529,0.6660313],"genre_scores_gemma":[0.8353371,0.043869033,0.024421496,0.007229367,0.0040783817,0.00007875206,0.00052029477,0.0004563019,0.08400927],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.999358,0.0001777185,0.00004741332,0.00020124456,0.00014142346,0.00007424458],"domain_scores_gemma":[0.99902093,0.0004578686,0.00008789094,0.00012818973,0.00021393958,0.00009121883],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009178184,0.0006046483,0.00062564865,0.0023021202,0.004188353,0.006218386,0.00119757,0.0018937826,0.0057371785],"category_scores_gemma":[0.0022663309,0.00051627134,0.00042529346,0.003642235,0.008588494,0.012357381,0.001980349,0.0033766702,0.00076951],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00002314451,0.000008562449,0.00045423422,0.00014518529,0.0000065325253,0.0001582923,0.0072106463,0.0000901906,0.00040369097,0.9665776,0.008716594,0.016205352],"study_design_scores_gemma":[0.0000140851225,0.000021193533,0.0035819542,0.00019354794,0.000055097014,0.0005480687,0.006704514,0.0011036552,0.00060638145,0.7306748,0.2564575,0.00003921552],"about_ca_topic_score_codex":0.016175553,"about_ca_topic_score_gemma":0.022272632,"teacher_disagreement_score":0.016175553,"about_ca_system_score_codex":0.0017738146,"about_ca_system_score_gemma":0.0016864224,"threshold_uncertainty_score":0.032162845},"labels":[],"label_agreement":null},{"id":"W1579920339","doi":"10.1007/978-3-540-24840-8_50","title":"An Algorithm for Anaphora Resolution in Aviation Safety Reports","year":2004,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Anaphora (linguistics); Aviation; Resolution (logic); Computer science; Aviation safety; Algorithm; Engineering; Artificial intelligence; Aerospace engineering","score_opus":0.010734819343064098,"score_gpt":0.2697041559693742,"score_spread":0.2589693366263101,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1579920339","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0077831554,0.00029332953,0.98053163,0.00037839738,0.000095212534,0.00031751802,0.00044113857,0.006650356,0.0035092437],"genre_scores_gemma":[0.03433599,0.00012384783,0.9608998,0.000092793845,0.000041111256,0.00016923665,0.0007973027,0.00030923614,0.0032308025],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9981224,0.00041458855,0.0002455836,0.00048124974,0.0005850498,0.00015113209],"domain_scores_gemma":[0.9968645,0.0017139163,0.00014317385,0.00039711292,0.0008089411,0.00007235023],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023077882,0.0011096372,0.0015538578,0.0035496356,0.0023484218,0.005032514,0.0028179865,0.0030629407,0.018096497],"category_scores_gemma":[0.0069045178,0.0010337392,0.0014882081,0.0031562587,0.0009277452,0.0050233873,0.0027647507,0.001746249,0.006592017],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005140981,0.00023108286,0.00084784016,0.00035642422,0.00008465208,0.00012558463,0.00033552258,0.014131155,0.009366044,0.030607175,0.0172578,0.9261427],"study_design_scores_gemma":[0.00039237193,0.0001842776,0.0010214755,0.00014572554,0.00025703953,0.00060903165,0.000510103,0.82028425,0.032807805,0.10685967,0.03684357,0.00008474939],"about_ca_topic_score_codex":0.0031497707,"about_ca_topic_score_gemma":0.004154551,"teacher_disagreement_score":0.018096497,"about_ca_system_score_codex":0.0015575456,"about_ca_system_score_gemma":0.0025139004,"threshold_uncertainty_score":0.06053883},"labels":[],"label_agreement":null},{"id":"W1580822516","doi":"","title":"Imposing Hierarchical Browsing Structures onto Spoken Documents","year":2010,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"University of Toronto; National Research Council Canada","funders":"","keywords":"Computer science; Baseline (sea); Dynamic programming; Hierarchical database model; Scheme (mathematics); Hierarchical organization; Artificial intelligence; Data mining; Algorithm; Mathematics","score_opus":0.006584873957553026,"score_gpt":0.2714422880427332,"score_spread":0.26485741408518015,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1580822516","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08424263,0.0003054313,0.906381,0.0001966737,0.000039646307,0.000111628884,0.00028708408,0.005314879,0.0031209528],"genre_scores_gemma":[0.43524975,0.00050611515,0.5530049,0.0001544658,0.00010429486,0.00023195472,0.0022460222,0.0008600593,0.0076424535],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99718755,0.0010804199,0.00013470036,0.00069999596,0.00069033407,0.00020698155],"domain_scores_gemma":[0.9897475,0.0060968716,0.0007143109,0.0023684932,0.00079550163,0.0002773095],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026226367,0.0009937586,0.0012331066,0.0008146868,0.0011456513,0.0018139845,0.0017619377,0.0015328155,0.0050011408],"category_scores_gemma":[0.013913281,0.000994947,0.0009214849,0.0015378073,0.001249115,0.0054898653,0.0030501192,0.0027556117,0.0021671134],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010453638,0.0005004663,0.0046224785,0.00096849963,0.00012247215,0.0009264917,0.0028955508,0.12419027,0.10315354,0.03129497,0.008023974,0.7222559],"study_design_scores_gemma":[0.00012044072,0.00060805073,0.0035537356,0.00014224066,0.00012540395,0.00096232287,0.0017972335,0.8557862,0.075513765,0.044962324,0.016286371,0.00014183795],"about_ca_topic_score_codex":0.004820071,"about_ca_topic_score_gemma":0.01030634,"teacher_disagreement_score":0.0050011408,"about_ca_system_score_codex":0.0005622805,"about_ca_system_score_gemma":0.0018762641,"threshold_uncertainty_score":0.016730428},"labels":[],"label_agreement":null},{"id":"W1582613544","doi":"10.1007/978-3-642-13193-6_13","title":"Fast FPT Algorithms for Computing Rooted Agreement Forests: Theory and Experiments","year":2010,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":53,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Intuition; Algorithm; Phylogenetic tree; Computer science; Branching (polymer chemistry); Tree (set theory); Combinatorics; Running time; Mathematics; Chemistry","score_opus":0.016008449975772172,"score_gpt":0.29301237441610245,"score_spread":0.2770039244403303,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1582613544","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.32882142,0.0094005605,0.58189416,0.003474923,0.0019270622,0.0008426591,0.006153489,0.025164513,0.042321138],"genre_scores_gemma":[0.47746676,0.0013246787,0.50595546,0.00047218663,0.00035679722,0.00045010753,0.0054362654,0.0020504103,0.0064873183],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9918339,0.0019996068,0.0005402026,0.0016707341,0.0028031038,0.0011524357],"domain_scores_gemma":[0.9312489,0.04920665,0.0015916472,0.010653274,0.005656128,0.0016434586],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008767871,0.002297585,0.0026631055,0.00271687,0.0034836626,0.0045333914,0.0045227087,0.0036056307,0.01574115],"category_scores_gemma":[0.034460865,0.00096395635,0.0025138024,0.005654799,0.0022353758,0.015713941,0.0031141883,0.0050654123,0.0039902823],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.009425295,0.003076727,0.0053397524,0.0017167847,0.0003990401,0.00018555833,0.00080140744,0.1828876,0.011751732,0.07520239,0.07401741,0.63519627],"study_design_scores_gemma":[0.0014797312,0.00082864054,0.0012448963,0.00013035147,0.00023441768,0.0003058561,0.00051147485,0.7991072,0.013581995,0.17612782,0.0063629034,0.0000846781],"about_ca_topic_score_codex":0.010272369,"about_ca_topic_score_gemma":0.01024236,"teacher_disagreement_score":0.01574115,"about_ca_system_score_codex":0.00575572,"about_ca_system_score_gemma":0.0063741803,"threshold_uncertainty_score":0.052659333},"labels":[],"label_agreement":null},{"id":"W1583311084","doi":"10.1017/cbo9780511551789.008","title":"Comment clauses with <i>look</i>","year":2008,"lang":"en","type":"book-chapter","venue":"Cambridge University Press eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; History","score_opus":0.017594050018705808,"score_gpt":0.1991997017478804,"score_spread":0.1816056517291746,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1583311084","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0010358879,0.0010987818,0.012080055,0.014224501,0.015419831,0.00039007128,0.0087079145,0.0050557484,0.9419871],"genre_scores_gemma":[0.011060557,0.00091138063,0.004825689,0.008056014,0.0030918233,0.000265641,0.0062278565,0.006528851,0.95903224],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99935323,0.00013926861,0.000050130195,0.000084734434,0.0003001382,0.00007246905],"domain_scores_gemma":[0.99681026,0.0013434496,0.00017763113,0.00031050824,0.0012317434,0.00012628168],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0010291943,0.001250192,0.00070371124,0.001347334,0.0019502948,0.0029570973,0.0014600286,0.002353242,0.6106415],"category_scores_gemma":[0.006636852,0.00060482294,0.00081602664,0.0016616904,0.0010485497,0.006310279,0.0020336378,0.0027831977,0.35181347],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003614879,0.000008985978,0.000049232403,0.0001687385,0.0000022790682,0.000064193395,0.00020234624,0.000019405064,0.0005762885,0.015086114,0.97157574,0.012210595],"study_design_scores_gemma":[0.0000064079595,0.0000046074974,0.00008361202,0.00007296269,0.0000022222853,0.000052241397,0.00013133285,0.00004106834,0.00051921135,0.0015532018,0.9975255,0.00000766887],"about_ca_topic_score_codex":0.005423511,"about_ca_topic_score_gemma":0.0065992153,"teacher_disagreement_score":0.6106415,"about_ca_system_score_codex":0.0018133257,"about_ca_system_score_gemma":0.0010819827,"threshold_uncertainty_score":0.55537266},"labels":[],"label_agreement":null},{"id":"W1586468330","doi":"10.1007/11424918_45","title":"Producing Headline Summaries for Newspaper Articles","year":2005,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Lethbridge","funders":"","keywords":"Headline; Newspaper; Computer science; Natural language processing; Compression (physics); Artificial intelligence; Information retrieval; Linguistics; Advertising","score_opus":0.018022642072488673,"score_gpt":0.2686828907352826,"score_spread":0.25066024866279396,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1586468330","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.036192764,0.00957139,0.2461937,0.0075009116,0.015015875,0.004861868,0.23815566,0.17436872,0.26813912],"genre_scores_gemma":[0.06844776,0.008801331,0.42704666,0.0009819645,0.0077188527,0.0017173565,0.21458267,0.030356348,0.24034712],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99863726,0.00035835252,0.00021702496,0.00018161235,0.0005366362,0.000069052025],"domain_scores_gemma":[0.98025274,0.0071228305,0.001775571,0.0015399867,0.008551465,0.0007573551],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019000127,0.0021347092,0.0011119838,0.00994294,0.0012949173,0.003956589,0.0010176159,0.0010255342,0.15810688],"category_scores_gemma":[0.014908941,0.0011952599,0.000813465,0.008333504,0.00023659127,0.0027729175,0.0013677243,0.00093510293,0.16146842],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002481774,0.000068872425,0.00060459,0.0011607612,0.000049614664,0.00025445232,0.00044264842,0.00040181476,0.006177389,0.00080780115,0.7239147,0.26586926],"study_design_scores_gemma":[0.00013830086,0.00032390287,0.0028358814,0.00033492217,0.00025063145,0.0004012121,0.0011401328,0.0029360037,0.01603981,0.0019762283,0.97354263,0.00008021831],"about_ca_topic_score_codex":0.0015336425,"about_ca_topic_score_gemma":0.0037413978,"teacher_disagreement_score":0.15810688,"about_ca_system_score_codex":0.00061164773,"about_ca_system_score_gemma":0.0010888708,"threshold_uncertainty_score":0.5289202},"labels":[],"label_agreement":null},{"id":"W1587116172","doi":"10.1007/978-3-642-13059-5_32","title":"Automatic Text Segmentation for Movie Subtitles","year":2010,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Segmentation; Artificial intelligence; Text segmentation; Natural language processing; Information retrieval; Order (exchange); Image segmentation; Computer vision; Computer graphics (images)","score_opus":0.01334097530898132,"score_gpt":0.2748172718560952,"score_spread":0.2614762965471139,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1587116172","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.18010128,0.011733111,0.590209,0.0011625416,0.001797243,0.0017835883,0.06491974,0.10624768,0.042045787],"genre_scores_gemma":[0.20871416,0.0023844375,0.6434066,0.00029023582,0.00068954343,0.00073201273,0.10077464,0.0046653864,0.038342983],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996253,0.000028067829,0.000032306027,0.00015227935,0.000088299086,0.00007376273],"domain_scores_gemma":[0.9990362,0.00026874398,0.000085670756,0.000085845124,0.00043482642,0.000088820176],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00028212255,0.0018181284,0.0012280863,0.0052292314,0.0012113831,0.001451728,0.0010358632,0.0009718362,0.019568209],"category_scores_gemma":[0.0011357507,0.000546383,0.0011412764,0.0031793388,0.0003094002,0.0013789377,0.0007756375,0.00083686865,0.01476731],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007274895,0.00011973684,0.0016696315,0.00094145926,0.00006912616,0.00035876993,0.00025414833,0.0010518244,0.16750528,0.0013845123,0.077488,0.7484301],"study_design_scores_gemma":[0.00022628543,0.00082910067,0.04423294,0.00056855177,0.0006121955,0.002601357,0.001806497,0.32574037,0.39376783,0.008491863,0.22090138,0.00022159336],"about_ca_topic_score_codex":0.009670514,"about_ca_topic_score_gemma":0.018932525,"teacher_disagreement_score":0.019568209,"about_ca_system_score_codex":0.00066352973,"about_ca_system_score_gemma":0.0009570326,"threshold_uncertainty_score":0.06546217},"labels":[],"label_agreement":null},{"id":"W1587851455","doi":"10.1007/978-3-642-22218-4_41","title":"A Token Centric Part-of-Speech Tagger for Biomedical Text","year":2011,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"","keywords":"Computer science; Security token; Natural language processing; Artificial intelligence; Speech recognition; Computer network","score_opus":0.02049503366701117,"score_gpt":0.26729686620843934,"score_spread":0.24680183254142818,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1587851455","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016215969,0.002251346,0.87915677,0.00066444516,0.001454748,0.0004203652,0.03168801,0.06369331,0.0044550453],"genre_scores_gemma":[0.10563456,0.0010329208,0.82657117,0.0010609952,0.00048984063,0.00081429305,0.047537647,0.0022602864,0.01459835],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9989384,0.00016939828,0.00016490798,0.00036100115,0.0002696696,0.00009652409],"domain_scores_gemma":[0.99690187,0.0015073029,0.00029351807,0.00046293365,0.00065884675,0.00017551721],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017416892,0.0012606748,0.0014891664,0.003414372,0.00072320714,0.0014844927,0.0018577998,0.0022002158,0.008989043],"category_scores_gemma":[0.003253343,0.00069038523,0.00109055,0.0025671646,0.0005483865,0.0020120926,0.0017768583,0.0015397297,0.02185405],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012949174,0.0002233581,0.002196839,0.0012309237,0.00024041753,0.001104166,0.00023675137,0.0046351044,0.18374328,0.0052318396,0.07411997,0.72574246],"study_design_scores_gemma":[0.00043399024,0.0009789303,0.015225164,0.0004810688,0.00089853245,0.006179836,0.00055190193,0.26327,0.4274494,0.058571406,0.22546765,0.00049208716],"about_ca_topic_score_codex":0.0011101718,"about_ca_topic_score_gemma":0.0029222853,"teacher_disagreement_score":0.008989043,"about_ca_system_score_codex":0.00055455865,"about_ca_system_score_gemma":0.0019032547,"threshold_uncertainty_score":0.030071318},"labels":[],"label_agreement":null},{"id":"W1588233381","doi":"","title":"Coordination of Standard Arabic Subject Markers: Implementing the Agreement Asymmetries in the ACCG Framework","year":2010,"lang":"en","type":"article","venue":"The Florida AI Research Society","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal; Université du Québec à Trois-Rivières","funders":"","keywords":"Subject (documents); Agreement; Object (grammar); Computer science; Linguistics; Arabic; Discourse marker; Modern Standard Arabic; Natural language processing; Grammar; Artificial intelligence; Philosophy; World Wide Web","score_opus":0.032254497413874796,"score_gpt":0.3851662372041153,"score_spread":0.3529117397902405,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1588233381","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.22854702,0.00022134857,0.717722,0.00092137407,0.0002462274,0.00022769229,0.00032642877,0.0074678436,0.044320006],"genre_scores_gemma":[0.80987924,0.000073347,0.18532272,0.00017475212,0.000056558627,0.00007972744,0.00012512995,0.0010701441,0.0032183093],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99789786,0.0007203958,0.00017093236,0.0006005176,0.000338587,0.00027166185],"domain_scores_gemma":[0.9966445,0.0009847869,0.00028646487,0.001241132,0.0006764093,0.00016681723],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023932185,0.0006805096,0.0006573091,0.0009462103,0.0012813379,0.0027656131,0.0013309543,0.0010779615,0.0055996426],"category_scores_gemma":[0.0046921927,0.00071894284,0.0006160068,0.0008694382,0.0033543045,0.003482446,0.0039095,0.0012875096,0.001407134],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003467985,0.0000858718,0.005603562,0.00029562737,0.000042647647,0.0016474184,0.007869518,0.013536647,0.038959477,0.82126206,0.0039156103,0.10643483],"study_design_scores_gemma":[0.00019853124,0.00025091102,0.0044512977,0.00023323152,0.00027333762,0.0018644407,0.0035823404,0.24026294,0.08215539,0.5488469,0.11768336,0.00019736629],"about_ca_topic_score_codex":0.0033757652,"about_ca_topic_score_gemma":0.0036441581,"teacher_disagreement_score":0.0055996426,"about_ca_system_score_codex":0.001652137,"about_ca_system_score_gemma":0.0017960268,"threshold_uncertainty_score":0.018732667},"labels":[],"label_agreement":null},{"id":"W1591871317","doi":"10.1007/978-3-540-75175-5_55","title":"Manageable Phrase-Based Statistical Machine Translation Models","year":2007,"lang":"en","type":"book-chapter","venue":"Advances in intelligent and soft computing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Machine translation; Computer science; Phrase; Natural language processing; Artificial intelligence; Translation (biology); Memory footprint; Task (project management); Language model; Speech recognition; Programming language; Engineering","score_opus":0.03504070139893109,"score_gpt":0.30938949450004477,"score_spread":0.2743487931011137,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1591871317","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0060502407,0.00050662295,0.9779455,0.00048057945,0.00014272741,0.000107335014,0.001591236,0.0065036607,0.006672157],"genre_scores_gemma":[0.20335639,0.0011603084,0.7727335,0.00033760795,0.00029323468,0.00055023236,0.008050765,0.0014283706,0.012089619],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9982222,0.000586771,0.00016901646,0.0003249233,0.000602689,0.00009438042],"domain_scores_gemma":[0.9952726,0.002468879,0.0002119497,0.0014403111,0.000545454,0.000060901202],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020747217,0.0008255049,0.0010471999,0.0012115049,0.00067459897,0.0028605177,0.0027010248,0.0013514529,0.011074072],"category_scores_gemma":[0.010452181,0.000840343,0.001132227,0.0027916501,0.0008291186,0.0062070163,0.0024764803,0.0022107079,0.008385706],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037285304,0.00025425255,0.0007310562,0.0005173465,0.00014809091,0.00046751893,0.0004672323,0.12745573,0.015665371,0.32771495,0.037733845,0.48847184],"study_design_scores_gemma":[0.000056710094,0.00008560088,0.00018284922,0.000043702974,0.00005823382,0.00022245459,0.000060278417,0.61585253,0.008662296,0.34903747,0.025705777,0.000032170017],"about_ca_topic_score_codex":0.0012471392,"about_ca_topic_score_gemma":0.0024327103,"teacher_disagreement_score":0.011074072,"about_ca_system_score_codex":0.0007636677,"about_ca_system_score_gemma":0.001345415,"threshold_uncertainty_score":0.037046432},"labels":[],"label_agreement":null},{"id":"W1592805114","doi":"","title":"Analyzing linguistic data : a practical introduction to statistics using R","year":2008,"lang":"en","type":"book","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2077,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Variety (cybernetics); Computer science; Variation (astronomy); Construct (python library); Range (aeronautics); Visualization; Natural language processing; Measure (data warehouse); Linguistics; Artificial intelligence; Statistical model; Computational statistics; Data science; Data mining; Machine learning; Engineering","score_opus":0.06438462978172278,"score_gpt":0.36881849231843616,"score_spread":0.3044338625367134,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1592805114","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00036266888,0.018702187,0.9054471,0.010361086,0.0026249734,0.0004689365,0.0055190157,0.016544145,0.039969955],"genre_scores_gemma":[0.0038055032,0.017073132,0.9218673,0.0043822485,0.002379028,0.0015917821,0.0038562275,0.007889781,0.037154987],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9944729,0.0024065,0.0006055127,0.00051062857,0.0019179176,0.00008640739],"domain_scores_gemma":[0.97469056,0.019848904,0.00067101995,0.0017769485,0.0027771562,0.00023542461],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00769965,0.0019558866,0.0020017389,0.0037012326,0.00084224896,0.004002429,0.0028893147,0.0023096204,0.059171673],"category_scores_gemma":[0.030453287,0.0021017052,0.001763541,0.0050245076,0.001835347,0.004937917,0.0019495868,0.0059646466,0.081567034],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000018630943,0.00003959185,0.00026741848,0.0013302416,0.00008458409,0.00026692686,0.0004819488,0.002408553,0.0011368484,0.053739622,0.6551655,0.28506005],"study_design_scores_gemma":[0.000015878048,0.00003246352,0.0006770855,0.0005696905,0.000016738843,0.0005497842,0.00012894838,0.0036252246,0.00047756408,0.10505386,0.8887912,0.00006158437],"about_ca_topic_score_codex":0.0013093812,"about_ca_topic_score_gemma":0.0018575214,"teacher_disagreement_score":0.059171673,"about_ca_system_score_codex":0.0009599385,"about_ca_system_score_gemma":0.0021279813,"threshold_uncertainty_score":0.19794899},"labels":[],"label_agreement":null},{"id":"W1593794771","doi":"","title":"Analyse d'erreurs lexicales d'apprenants du FLS : démarche empirique pour l'élaboration d'un dictionnaire d'apprentissage","year":2007,"lang":"fr","type":"article","venue":"DOAJ (DOAJ: Directory of Open Access Journals)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Linguistics; Lexical item; Meaning (existential); Typology; Computer science; Lexical database; Lexical density; Artificial intelligence; Natural language processing; Lexical functional grammar; Elaboration; Psychology; Sociology; Humanities; Philosophy; Grammar; WordNet","score_opus":0.19628999571085737,"score_gpt":0.5266474201592246,"score_spread":0.33035742444836724,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1593794771","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8178072,0.00283664,0.15611543,0.0006174383,0.00007255983,0.00037212347,0.0047741723,0.0008363537,0.016568141],"genre_scores_gemma":[0.8998315,0.0012332093,0.08949187,0.00007202159,0.00003618635,0.00031820827,0.0048956466,0.00032258342,0.0037988236],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9908813,0.0035402174,0.0016515024,0.001240698,0.0024196892,0.00026656225],"domain_scores_gemma":[0.9317186,0.05542073,0.0020958,0.003232221,0.0072215954,0.0003110628],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0070310812,0.0009262383,0.0014528664,0.010691135,0.0011450731,0.0049242694,0.0014997541,0.0010744245,0.0061576124],"category_scores_gemma":[0.037972484,0.0007844877,0.0013223459,0.005671349,0.0018967186,0.0034952627,0.0020532755,0.0013795005,0.0018620013],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010591631,0.00046645533,0.36286634,0.0023693978,0.0006289255,0.0018073652,0.02136881,0.016261034,0.030196844,0.024501218,0.0030145282,0.53546],"study_design_scores_gemma":[0.00026720125,0.0007685747,0.55583066,0.0017958565,0.00091413717,0.0037206432,0.032098193,0.20718876,0.087957226,0.02134829,0.08772645,0.00038392746],"about_ca_topic_score_codex":0.025741292,"about_ca_topic_score_gemma":0.025619969,"teacher_disagreement_score":0.025741292,"about_ca_system_score_codex":0.0020372653,"about_ca_system_score_gemma":0.001872979,"threshold_uncertainty_score":0.051182926},"labels":[],"label_agreement":null},{"id":"W1594038170","doi":"10.7202/045696ar","title":"Yuste Rodrigo, Elia, ed. (2008): Topics in Language Resources for Translation and Localisation. Amsterdam: John Benjamins, 220 p.","year":2010,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Humanities; Translation (biology); Linguistics; Philosophy; Chemistry","score_opus":0.019295308658533374,"score_gpt":0.2716752873495652,"score_spread":0.25237997869103185,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1594038170","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00038608856,0.93908525,0.018255396,0.01007205,0.004062895,0.000029914569,0.0018231064,0.0006586433,0.025626626],"genre_scores_gemma":[0.006390014,0.889397,0.03189464,0.0019157704,0.0038320452,0.0001306055,0.0037900251,0.0010730481,0.061576825],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9988047,0.00030709006,0.0001664875,0.00022813377,0.0003975454,0.000096012875],"domain_scores_gemma":[0.9960051,0.0025003396,0.00029717464,0.00019667407,0.0008117995,0.00018886132],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028725513,0.0021352903,0.0020740416,0.0071265018,0.0010400926,0.0053244974,0.001588663,0.0025296875,0.06576589],"category_scores_gemma":[0.005024412,0.001617268,0.0012097498,0.012877305,0.0017292679,0.012360711,0.0017651083,0.0032030463,0.057917643],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007684362,0.000021082558,0.0003739372,0.0034184298,0.000049238857,0.00019038122,0.0015592959,0.00034249967,0.0009324474,0.017173791,0.62141687,0.35444513],"study_design_scores_gemma":[0.0000090266485,0.000011599657,0.0007208659,0.001236002,0.000030076639,0.00049789675,0.0004801219,0.00018852603,0.00052128424,0.009296552,0.98698246,0.000025536376],"about_ca_topic_score_codex":0.012074375,"about_ca_topic_score_gemma":0.01893465,"teacher_disagreement_score":0.06576589,"about_ca_system_score_codex":0.0018509033,"about_ca_system_score_gemma":0.0033443528,"threshold_uncertainty_score":0.22000879},"labels":[],"label_agreement":null},{"id":"W1594864214","doi":"10.1007/978-3-662-05320-1_11","title":"Exploiting the Web as Parallel Corpora for Cross-Language Information Retrieval","year":2003,"lang":"en","type":"book-chapter","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Query expansion; Information retrieval; Unavailability; Web query classification; Cross-language information retrieval; Web search query; Query language; Machine translation; Natural language processing; Set (abstract data type); Web page; Query optimization; Artificial intelligence; Parallel corpora; Search engine; World Wide Web; Programming language","score_opus":0.019339004576255998,"score_gpt":0.28320103519344314,"score_spread":0.26386203061718716,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1594864214","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019247707,0.013298512,0.88854617,0.0014119397,0.0009930183,0.00036390484,0.0024201018,0.012952506,0.060766116],"genre_scores_gemma":[0.094255865,0.012960257,0.8372649,0.00047899928,0.00063348864,0.00048808192,0.010961994,0.0024749925,0.040481407],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99910575,0.00029227047,0.00008331725,0.0001425777,0.00032293133,0.00005306174],"domain_scores_gemma":[0.99814165,0.00094975426,0.00006612253,0.0005087413,0.00028957968,0.0000441168],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001608677,0.000856977,0.0009348294,0.0040892367,0.0009386435,0.0032349667,0.0013279029,0.0009032372,0.010920252],"category_scores_gemma":[0.004180453,0.0009086065,0.0008220067,0.0063206176,0.00083139946,0.008814416,0.002219072,0.001448074,0.008497865],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019243092,0.00020968453,0.0006336676,0.00088153366,0.00012480706,0.0004008626,0.00037276215,0.0070438334,0.02216795,0.060163964,0.06560238,0.8422061],"study_design_scores_gemma":[0.00012936912,0.00016788869,0.0019486865,0.00029546904,0.0003344082,0.0019252647,0.00050641893,0.23141104,0.06777326,0.30207226,0.39331758,0.00011830753],"about_ca_topic_score_codex":0.0017911547,"about_ca_topic_score_gemma":0.003401316,"teacher_disagreement_score":0.010920252,"about_ca_system_score_codex":0.00054925296,"about_ca_system_score_gemma":0.00079572306,"threshold_uncertainty_score":0.036531925},"labels":[],"label_agreement":null},{"id":"W1594980056","doi":"10.1023/a:1012244918183","title":"Machine Translation of Closed Captions","year":2000,"lang":"en","type":"article","venue":"Machine Translation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":28,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Machine translation; Natural language processing; Artificial intelligence; Domain (mathematical analysis); Closed captioning; Speech translation; Computational linguistics","score_opus":0.01603385525805243,"score_gpt":0.2694640035533037,"score_spread":0.25343014829525123,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1594980056","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.038122762,0.004848077,0.86201525,0.0033127922,0.0052054455,0.0006429694,0.010299177,0.02153709,0.05401655],"genre_scores_gemma":[0.2592015,0.0036404955,0.66690004,0.00095455605,0.0015534223,0.0005857863,0.034956522,0.005610681,0.026596997],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9968874,0.0012054464,0.0002786055,0.0007499731,0.0006841851,0.00019444253],"domain_scores_gemma":[0.9895068,0.0052350666,0.0005020602,0.0017680097,0.0028162128,0.00017183645],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015066938,0.0018612285,0.0019725112,0.0029082873,0.0021522401,0.0044415705,0.0019653281,0.002502593,0.022896731],"category_scores_gemma":[0.013598675,0.0011588626,0.0016333526,0.0035114284,0.001431201,0.004635796,0.0028021804,0.002777642,0.012440886],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008882449,0.00048553332,0.0006481786,0.0033460383,0.00023426027,0.0032357823,0.001665476,0.021310398,0.048906773,0.14666116,0.18117146,0.5914466],"study_design_scores_gemma":[0.00035703983,0.00045143752,0.0013114085,0.0008153106,0.00037630126,0.0027359964,0.0010065187,0.3425392,0.1040153,0.268793,0.27736637,0.00023209707],"about_ca_topic_score_codex":0.0015022863,"about_ca_topic_score_gemma":0.0015934979,"teacher_disagreement_score":0.022896731,"about_ca_system_score_codex":0.0012269387,"about_ca_system_score_gemma":0.0018458422,"threshold_uncertainty_score":0.076597214},"labels":[],"label_agreement":null},{"id":"W1596134004","doi":"","title":"Fast and Accurate Arc Filtering for Dependency Parsing","year":2010,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"National Research Council Canada; University of Alberta","funders":"","keywords":"Computer science; Dependency (UML); Parsing; Dependency grammar; Graph; Dependency graph; Quadratic equation; Cascade; Time complexity; Context (archaeology); Algorithm; Sentence; Artificial intelligence; Overhead (engineering); Theoretical computer science; Mathematics","score_opus":0.014835633851801349,"score_gpt":0.2776293965862871,"score_spread":0.26279376273448574,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1596134004","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005170499,0.0002429582,0.9829441,0.00016390157,0.000051207713,0.000047713635,0.00034555735,0.010010793,0.0010233585],"genre_scores_gemma":[0.06261073,0.000289186,0.93169624,0.00016462886,0.00006651093,0.000096435186,0.0014730765,0.0009887619,0.0026143834],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9985343,0.00034073758,0.00010275855,0.0004549745,0.00044309112,0.00012415016],"domain_scores_gemma":[0.9966658,0.0016513598,0.00021519088,0.0007570646,0.0006545284,0.000056038865],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011171795,0.0012704505,0.00083620223,0.002016429,0.0009958626,0.0010192066,0.0020260997,0.0012465417,0.006977655],"category_scores_gemma":[0.004751471,0.0008261014,0.00089640316,0.0019772951,0.0007670193,0.0035536739,0.0012355708,0.0021173754,0.0043425015],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015502513,0.00010982659,0.001547903,0.00033756488,0.00008170442,0.00024709568,0.00022786848,0.0411632,0.030439494,0.01919204,0.030220142,0.87627816],"study_design_scores_gemma":[0.000029452107,0.00004847115,0.0010776642,0.000035595545,0.00006427376,0.0003260656,0.00005693362,0.88371134,0.04493678,0.045060523,0.024600497,0.000052398613],"about_ca_topic_score_codex":0.0073068924,"about_ca_topic_score_gemma":0.013827138,"teacher_disagreement_score":0.0073068924,"about_ca_system_score_codex":0.0010944038,"about_ca_system_score_gemma":0.0016771379,"threshold_uncertainty_score":0.02334255},"labels":[],"label_agreement":null},{"id":"W1596349298","doi":"","title":"Thinking linguistically: A scientific approach to language by Maya Honda & Wayne O’Neil.","year":2010,"lang":"en","type":"article","venue":"DOAJ (DOAJ: Directory of Open Access Journals)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Maya; Linguistics; Art; History; Philosophy; Archaeology","score_opus":0.10379727144666787,"score_gpt":0.5063980020401471,"score_spread":0.40260073059347923,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1596349298","genre_codex":"commentary","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0038482358,0.19293469,0.040299896,0.6956364,0.012681842,0.000031282383,0.00015410525,0.00018533024,0.05422827],"genre_scores_gemma":[0.34320912,0.30315816,0.09881092,0.10921773,0.020668138,0.0004841152,0.00036888808,0.000688159,0.1233948],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99802244,0.0011312854,0.0000801886,0.00025029786,0.0004241827,0.00009165216],"domain_scores_gemma":[0.9934574,0.0047983536,0.000218927,0.00021212293,0.0009371718,0.0003760401],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0040146327,0.0006867353,0.0005698954,0.0017573542,0.0033589592,0.0047933375,0.0011995255,0.001727147,0.0025556122],"category_scores_gemma":[0.010480003,0.00039209204,0.00050310843,0.0015872901,0.00841895,0.011641885,0.0021112303,0.007305416,0.0010172955],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000067310735,0.00003843855,0.00040072916,0.00024507695,0.000030887477,0.00013413225,0.010462043,0.00033726182,0.000287421,0.5654959,0.33892578,0.08357505],"study_design_scores_gemma":[0.0000131508505,0.00001771447,0.00049281935,0.00031769235,0.000017249775,0.00020808243,0.0027253549,0.0008399108,0.0002897536,0.46935394,0.5256876,0.00003676111],"about_ca_topic_score_codex":0.007382418,"about_ca_topic_score_gemma":0.010007536,"teacher_disagreement_score":0.007382418,"about_ca_system_score_codex":0.00245489,"about_ca_system_score_gemma":0.0025463048,"threshold_uncertainty_score":0.021231651},"labels":[],"label_agreement":null},{"id":"W1597682916","doi":"","title":"El proyecto METIS-II","year":2005,"lang":"es","type":"article","venue":"Redalyc (Universidad Autónoma del Estado de México)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Metis; Computer science; Machine translation; Natural language processing; Artificial intelligence; Translation (biology); Linguistics; Programming language; World Wide Web; Philosophy; Chemistry","score_opus":0.012319843290440802,"score_gpt":0.25800437544150834,"score_spread":0.24568453215106753,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1597682916","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010149745,0.0010567022,0.95453864,0.0020234566,0.00062971486,0.0007826763,0.00090404705,0.0032825659,0.026632333],"genre_scores_gemma":[0.10358153,0.0013472683,0.8399977,0.0014832962,0.00088204234,0.002026227,0.0041156765,0.0020446659,0.04452152],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99514633,0.0012391144,0.00031472713,0.0007755644,0.0021706764,0.0003535661],"domain_scores_gemma":[0.997996,0.000545405,0.000097457414,0.00046059492,0.0007635359,0.00013707989],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026741948,0.0016861157,0.0013745348,0.0016438836,0.0012723216,0.004374278,0.002633101,0.0016270912,0.013720623],"category_scores_gemma":[0.005831999,0.0009054935,0.0011282165,0.0013857662,0.0016746329,0.00430448,0.0053405324,0.003031306,0.009697699],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010843322,0.00023872926,0.0038293519,0.0026798632,0.00018915535,0.0008642137,0.0016371298,0.00766804,0.07004754,0.28253877,0.052361127,0.5768617],"study_design_scores_gemma":[0.00023988969,0.00064803334,0.002660866,0.0006852911,0.0001599684,0.0023816272,0.0011651882,0.07849708,0.1556493,0.16010937,0.5976444,0.00015899971],"about_ca_topic_score_codex":0.0013079802,"about_ca_topic_score_gemma":0.0017033184,"teacher_disagreement_score":0.013720623,"about_ca_system_score_codex":0.001323422,"about_ca_system_score_gemma":0.002400865,"threshold_uncertainty_score":0.045900106},"labels":[],"label_agreement":null},{"id":"W1597790068","doi":"","title":"Un algoritmo lingüístico-estadístico para resumen automático de textos especializados","year":2009,"lang":"es","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Humanities; Philosophy","score_opus":0.014756250857085316,"score_gpt":0.26554930277562067,"score_spread":0.25079305191853535,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1597790068","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020857643,0.0027107906,0.95007604,0.0007439033,0.00051081297,0.0003906056,0.0018974075,0.018649217,0.0041636215],"genre_scores_gemma":[0.13240342,0.0016711283,0.84333783,0.000354608,0.00028214982,0.00049054984,0.0034512454,0.00132611,0.016682949],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99662775,0.00071882113,0.00050129375,0.0007786094,0.0011905335,0.00018292786],"domain_scores_gemma":[0.9951023,0.0014855774,0.0003209433,0.0009919989,0.0019615935,0.00013748539],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031068232,0.0013333021,0.0014984388,0.0050097043,0.0015079805,0.00453501,0.0015228608,0.0014683274,0.008052472],"category_scores_gemma":[0.009515764,0.0007206547,0.0013182103,0.003467911,0.0010346634,0.0031532897,0.0016757383,0.0013633845,0.005711177],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006941628,0.0001233943,0.0027836468,0.0016604558,0.00016991087,0.00027945417,0.0011822891,0.007223233,0.09269662,0.007384728,0.014243168,0.8715589],"study_design_scores_gemma":[0.00020628389,0.0009444594,0.01694316,0.0008122988,0.0008190161,0.0023328448,0.0021832515,0.32017088,0.31858718,0.022552207,0.31404507,0.0004033576],"about_ca_topic_score_codex":0.0055141957,"about_ca_topic_score_gemma":0.006205846,"teacher_disagreement_score":0.008052472,"about_ca_system_score_codex":0.0011495004,"about_ca_system_score_gemma":0.0026922787,"threshold_uncertainty_score":0.0269382},"labels":[],"label_agreement":null},{"id":"W1598101953","doi":"10.1007/978-3-642-39931-2_14","title":"Summarization of Legal Texts with High Cohesion and Automatic Compression Rate","year":2013,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Automatic summarization; Computer science; Exploit; Centroid; Artificial intelligence; Cohesion (chemistry); Cluster analysis; Weighting; Natural language processing; Graph; Sentence; Theoretical computer science; Data mining","score_opus":0.006764947891861104,"score_gpt":0.2258325785188064,"score_spread":0.2190676306269453,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1598101953","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.25117704,0.004758748,0.70326334,0.0013926345,0.0008839595,0.00082380587,0.0071063954,0.016295563,0.01429851],"genre_scores_gemma":[0.42560202,0.0015906825,0.52618617,0.00022463196,0.0010460493,0.0005148808,0.026953904,0.0013931345,0.016488502],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993231,0.00021994238,0.0000781318,0.00010952558,0.00021376707,0.000055598954],"domain_scores_gemma":[0.99733764,0.0010924832,0.00021997899,0.00029158997,0.0009827254,0.00007551502],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00059722946,0.00093339605,0.0008359386,0.0025457127,0.00059157045,0.0009719285,0.0005920821,0.00076187996,0.007418735],"category_scores_gemma":[0.003948125,0.0003099591,0.000550919,0.0019499339,0.00024844747,0.0011164746,0.00060624554,0.00059359387,0.0038645584],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00089192815,0.00019991619,0.000956039,0.0012180457,0.00013918626,0.0008856883,0.00060426496,0.010040648,0.14288786,0.004155998,0.03521524,0.8028051],"study_design_scores_gemma":[0.00042531668,0.001404615,0.011744222,0.00027988985,0.00095543877,0.0019274008,0.0012298711,0.52924687,0.33907074,0.017359952,0.09621031,0.00014550032],"about_ca_topic_score_codex":0.0008718491,"about_ca_topic_score_gemma":0.0015523363,"teacher_disagreement_score":0.007418735,"about_ca_system_score_codex":0.00027733337,"about_ca_system_score_gemma":0.0005575327,"threshold_uncertainty_score":0.024818122},"labels":[],"label_agreement":null},{"id":"W1599016936","doi":"","title":"The Winograd Schema Challenge","year":2011,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":865,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Schema (genetic algorithms); Ambiguity; Natural language processing; Artificial intelligence; Turing test; Sentence; Turing; Information retrieval; Programming language","score_opus":0.035399085143150454,"score_gpt":0.25911663256442086,"score_spread":0.2237175474212704,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1599016936","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03664895,0.011867831,0.24694051,0.26426324,0.0054756976,0.0002006964,0.003974223,0.0031032546,0.4275257],"genre_scores_gemma":[0.6646176,0.010950074,0.12046582,0.041160267,0.0054684235,0.0006727921,0.0069778236,0.003256379,0.14643084],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9951805,0.001854249,0.00031136494,0.0011165891,0.0010952723,0.00044201326],"domain_scores_gemma":[0.99098796,0.0043830373,0.0003171433,0.0026400206,0.0010671613,0.00060473016],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005606096,0.00052973646,0.0012111386,0.0012421464,0.004412415,0.007310462,0.0030235727,0.0046135252,0.044090007],"category_scores_gemma":[0.025919132,0.000673229,0.0011346024,0.00279534,0.006164418,0.027611002,0.008611195,0.0070862793,0.010430493],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00004136825,0.000012314237,0.00023040004,0.00006706834,0.000007696262,0.00007577547,0.0003174763,0.00032054662,0.000058740603,0.926314,0.045848735,0.026705833],"study_design_scores_gemma":[0.000022205099,0.0000057096304,0.000086278524,0.000057411216,0.000005527787,0.00021142786,0.00027573315,0.0012312384,0.0002078425,0.8497819,0.14810193,0.000012874423],"about_ca_topic_score_codex":0.0038237951,"about_ca_topic_score_gemma":0.0028321,"teacher_disagreement_score":0.044090007,"about_ca_system_score_codex":0.0027881914,"about_ca_system_score_gemma":0.0026957649,"threshold_uncertainty_score":0.14749575},"labels":[],"label_agreement":null},{"id":"W1599853666","doi":"10.1023/a:1024137719751","title":"D-LTAG System: Discourse Parsing with a Lexicalized Tree-Adjoining Grammar","year":2003,"lang":"en","type":"article","venue":"Journal of Logic Language and Information","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":65,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Sentence; Computer science; Parsing; Natural language processing; Grammar; Artificial intelligence; Linguistics; Semantics (computer science); Syntax; Programming language; Philosophy","score_opus":0.008249876731618465,"score_gpt":0.2518049367087167,"score_spread":0.2435550599770982,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1599853666","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012674384,0.00024259447,0.7906943,0.0003856061,0.0002479766,0.00025071693,0.006084513,0.18402377,0.0053960164],"genre_scores_gemma":[0.1547091,0.00028531704,0.80206764,0.000513156,0.00019655781,0.00037183074,0.017321024,0.015193741,0.009341683],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99923646,0.00018257895,0.00010914718,0.00029374592,0.00012348776,0.000054627828],"domain_scores_gemma":[0.9984413,0.0007958323,0.00007481302,0.00032945178,0.00029020055,0.000068375244],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011906447,0.0010807593,0.0015044981,0.0019268867,0.0008871568,0.002939503,0.0024317652,0.0018494335,0.018401949],"category_scores_gemma":[0.0041290163,0.001243914,0.0011025917,0.0015129016,0.00093511277,0.003587064,0.0028077904,0.0014622221,0.011728081],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010866157,0.0004323051,0.0023579905,0.0013009256,0.00018186834,0.0009708737,0.0014215532,0.010617486,0.062379662,0.08469769,0.16733818,0.6672149],"study_design_scores_gemma":[0.00046277477,0.00024926168,0.0013812035,0.00024091188,0.00043735106,0.00074292615,0.0004887717,0.5879203,0.12831748,0.13027544,0.149243,0.00024066366],"about_ca_topic_score_codex":0.0032256623,"about_ca_topic_score_gemma":0.0035029326,"teacher_disagreement_score":0.018401949,"about_ca_system_score_codex":0.00064290204,"about_ca_system_score_gemma":0.0016897107,"threshold_uncertainty_score":0.06156063},"labels":[],"label_agreement":null},{"id":"W1600429586","doi":"10.1007/978-3-540-30586-6_34","title":"Customisable Semantic Analysis of Texts","year":2005,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Parsing; Natural language processing; Artificial intelligence; Syntactic predicate; Relation (database); Semantic analysis (machine learning); Semantic relation; Information retrieval; Database","score_opus":0.010644060282955066,"score_gpt":0.2613522510840163,"score_spread":0.2507081908010612,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1600429586","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009472642,0.00030967637,0.95968616,0.00028076468,0.00011231856,0.00015704356,0.0015312969,0.019540992,0.0089091305],"genre_scores_gemma":[0.16449921,0.00064952933,0.8088804,0.00019669806,0.00018144539,0.00021580611,0.007719936,0.0051684226,0.012488478],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9975204,0.0004806855,0.00034036464,0.0007527154,0.0007729671,0.00013296888],"domain_scores_gemma":[0.9951983,0.0016658686,0.00022177439,0.0021189132,0.00070098805,0.000094134724],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024236043,0.0009343868,0.0007436231,0.0046446095,0.0009194591,0.0035228643,0.0020062726,0.0007671253,0.010486765],"category_scores_gemma":[0.0069754915,0.0007801853,0.002149441,0.0039515505,0.0020979424,0.007061612,0.004040554,0.0018150451,0.0042124786],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028723862,0.0001184142,0.0017327843,0.000795615,0.00015204714,0.00035308115,0.001729006,0.004969674,0.024743177,0.35498223,0.018304016,0.59183264],"study_design_scores_gemma":[0.00003657818,0.000066082226,0.0018186952,0.0003302996,0.00024635767,0.0007046781,0.00071014534,0.09105992,0.06628958,0.64233285,0.19631825,0.00008654164],"about_ca_topic_score_codex":0.0011439577,"about_ca_topic_score_gemma":0.0017536852,"teacher_disagreement_score":0.010486765,"about_ca_system_score_codex":0.0012483983,"about_ca_system_score_gemma":0.0011805203,"threshold_uncertainty_score":0.035081685},"labels":[],"label_agreement":null},{"id":"W1601652018","doi":"","title":"A Comparison of Syntactically Motivated Word Alignment Spaces","year":2006,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Redundancy (engineering); Parsing; Word (group theory); Dependency grammar; Syntax; Limiting; Focus (optics); Dependency (UML); Rule-based machine translation; Reduction (mathematics); Natural language processing; Word order; Artificial intelligence; Space (punctuation); Mathematics","score_opus":0.015271534343428178,"score_gpt":0.304018159735593,"score_spread":0.2887466253921648,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1601652018","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6298862,0.0068492857,0.33058485,0.0013714414,0.00025854702,0.00035079895,0.0019636515,0.009036781,0.019698441],"genre_scores_gemma":[0.79424655,0.0014745899,0.19628727,0.00037967626,0.000060098148,0.0004030955,0.0043677245,0.0012890483,0.0014920332],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99203694,0.0043126238,0.00087487715,0.00066233997,0.0017118747,0.00040129508],"domain_scores_gemma":[0.977211,0.015730945,0.0010851122,0.0032850674,0.0020942772,0.00059355993],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0055964678,0.0007886807,0.001403147,0.0032602376,0.0009952685,0.00330979,0.0016293803,0.0016974784,0.0031430018],"category_scores_gemma":[0.027754074,0.000504104,0.0010257361,0.0035037252,0.0018437295,0.0051244632,0.0027645982,0.0014637978,0.00090061396],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0031238352,0.0011845519,0.011461228,0.002131057,0.00037158086,0.00035892308,0.0021281203,0.24561714,0.04160993,0.20626804,0.009743595,0.476002],"study_design_scores_gemma":[0.00055783923,0.0036532271,0.010977244,0.00038701948,0.00033220768,0.00090212334,0.002302146,0.6951662,0.04561908,0.21049584,0.02927217,0.00033493008],"about_ca_topic_score_codex":0.00089800055,"about_ca_topic_score_gemma":0.0015414819,"teacher_disagreement_score":0.0055964678,"about_ca_system_score_codex":0.0013520902,"about_ca_system_score_gemma":0.002340296,"threshold_uncertainty_score":0.029597342},"labels":[],"label_agreement":null},{"id":"W1603271854","doi":"10.1007/978-3-642-13059-5_26","title":"Exploiting Frame Information for Prepositional Phrase Semantic Role Labeling","year":2010,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Athabasca University","funders":"","keywords":"Computer science; Semantic role labeling; Natural language processing; Predicate (mathematical logic); Artificial intelligence; Phrase; Frame (networking); Argument (complex analysis); Noun phrase; FrameNet; Task (project management); Word-sense disambiguation; Parsing; Noun; Sentence; WordNet","score_opus":0.009039467467741818,"score_gpt":0.24888950526273548,"score_spread":0.23985003779499367,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1603271854","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015019912,0.0006765668,0.9668179,0.00035378014,0.00021329436,0.00013498828,0.0018627247,0.00458926,0.010331549],"genre_scores_gemma":[0.21402694,0.0009159482,0.7695678,0.00019772661,0.00018775502,0.00020645688,0.007973503,0.0011492842,0.005774643],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992674,0.00018773662,0.000056463712,0.00020202623,0.0001833144,0.00010303228],"domain_scores_gemma":[0.9986688,0.00061386405,0.00007087012,0.00038010397,0.00022031904,0.000046005633],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011041116,0.0011179104,0.0009860825,0.0024538182,0.0009802043,0.001795302,0.001997608,0.001297683,0.0104868645],"category_scores_gemma":[0.0028549943,0.000997935,0.0014540711,0.0026330547,0.00067246245,0.0058260877,0.0017563652,0.001739621,0.004392912],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007457794,0.00032704213,0.0009480396,0.00060460524,0.00008928062,0.00042949087,0.0006094948,0.013473871,0.04427093,0.11686119,0.028654505,0.7929857],"study_design_scores_gemma":[0.00010472424,0.00012427881,0.0010626805,0.00020078303,0.00021235163,0.0003763831,0.00042342232,0.5090587,0.06268909,0.37115508,0.05447373,0.000118690856],"about_ca_topic_score_codex":0.0060227737,"about_ca_topic_score_gemma":0.011245775,"teacher_disagreement_score":0.0104868645,"about_ca_system_score_codex":0.0010344993,"about_ca_system_score_gemma":0.0011774551,"threshold_uncertainty_score":0.0350821},"labels":[],"label_agreement":null},{"id":"W1604065198","doi":"10.63317/4oo7so8h9jey","title":"MOOD: A Modular Object-Oriented Decoder for Statistical Machine Translation","year":2006,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Machine translation; Modular design; Mood; Translation (biology); Object (grammar); clone (Java method); Artificial intelligence; Programming language; Theoretical computer science; Psychology","score_opus":0.009332769535340214,"score_gpt":0.27360339398523154,"score_spread":0.2642706244498913,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1604065198","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004579934,0.0002160185,0.8839989,0.00018527312,0.0004479738,0.00013980042,0.0024059075,0.10237389,0.005652371],"genre_scores_gemma":[0.09735453,0.000340676,0.85430014,0.0006673169,0.00042766522,0.0004005498,0.010165023,0.020407282,0.015936773],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999032,0.0002467843,0.00011375089,0.00025873847,0.0002529587,0.00009582595],"domain_scores_gemma":[0.9982526,0.00064195116,0.00008598295,0.0003851084,0.00054327585,0.000091038004],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013605114,0.0015365873,0.0012492577,0.0017789352,0.00066173053,0.0023120102,0.0016955053,0.0016498624,0.020385545],"category_scores_gemma":[0.0043141143,0.0010431366,0.0011063479,0.001291433,0.00046572488,0.0022018515,0.0024402053,0.0015961173,0.022397693],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013637275,0.00025447126,0.0013232261,0.00064600736,0.00024143571,0.0006189354,0.000332884,0.006617869,0.05416175,0.0323438,0.1391702,0.76292557],"study_design_scores_gemma":[0.00067473756,0.0005599236,0.0014649727,0.00020249917,0.00040654102,0.0012924955,0.00027378154,0.5134568,0.17375407,0.12691669,0.1807086,0.00028896146],"about_ca_topic_score_codex":0.0013505252,"about_ca_topic_score_gemma":0.0031964232,"teacher_disagreement_score":0.020385545,"about_ca_system_score_codex":0.0004097483,"about_ca_system_score_gemma":0.001324628,"threshold_uncertainty_score":0.068196416},"labels":[],"label_agreement":null},{"id":"W1604477342","doi":"10.1007/978-3-540-30222-3_53","title":"Quantum, a French/English Cross-Language Question Answering System","year":2004,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Question answering; Natural language processing; Set (abstract data type); Artificial intelligence; Machine translation; Translation system; Information retrieval; Programming language","score_opus":0.00858013613596664,"score_gpt":0.26370821633179237,"score_spread":0.2551280801958257,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1604477342","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10550315,0.0030841022,0.58151776,0.003753522,0.0008106536,0.0007717443,0.020463178,0.22816758,0.05592821],"genre_scores_gemma":[0.47114196,0.0009990414,0.442595,0.002144912,0.00032689446,0.00054179534,0.030634305,0.0069537484,0.044662483],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991972,0.00021649673,0.00006594515,0.00022538478,0.00021053298,0.00008443027],"domain_scores_gemma":[0.99862504,0.00059249485,0.000051781004,0.00021146897,0.00041250535,0.00010668879],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012927171,0.0007526094,0.0011651742,0.0015904459,0.0016808866,0.001741757,0.0014866472,0.001207394,0.021673238],"category_scores_gemma":[0.0044522816,0.00055068225,0.0006698087,0.0014006991,0.00078824104,0.0045824847,0.002223861,0.0008425942,0.0070125926],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019189598,0.00062882475,0.0034256608,0.0013382421,0.00020272062,0.00068977056,0.0015144541,0.0061104023,0.058005184,0.07173902,0.32167783,0.53274894],"study_design_scores_gemma":[0.0010454053,0.00089221564,0.011589993,0.00019765054,0.00049820775,0.0011168448,0.0021140093,0.31881437,0.074897535,0.12799309,0.46050423,0.00033643458],"about_ca_topic_score_codex":0.022832144,"about_ca_topic_score_gemma":0.021725804,"teacher_disagreement_score":0.022832144,"about_ca_system_score_codex":0.0012211173,"about_ca_system_score_gemma":0.0018240467,"threshold_uncertainty_score":0.07250416},"labels":[],"label_agreement":null},{"id":"W1604751532","doi":"10.1111/j.1467-9922.2010.00616.x","title":"Lexical Frequency Profiles and Zipf's Law","year":2010,"lang":"en","type":"article","venue":"Language Learning","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":35,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University; University of Victoria","funders":"","keywords":"Zipf's law; Vocabulary; Word lists by frequency; Natural language processing; Artificial intelligence; Probabilistic logic; Word (group theory); Computer science; Homogeneous; Linguistics; Psychology; Statistics; Cognitive psychology; Statistical physics; Mathematics; Sentence","score_opus":0.0052063690556130495,"score_gpt":0.2558624053061282,"score_spread":0.25065603625051514,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1604751532","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.32691035,0.0023242538,0.6344466,0.002465076,0.000109119566,0.0003424719,0.0013276316,0.00095214316,0.031122405],"genre_scores_gemma":[0.9517121,0.0009828242,0.04231907,0.00025495354,0.00018161825,0.00034238535,0.0005562752,0.00007787613,0.0035730253],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985322,0.00052296586,0.00007498677,0.00031962225,0.00039197743,0.0001582293],"domain_scores_gemma":[0.979122,0.015463816,0.001580325,0.0017956062,0.001596045,0.00044230037],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032101152,0.00038731823,0.0009065774,0.0034827474,0.0009772675,0.0024560718,0.0013858613,0.0015167623,0.0040044216],"category_scores_gemma":[0.040474206,0.0005110209,0.0006791963,0.002392787,0.0026088855,0.0061262813,0.00089336775,0.0012608927,0.0012175953],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021728249,0.00008604936,0.025395036,0.00015738218,0.00008930208,0.0004442531,0.0013016489,0.108861394,0.0017922369,0.79128444,0.005505046,0.06486596],"study_design_scores_gemma":[0.000046691224,0.000048500657,0.013424967,0.00005841326,0.00001970189,0.00043438099,0.00025598885,0.32894066,0.0007138711,0.650872,0.005096481,0.00008840349],"about_ca_topic_score_codex":0.0051111947,"about_ca_topic_score_gemma":0.0018119342,"teacher_disagreement_score":0.0051111947,"about_ca_system_score_codex":0.0019064421,"about_ca_system_score_gemma":0.0006555058,"threshold_uncertainty_score":0.016976893},"labels":[],"label_agreement":null},{"id":"W1605180730","doi":"10.1007/3-540-36271-1_17","title":"Opening Statistical Translation Engines to Terminological Resources","year":2002,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Terminology; Computer science; Machine translation; Natural language processing; Field (mathematics); Word (group theory); Word error rate; Artificial intelligence; Translation (biology); Work (physics); Reduction (mathematics); Information retrieval; Linguistics; Mechanical engineering","score_opus":0.029075296226176665,"score_gpt":0.27788428607011967,"score_spread":0.248808989843943,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1605180730","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0048908535,0.001116384,0.9537859,0.0010859099,0.00058429356,0.000119182965,0.0011018655,0.025968788,0.011346746],"genre_scores_gemma":[0.12476864,0.003657254,0.8136613,0.0010854879,0.0012116095,0.00043303706,0.010105064,0.018488627,0.026588982],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9974759,0.00069779856,0.00037430372,0.00034595456,0.000953326,0.00015276035],"domain_scores_gemma":[0.992,0.0045182565,0.00025758383,0.002179679,0.00090957124,0.00013492547],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032095935,0.00094416377,0.0015923346,0.00267355,0.001102012,0.005521661,0.0023310934,0.001668185,0.019681428],"category_scores_gemma":[0.01314603,0.0018852423,0.0015083913,0.004434601,0.0021279813,0.009788621,0.0045243385,0.0031689976,0.014962265],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00030338796,0.00012754637,0.00033959118,0.0007307255,0.000110018555,0.0005298131,0.0008108443,0.010115696,0.016090725,0.3514775,0.09215479,0.52720934],"study_design_scores_gemma":[0.00008788904,0.000059289414,0.00022965907,0.00019828515,0.00011497884,0.0004460974,0.0002230167,0.11813589,0.03734104,0.61339957,0.22967997,0.00008425753],"about_ca_topic_score_codex":0.0007401423,"about_ca_topic_score_gemma":0.0011516134,"teacher_disagreement_score":0.019681428,"about_ca_system_score_codex":0.00091443065,"about_ca_system_score_gemma":0.001451755,"threshold_uncertainty_score":0.0658409},"labels":[],"label_agreement":null},{"id":"W1608511509","doi":"10.18438/b8mc85","title":"Measuring the Extent of the Synonym Problem in Full-Text Searching","year":2008,"lang":"en","type":"article","venue":"Evidence Based Library and Information Practice","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Synonym (taxonomy); Information retrieval; Computer science; Term (time); Value (mathematics); Web page; Word (group theory); Search engine; World Wide Web; Mathematics; Genus; Machine learning; Biology","score_opus":0.021987962981900634,"score_gpt":0.25361863839317306,"score_spread":0.2316306754112724,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1608511509","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.97733885,0.0022230002,0.014953189,0.00025156894,0.000056939793,0.00028881716,0.0007156471,0.00019902088,0.003972905],"genre_scores_gemma":[0.98312265,0.0003523777,0.014173296,0.00010465045,0.00006299993,0.00014018502,0.0014137191,0.00006918186,0.000560885],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.93214804,0.02159761,0.013037324,0.00593091,0.02603412,0.0012519375],"domain_scores_gemma":[0.68149036,0.22258127,0.060458153,0.012868715,0.019828498,0.0027730812],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.021336006,0.0009839451,0.0017163663,0.011792548,0.0011769072,0.0030120166,0.0013584968,0.0016253833,0.001579915],"category_scores_gemma":[0.17064464,0.0005416581,0.0012529579,0.010183255,0.0019287318,0.010206554,0.004599862,0.0010813731,0.0007404954],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.003396122,0.0011510237,0.7567494,0.003984295,0.0018855338,0.0013427344,0.0074536516,0.008790526,0.016020007,0.0015424977,0.003630469,0.19405372],"study_design_scores_gemma":[0.0002810982,0.0036155721,0.86210984,0.0008923324,0.0013445072,0.014418757,0.009006654,0.06888043,0.021760857,0.006692254,0.010458388,0.0005392491],"about_ca_topic_score_codex":0.0014050804,"about_ca_topic_score_gemma":0.0011896373,"teacher_disagreement_score":0.021336006,"about_ca_system_score_codex":0.001003144,"about_ca_system_score_gemma":0.0010892734,"threshold_uncertainty_score":0.1128369},"labels":[],"label_agreement":null},{"id":"W160940076","doi":"10.63317/3vk8aoyuzdat","title":"Targeting Chinese Nominal Compounds in Corpora","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Pipeline (software); Lexicon; Sentence; Morpheme; Set (abstract data type); Speech recognition","score_opus":0.013105572708701833,"score_gpt":0.25896367309242246,"score_spread":0.24585810038372063,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W160940076","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.34013596,0.0062856814,0.50377357,0.0032780587,0.0007725915,0.0024551584,0.0554485,0.030832406,0.05701808],"genre_scores_gemma":[0.31519082,0.001888533,0.55072445,0.0005583308,0.00025274823,0.0014045117,0.112275705,0.0021858928,0.015519057],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99698454,0.00058168976,0.0002797309,0.0011686425,0.0006815309,0.00030392606],"domain_scores_gemma":[0.99543613,0.0014603501,0.0003652359,0.000998996,0.00158411,0.00015514933],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031612106,0.0015938552,0.0014101631,0.005401108,0.00253269,0.0025098359,0.0015120029,0.0014668278,0.008545115],"category_scores_gemma":[0.008725627,0.000887328,0.0008115971,0.0064219474,0.0009953992,0.0035069012,0.0024489474,0.0010707771,0.007032413],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00091095426,0.00034464532,0.023217978,0.0029553634,0.0002032638,0.0021776317,0.002710617,0.0077263294,0.19047545,0.023352759,0.10400616,0.6419189],"study_design_scores_gemma":[0.00034262156,0.0004954558,0.0488,0.0003452182,0.00050009775,0.0026656296,0.002699875,0.1025208,0.34363213,0.017661344,0.48002023,0.00031654345],"about_ca_topic_score_codex":0.011851551,"about_ca_topic_score_gemma":0.017483432,"teacher_disagreement_score":0.011851551,"about_ca_system_score_codex":0.0017105793,"about_ca_system_score_gemma":0.0032786091,"threshold_uncertainty_score":0.028586209},"labels":[],"label_agreement":null},{"id":"W161681728","doi":"","title":"Using context-dependent interpolation to combine statistical language and translation models for interactive machine translation","year":2000,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Trigram; Computer science; Machine translation; Natural language processing; Translation (biology); Artificial intelligence; Context (archaeology); Example-based machine translation; Interpolation (computer graphics); Machine translation software usability; Language model; Transfer-based machine translation; Evaluation of machine translation; Machine learning","score_opus":0.03654301589907547,"score_gpt":0.33580211899525425,"score_spread":0.29925910309617876,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W161681728","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013961752,0.0002168787,0.98231333,0.00012811826,0.000045111115,0.00004373299,0.000053413547,0.0023628175,0.0008749327],"genre_scores_gemma":[0.29327247,0.0002942368,0.7019919,0.00013161555,0.00012931963,0.0002077645,0.00042157623,0.0007949513,0.0027561295],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9978911,0.0011997633,0.00010092475,0.00038011294,0.00031277243,0.000115357325],"domain_scores_gemma":[0.9945445,0.0038785292,0.00025938387,0.00065054,0.0005351007,0.00013197251],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004411937,0.0010603337,0.0011863569,0.0022269143,0.0010397631,0.0015393391,0.001650556,0.0013552384,0.0041404436],"category_scores_gemma":[0.01472925,0.00096204615,0.0017474133,0.0028271717,0.00083487824,0.0029821922,0.0021724724,0.0023426085,0.0020773239],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00062579894,0.0002646973,0.004153095,0.00018558376,0.00028695463,0.000212657,0.00062968006,0.38751754,0.009467173,0.020034377,0.0025456164,0.57407683],"study_design_scores_gemma":[0.000012196104,0.000044230143,0.00028097304,0.000008714134,0.000018890612,0.000029280098,0.000018341128,0.988361,0.0013282655,0.009044634,0.0008349011,0.000018620742],"about_ca_topic_score_codex":0.010525992,"about_ca_topic_score_gemma":0.018171716,"teacher_disagreement_score":0.010525992,"about_ca_system_score_codex":0.0011797346,"about_ca_system_score_gemma":0.0013482294,"threshold_uncertainty_score":0.023332834},"labels":[],"label_agreement":null},{"id":"W161736330","doi":"","title":"Approaches to High Accuracy Retrieval: Phrase-Based Search Experiments in the HARD Track.","year":2004,"lang":"en","type":"article","venue":"Text REtrieval Conference","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Focus (optics); Phrase; Selection (genetic algorithm); Natural language processing; Word (group theory); Artificial intelligence; Noun phrase; Noun; Information retrieval; Mathematics","score_opus":0.1651441880432227,"score_gpt":0.33111074693969944,"score_spread":0.16596655889647674,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W161736330","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.897703,0.010359552,0.055448927,0.0027151993,0.0011525551,0.004998323,0.0074133985,0.005371933,0.0148371775],"genre_scores_gemma":[0.82566875,0.0015730729,0.14220762,0.0014371523,0.000602637,0.0030954455,0.015439944,0.0009842828,0.008991082],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9774818,0.014171603,0.002023522,0.0016711794,0.0038148665,0.00083688844],"domain_scores_gemma":[0.8522637,0.12971237,0.002298871,0.008496322,0.004538774,0.0026899641],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.026072672,0.0023123887,0.0029715397,0.0026154423,0.0025587913,0.00318797,0.0038222496,0.0057508303,0.009843691],"category_scores_gemma":[0.11008709,0.0010825219,0.0016291032,0.0034346944,0.0019941658,0.008370861,0.0036976966,0.0050196825,0.0037226884],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0617849,0.04685712,0.02785149,0.016223544,0.0044765305,0.0023694849,0.008600711,0.098679684,0.041934926,0.010081804,0.10548579,0.57565403],"study_design_scores_gemma":[0.01808015,0.04322507,0.03478437,0.000553309,0.0019644701,0.002539945,0.0046634614,0.79538757,0.04086543,0.028733382,0.028430022,0.00077286264],"about_ca_topic_score_codex":0.010601699,"about_ca_topic_score_gemma":0.009047964,"teacher_disagreement_score":0.026072672,"about_ca_system_score_codex":0.0017748887,"about_ca_system_score_gemma":0.0020058488,"threshold_uncertainty_score":0.13788706},"labels":[],"label_agreement":null},{"id":"W1619002","doi":"10.1007/978-3-642-38457-8_31","title":"Improved Arabic-French Machine Translation through Preprocessing Schemes and Language Analysis","year":2013,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Fowler Kennedy Sport Medicine Clinic","funders":"","keywords":"Machine translation; Computer science; Preprocessor; Arabic; Natural language processing; BLEU; Artificial intelligence; Machine translation software usability; Evaluation of machine translation; Translation (biology); Machine translation system; Example-based machine translation; Speech recognition; Linguistics","score_opus":0.012250881863594531,"score_gpt":0.26619015813311836,"score_spread":0.25393927626952384,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1619002","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.052574284,0.0023611425,0.86472845,0.001265846,0.0020179981,0.0003322785,0.0047510806,0.04831756,0.023651283],"genre_scores_gemma":[0.14290549,0.0011141128,0.81663567,0.00048352615,0.000527871,0.0002411815,0.012296633,0.005260862,0.02053461],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99889463,0.00038044676,0.00012896338,0.00025289785,0.00022284235,0.00012028851],"domain_scores_gemma":[0.9977168,0.00053724746,0.00007966317,0.00039676984,0.0012209646,0.000048492286],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008280299,0.0024337627,0.0012285543,0.0021812776,0.001433552,0.0025084587,0.0008559537,0.00091204693,0.017243577],"category_scores_gemma":[0.0033698732,0.0004766401,0.0012832468,0.0021443784,0.00039121407,0.0014710305,0.0013630651,0.0013609632,0.018597987],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00076664274,0.00021066677,0.00093118136,0.0008319098,0.00014690413,0.0011603963,0.00056250475,0.007693347,0.096899144,0.011756736,0.05482723,0.82421327],"study_design_scores_gemma":[0.0003270225,0.00068927224,0.0044041607,0.00030414888,0.0007664386,0.0030134069,0.0011276839,0.27745768,0.39340153,0.02357567,0.2946347,0.0002981879],"about_ca_topic_score_codex":0.005826787,"about_ca_topic_score_gemma":0.008340821,"teacher_disagreement_score":0.017243577,"about_ca_system_score_codex":0.00065602263,"about_ca_system_score_gemma":0.0016108934,"threshold_uncertainty_score":0.057685494},"labels":[],"label_agreement":null},{"id":"W1625323","doi":"10.1007/bf00233285","title":"Using LTAG-Based Features for Semantic Role Labeling","year":2006,"lang":"en","type":"article","venue":"The Journal of Membrane Biology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Semantic role labeling; Discriminative model; Artificial intelligence; Natural language processing; Parsing; Task (project management); Tree (set theory); Grammar; Decision tree; Mathematics; Linguistics; Sentence","score_opus":0.017743721320356964,"score_gpt":0.2973177865186044,"score_spread":0.27957406519824746,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1625323","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025682125,0.00019140549,0.9244281,0.00018607381,0.00018369303,0.00020063887,0.003877381,0.03686304,0.008387596],"genre_scores_gemma":[0.21331024,0.0003077333,0.74940634,0.00033349294,0.00007087721,0.0005899322,0.018984536,0.005763112,0.011233645],"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9996177,0.000049255686,0.000036758513,0.00012791778,0.000095903626,0.000072442606],"domain_scores_gemma":[0.99947745,0.00009266496,0.000053715128,0.00020790208,0.00012915659,0.00003912091],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004888335,0.0006255804,0.000377691,0.0008300518,0.00038955818,0.0012972184,0.00092896883,0.00092832063,0.0076059015],"category_scores_gemma":[0.000754106,0.00038944194,0.00056600216,0.0011935828,0.00041903608,0.0018601179,0.0010871569,0.00124127,0.008531567],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00079030334,0.00018059034,0.0018901579,0.00056167867,0.00004575856,0.0008211682,0.0007966416,0.0037497797,0.6556531,0.05381752,0.022534791,0.25915858],"study_design_scores_gemma":[0.00007296996,0.00035667518,0.0031957726,0.00008706085,0.000075621465,0.0010874117,0.00031766185,0.11418567,0.3959075,0.04657193,0.4380644,0.000077308534],"about_ca_topic_score_codex":0.0014125871,"about_ca_topic_score_gemma":0.001454121,"teacher_disagreement_score":0.0076059015,"about_ca_system_score_codex":0.00072526006,"about_ca_system_score_gemma":0.0004879736,"threshold_uncertainty_score":0.025444329},"labels":[],"label_agreement":null},{"id":"W1627936606","doi":"10.1109/icassp.1995.479657","title":"Improved decision trees for phonetic modeling","year":2002,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Computer Research Institute of Montréal","funders":"","keywords":"Computer science; Decision tree; Pruning; Schema (genetic algorithms); Context (archaeology); Phone; Artificial intelligence; Tree (set theory); Poisson distribution; Feature (linguistics); Function (biology); Machine learning; Speech recognition; Mathematics; Linguistics; Statistics","score_opus":0.024767869919965883,"score_gpt":0.27231470086325527,"score_spread":0.2475468309432894,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1627936606","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0040408396,0.001048486,0.99214023,0.00019184007,0.000100129844,0.00006396775,0.00028240643,0.0009011987,0.0012308285],"genre_scores_gemma":[0.13201988,0.0010939656,0.8604662,0.00029850833,0.00025007757,0.00043873827,0.0018302568,0.00036921885,0.0032332134],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99551094,0.0023562277,0.00032996494,0.00067212776,0.00086697977,0.00026376912],"domain_scores_gemma":[0.9893378,0.0085287215,0.00032629824,0.0005059901,0.001128944,0.0001722205],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0051958216,0.0015032358,0.0023520745,0.0027543148,0.0009968099,0.0016432812,0.0023647265,0.0019106325,0.007129336],"category_scores_gemma":[0.01727772,0.001148622,0.002484313,0.0028753185,0.00085590227,0.0025714745,0.0018343821,0.0040299585,0.002668756],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001998521,0.00009342542,0.00081151107,0.0002139843,0.00013622934,0.0001194329,0.00020123216,0.71417314,0.0011374434,0.05001998,0.0058243377,0.22706933],"study_design_scores_gemma":[0.000016331753,0.0000134779575,0.000060387254,0.000017806487,0.000010358941,0.000012202557,0.000006642025,0.974939,0.00021979932,0.023295738,0.001398802,0.00000950037],"about_ca_topic_score_codex":0.009569879,"about_ca_topic_score_gemma":0.011028841,"teacher_disagreement_score":0.009569879,"about_ca_system_score_codex":0.0017278764,"about_ca_system_score_gemma":0.0016601819,"threshold_uncertainty_score":0.027478456},"labels":[],"label_agreement":null},{"id":"W1629368090","doi":"10.48550/arxiv.1407.8215","title":"Two-pass Discourse Segmentation with Pairing and Global Features","year":2014,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Pairing; Segmentation; Computer science; Artificial intelligence; Physics","score_opus":0.023869644782219767,"score_gpt":0.21761498427210396,"score_spread":0.1937453394898842,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1629368090","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16636564,0.0009376831,0.7871195,0.00059002236,0.0002766784,0.00061166164,0.0018538878,0.032105755,0.010139153],"genre_scores_gemma":[0.48432773,0.00017749898,0.4979377,0.00019864656,0.00017825846,0.00037388477,0.003991887,0.0008276071,0.011986827],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99850094,0.00032724784,0.00011365292,0.0006249754,0.0002721983,0.00016095194],"domain_scores_gemma":[0.9974553,0.0011261348,0.0002685673,0.00049996906,0.0004920516,0.00015790832],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020772398,0.0021512175,0.0015035822,0.0024357694,0.0008248617,0.0018288229,0.0015646382,0.001911875,0.006789186],"category_scores_gemma":[0.0040507005,0.0004737132,0.0010451204,0.0012506198,0.00064428186,0.0029269867,0.0017901583,0.0020187506,0.0047192313],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016670261,0.0004974118,0.005642605,0.0003705695,0.00013937367,0.0003731021,0.00072405476,0.014305597,0.097142,0.004527816,0.011990536,0.86261994],"study_design_scores_gemma":[0.00010556563,0.0007667403,0.008133016,0.000080565784,0.00016200545,0.0004380484,0.00049890863,0.79581475,0.16492498,0.011302401,0.017654063,0.000118971795],"about_ca_topic_score_codex":0.0029298214,"about_ca_topic_score_gemma":0.0047528874,"teacher_disagreement_score":0.006789186,"about_ca_system_score_codex":0.00073972903,"about_ca_system_score_gemma":0.0012760188,"threshold_uncertainty_score":0.022712052},"labels":[],"label_agreement":null},{"id":"W1630014687","doi":"","title":"Exploring the Effects of First- and Second-Language Proficiency on Summarizing in French as a Second Language.","year":2000,"lang":"en","type":"article","venue":"DOAJ (DOAJ: Directory of Open Access Journals)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Acadia University","funders":"","keywords":"Automatic summarization; Computer science; Comprehension approach; Language assessment; Linguistics; Language proficiency; Second-language attrition; Language education; Second-language acquisition; Language pedagogy; Psychology; Natural language processing; Natural language; Mathematics education","score_opus":0.1059457975495797,"score_gpt":0.455930479286816,"score_spread":0.3499846817372363,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1630014687","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9982779,0.000061976316,0.00026086485,0.000043499866,0.0000022141148,0.000009735482,0.000039228424,0.000020572314,0.0012839189],"genre_scores_gemma":[0.99849474,0.00004518906,0.0005059765,0.00002150686,0.0000022013803,0.000011587706,0.000066396446,0.000009260418,0.0008431803],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9983619,0.00084096025,0.000102527236,0.0002628502,0.00027041996,0.00016139461],"domain_scores_gemma":[0.95206505,0.038129672,0.0051027965,0.0011211007,0.0022495082,0.0013319046],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027572094,0.0005625469,0.00038530174,0.0007114351,0.00048092852,0.0024584155,0.00044029372,0.0006118282,0.005217544],"category_scores_gemma":[0.025646077,0.00018374901,0.0003724026,0.00044508072,0.0006598174,0.0014905205,0.00058986683,0.0006302678,0.0005661859],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021691734,0.003657795,0.68128586,0.0006836516,0.00043510462,0.0020869293,0.07776274,0.0037601285,0.087528475,0.0020065363,0.0015997975,0.13702385],"study_design_scores_gemma":[0.000048193306,0.0032278975,0.9520225,0.000083966326,0.00015594704,0.00066133786,0.021720242,0.0045640822,0.012515816,0.0011821338,0.0037361921,0.00008171114],"about_ca_topic_score_codex":0.015817115,"about_ca_topic_score_gemma":0.012842705,"teacher_disagreement_score":0.015817115,"about_ca_system_score_codex":0.0007564404,"about_ca_system_score_gemma":0.00070595805,"threshold_uncertainty_score":0.031450093},"labels":[],"label_agreement":null},{"id":"W1640443237","doi":"10.1613/jair.4584","title":"Knowledge-Based Textual Inference via Parse-Tree Transformations","year":2015,"lang":"en","type":"article","venue":"Journal of Artificial Intelligence Research","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"FP7 Information and Communication Technologies; Azrieli Foundation; Bar-Ilan University","keywords":"Computer science; Inference; Natural language processing; Parsing; Artificial intelligence; Logical consequence; Formalism (music); Correctness; Parse tree; Natural language understanding; Natural language; Representation (politics); Tree structure; Algorithm; Programming language; Data structure","score_opus":0.23184268415197817,"score_gpt":0.46167685917493306,"score_spread":0.2298341750229549,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1640443237","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.002782991,0.00006071873,0.98968214,0.00011491104,0.00002106624,0.00014069972,0.0007570587,0.0050140796,0.0014262964],"genre_scores_gemma":[0.06049835,0.00016504813,0.9317976,0.00012515337,0.00005327643,0.0002621663,0.0046602986,0.0010063753,0.0014316012],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9965564,0.0007941349,0.00040444446,0.0008660828,0.0011970456,0.00018204428],"domain_scores_gemma":[0.9927672,0.0046567875,0.0003437273,0.0010596141,0.0010650089,0.0001076642],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00264024,0.0012558372,0.0011306017,0.0039298786,0.001402543,0.003101728,0.0032214643,0.001394953,0.0073282258],"category_scores_gemma":[0.020018164,0.0010477798,0.002843345,0.0035709466,0.0017397004,0.005539873,0.003441072,0.0028931445,0.004500958],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035498443,0.00033486955,0.00234296,0.00089867477,0.00020587994,0.0008790349,0.0013631675,0.08641733,0.019926695,0.23600797,0.027658263,0.62361014],"study_design_scores_gemma":[0.00007249496,0.00003732073,0.0005316376,0.00008172255,0.00009033585,0.00022685321,0.00017572573,0.63248426,0.021351136,0.3251635,0.019729434,0.000055605757],"about_ca_topic_score_codex":0.0036052656,"about_ca_topic_score_gemma":0.0060021277,"teacher_disagreement_score":0.0073282258,"about_ca_system_score_codex":0.0012536186,"about_ca_system_score_gemma":0.0025201354,"threshold_uncertainty_score":0.02451533},"labels":[],"label_agreement":null},{"id":"W1649009502","doi":"10.63317/4jcxttuiia7j","title":"Evaluation of a Machine Translation System for Low Resource Languages: METIS-II","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Metis; Computer science; Parsing; Natural language processing; German; Set (abstract data type); Machine translation; Test (biology); Artificial intelligence; Programming language; Linguistics; World Wide Web","score_opus":0.03617233535713793,"score_gpt":0.31135595602391064,"score_spread":0.2751836206667727,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1649009502","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8329578,0.002955744,0.06506224,0.0012895239,0.00085525453,0.0018842112,0.023728818,0.027808867,0.043457527],"genre_scores_gemma":[0.67340636,0.00089524145,0.16903837,0.00049219903,0.00025502872,0.001140924,0.12441852,0.0037964578,0.02655691],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99602026,0.0015113003,0.00042362293,0.0006092065,0.0012540793,0.00018157401],"domain_scores_gemma":[0.9964432,0.0012963376,0.0001273673,0.0005037777,0.0013990088,0.00023024536],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034139208,0.0012980637,0.0014032411,0.0014316284,0.0010233013,0.0018161597,0.0014302969,0.0009313196,0.0067544845],"category_scores_gemma":[0.006874519,0.00033479524,0.0005616433,0.0014547374,0.00046909702,0.0019620154,0.0012568514,0.00090505276,0.0041450923],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.005589721,0.0018614447,0.013132173,0.005274898,0.00097233587,0.0029324826,0.0039894083,0.031490684,0.28578162,0.00792684,0.10623582,0.5348126],"study_design_scores_gemma":[0.003322696,0.0060185073,0.054511532,0.00042687843,0.0007854574,0.004788245,0.004777711,0.15740556,0.5000494,0.002679761,0.26486576,0.00036846453],"about_ca_topic_score_codex":0.0037741978,"about_ca_topic_score_gemma":0.0044729663,"teacher_disagreement_score":0.0067544845,"about_ca_system_score_codex":0.0010454352,"about_ca_system_score_gemma":0.0015194779,"threshold_uncertainty_score":0.022596002},"labels":[],"label_agreement":null},{"id":"W1650129375","doi":"","title":"Incremental Decoding for Phrase-Based Statistical Machine Translation","year":2010,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Machine translation; Decoding methods; Phrase; Computer science; Synchronous context-free grammar; Translation (biology); Artificial intelligence; Natural language processing; Transfer-based machine translation; Focus (optics); Computation; Speech recognition; Word (group theory); Example-based machine translation; Algorithm; Linguistics","score_opus":0.01888748214207734,"score_gpt":0.309112405897886,"score_spread":0.29022492375580866,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1650129375","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0020652201,0.000161905,0.9950984,0.00008084422,0.000039190953,0.000036949983,0.000091370224,0.0015379104,0.00088821055],"genre_scores_gemma":[0.08226996,0.00037093647,0.9132281,0.00016853538,0.00012035839,0.00022311727,0.0008847236,0.00067753554,0.0020567067],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985819,0.0006335086,0.00009448216,0.00018336065,0.00044552074,0.0000611851],"domain_scores_gemma":[0.9965603,0.0020971121,0.00013109557,0.00048952864,0.00067958585,0.000042421347],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012690665,0.0008126024,0.0008260353,0.0008890968,0.0005720747,0.0009923427,0.0015670448,0.00093337253,0.006636723],"category_scores_gemma":[0.007317252,0.000519514,0.00070729514,0.0016493504,0.0006399027,0.0018808526,0.0012284823,0.001337231,0.0044040037],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00039038586,0.00010375742,0.00058750133,0.0005341937,0.00015132305,0.000278981,0.00021535257,0.13763978,0.04603571,0.09853648,0.013038566,0.70248795],"study_design_scores_gemma":[0.000045500892,0.00009900979,0.00023094767,0.000025431818,0.00004920899,0.00022898725,0.000025391322,0.89648086,0.024671508,0.068559095,0.009540785,0.00004332791],"about_ca_topic_score_codex":0.0022738294,"about_ca_topic_score_gemma":0.003802953,"teacher_disagreement_score":0.006636723,"about_ca_system_score_codex":0.0005947514,"about_ca_system_score_gemma":0.0012511631,"threshold_uncertainty_score":0.022202015},"labels":[],"label_agreement":null},{"id":"W1657275562","doi":"","title":"Cross-Lingual Distributional Profiles of Concepts for Measuring Semantic Distance","year":2007,"lang":"en","type":"article","venue":"TUbilio (Technical University of Darmstadt)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":38,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Bootstrapping (finance); Natural language processing; WordNet; Artificial intelligence; Semantic similarity; Task (project management); Word (group theory); Distributional semantics; Lexicon; Ranking (information retrieval); Linguistics; Mathematics","score_opus":0.0174102123637127,"score_gpt":0.2928284967007948,"score_spread":0.2754182843370821,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1657275562","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.44416916,0.00040289905,0.54515886,0.00009407046,0.00004554928,0.00017945337,0.0019970092,0.001010384,0.006942579],"genre_scores_gemma":[0.8241638,0.00012521894,0.17230405,0.000030604388,0.000017598537,0.000254326,0.0024062966,0.00017716894,0.0005208984],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9962603,0.0011094213,0.00047639044,0.0008801834,0.0011103898,0.00016339809],"domain_scores_gemma":[0.9890455,0.004850537,0.0011450117,0.0022063323,0.0022782462,0.00047444663],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032447525,0.0004901135,0.0006483757,0.0075283954,0.0009394532,0.0019798062,0.0008662162,0.0008633886,0.0019141611],"category_scores_gemma":[0.020134583,0.00033849344,0.00067185063,0.005536013,0.00094112544,0.0041638273,0.0030395656,0.0009800078,0.0010152947],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012846147,0.0006526291,0.18353426,0.00074465695,0.0007470003,0.0005351995,0.005212966,0.031529956,0.060760546,0.031568546,0.00376527,0.6796643],"study_design_scores_gemma":[0.00010489341,0.0011744256,0.26350558,0.0002536441,0.00032095853,0.0038369275,0.005779499,0.508418,0.0881176,0.103756644,0.024284322,0.00044743656],"about_ca_topic_score_codex":0.0012178296,"about_ca_topic_score_gemma":0.0017404046,"teacher_disagreement_score":0.0075283954,"about_ca_system_score_codex":0.0006163489,"about_ca_system_score_gemma":0.00059079455,"threshold_uncertainty_score":0.017160118},"labels":[],"label_agreement":null},{"id":"W1662715038","doi":"10.18806/tesl.v29i1.1092","title":"The Myth of FANBOYS: Coordination, Commas, and College Composition Classes","year":2012,"lang":"en","type":"article","venue":"TESL Canada Journal","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Composition (language); Linguistics; Mythology; Psychology; Mathematics education; Pedagogy; Philosophy; Literature; Art","score_opus":0.007090626090631253,"score_gpt":0.23135779090299669,"score_spread":0.22426716481236544,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1662715038","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.35724825,0.0025554493,0.044198114,0.051260665,0.0007837582,0.00005549857,0.00034817454,0.00045173214,0.54309833],"genre_scores_gemma":[0.96837395,0.00041574534,0.005398748,0.00097532256,0.00020682886,0.000024619305,0.000057169753,0.0000845214,0.024463015],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9995055,0.00024608718,0.00001709264,0.000083775514,0.0000744761,0.000073070994],"domain_scores_gemma":[0.9984522,0.0010929157,0.00010473272,0.00012558482,0.00009745604,0.00012706252],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010234773,0.00023857196,0.00018663336,0.0007446656,0.0035908169,0.004237646,0.0005238381,0.0012170092,0.016992237],"category_scores_gemma":[0.004042141,0.00022457562,0.00018524163,0.0007489849,0.007300155,0.0077180844,0.0018712076,0.0020577132,0.0006580326],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010908413,0.00003344291,0.002079635,0.000027626142,0.0000032383343,0.00025765813,0.0152915595,0.00014016834,0.00056672224,0.9475913,0.010601406,0.02329815],"study_design_scores_gemma":[0.00003823918,0.00006471335,0.006902616,0.0001598939,0.000018006274,0.0007224493,0.042769615,0.0029492702,0.00233673,0.69735473,0.24663806,0.000045713827],"about_ca_topic_score_codex":0.011723881,"about_ca_topic_score_gemma":0.022325223,"teacher_disagreement_score":0.016992237,"about_ca_system_score_codex":0.0018288654,"about_ca_system_score_gemma":0.0009826766,"threshold_uncertainty_score":0.05684471},"labels":[],"label_agreement":null},{"id":"W16815992","doi":"10.1542/pir.27-7-243","title":"A Real-Time N-Gram Approach to Choosing Synonyms Based on Context","year":2015,"lang":"en","type":"article","venue":"Pediatrics in Review","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Context (archaeology); Computer science; n-gram; Gram; Artificial intelligence; Geography; Geology","score_opus":0.037652168712233244,"score_gpt":0.31607313654078045,"score_spread":0.2784209678285472,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W16815992","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025928063,0.005910259,0.93874925,0.0017784359,0.00084663415,0.00058423606,0.004346123,0.010887599,0.010969434],"genre_scores_gemma":[0.13871443,0.0024576327,0.8459236,0.00052463403,0.00057348475,0.0003379972,0.005693812,0.0005054656,0.0052688913],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9967874,0.0011665416,0.00041117732,0.0007970614,0.0006822277,0.00015561185],"domain_scores_gemma":[0.9948901,0.0026081658,0.0003311335,0.0005767222,0.0013883286,0.00020551312],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002473463,0.0020839076,0.0016456563,0.006101719,0.0018259727,0.0033430674,0.0023037186,0.001587219,0.009361191],"category_scores_gemma":[0.011280269,0.00065854617,0.0013165296,0.0068395594,0.00084847846,0.0063260198,0.002823774,0.0029622526,0.010330181],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005966952,0.00023439399,0.0022114115,0.0008390106,0.00030963295,0.00029791938,0.00063454814,0.0037617339,0.029109322,0.009262297,0.035277013,0.91746616],"study_design_scores_gemma":[0.00024974463,0.0006079862,0.0066680354,0.00057122693,0.00087014824,0.0021850024,0.0032840131,0.757752,0.03954794,0.087272294,0.100619175,0.00037237786],"about_ca_topic_score_codex":0.007118995,"about_ca_topic_score_gemma":0.022693342,"teacher_disagreement_score":0.009361191,"about_ca_system_score_codex":0.00088868005,"about_ca_system_score_gemma":0.0027989938,"threshold_uncertainty_score":0.03131628},"labels":[],"label_agreement":null},{"id":"W1686963956","doi":"","title":"Improved Phrase-based SMT with Syntactic Reordering Patterns Learned from Lattice Scoring.","year":2010,"lang":"en","type":"article","venue":"Arrow@dit (Dublin Institute of Technology)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Atomic Energy of Canada Limited","keywords":"Computer science; Parsing; Natural language processing; Phrase; NIST; Artificial intelligence; Word (group theory); Machine translation; Task (project management); Exploit; Speech recognition; Linguistics","score_opus":0.014478758979176411,"score_gpt":0.2509120185420485,"score_spread":0.23643325956287206,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1686963956","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.035716295,0.00058972527,0.94167006,0.0004037592,0.00028755938,0.00028790184,0.0016023706,0.012725427,0.0067168595],"genre_scores_gemma":[0.2796943,0.00020892386,0.7057112,0.0002451807,0.00017262416,0.0002876939,0.004551974,0.000936681,0.008191445],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9984078,0.00072463386,0.0001087256,0.0002573384,0.00043920174,0.000062339706],"domain_scores_gemma":[0.99758196,0.0011762201,0.00017688116,0.0003930475,0.00059232546,0.00007956761],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014708443,0.0009083767,0.0007391069,0.0011746539,0.00038732582,0.00082313234,0.0012022542,0.0008313765,0.004497885],"category_scores_gemma":[0.005381522,0.0003716307,0.0006378,0.001360336,0.00033794157,0.0016781419,0.0010641803,0.0013896802,0.0046443143],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000245553,0.0002537697,0.002520954,0.0002731026,0.0001468429,0.00017061208,0.00013473726,0.07416567,0.02744032,0.0052357824,0.014692507,0.8747202],"study_design_scores_gemma":[0.000047483285,0.00011039626,0.00088595646,0.000022959248,0.000027997805,0.0001250214,0.000052747255,0.9725691,0.010553161,0.009881471,0.0056940448,0.000029730325],"about_ca_topic_score_codex":0.0033469177,"about_ca_topic_score_gemma":0.008299915,"teacher_disagreement_score":0.004497885,"about_ca_system_score_codex":0.0005452433,"about_ca_system_score_gemma":0.0012927571,"threshold_uncertainty_score":0.015046954},"labels":[],"label_agreement":null},{"id":"W169063832","doi":"10.63317/2ghqfd9j9xh9","title":"A French Human Reference Corpus for Multi-Document Summarization and Sentence Compression","year":2010,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Automatic summarization; Computer science; Natural language processing; Multi-document summarization; Artificial intelligence; Sentence; Information retrieval; Compression (physics)","score_opus":0.04659203308536194,"score_gpt":0.33944365897139733,"score_spread":0.2928516258860354,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W169063832","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19595248,0.022474905,0.32661575,0.005669836,0.00398514,0.0033650396,0.325719,0.040958025,0.075259954],"genre_scores_gemma":[0.2262232,0.0034180998,0.2244327,0.00081022276,0.0009934473,0.0023861625,0.4994074,0.0030810006,0.039247785],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99581426,0.0021639247,0.00030072386,0.00077862374,0.0007335327,0.00020902611],"domain_scores_gemma":[0.98931235,0.0035695047,0.00025868032,0.0014250464,0.005083644,0.00035084932],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036339094,0.0016973047,0.0013696495,0.006154488,0.0024527586,0.0017971616,0.0013869096,0.0019593085,0.027633391],"category_scores_gemma":[0.010101381,0.00039059453,0.0007010418,0.0036587636,0.00077265344,0.0015043246,0.0012318939,0.0010732674,0.012502633],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010958676,0.0005608418,0.0021346486,0.0035324693,0.00019941118,0.0016990441,0.0014909912,0.0064170044,0.047577247,0.0062144278,0.40419513,0.52488303],"study_design_scores_gemma":[0.0005890031,0.0009314514,0.023474753,0.0007069369,0.0005004879,0.003872097,0.0017151327,0.0354522,0.08441817,0.004177558,0.84388196,0.00028017082],"about_ca_topic_score_codex":0.027758943,"about_ca_topic_score_gemma":0.02522557,"teacher_disagreement_score":0.027758943,"about_ca_system_score_codex":0.0013783744,"about_ca_system_score_gemma":0.003112436,"threshold_uncertainty_score":0.09244293},"labels":[],"label_agreement":null},{"id":"W169800830","doi":"","title":"Scaling up Analogical Learning","year":2008,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Scalability; Artificial intelligence; Simple (philosophy); Scaling; Identification (biology); Space (punctuation); Limit (mathematics); Machine learning; Theoretical computer science; Mathematics; Epistemology","score_opus":0.06614344725555805,"score_gpt":0.337967393007861,"score_spread":0.27182394575230295,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W169800830","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2751903,0.009223788,0.63613653,0.007733268,0.0010775697,0.0007592945,0.0011132215,0.015299004,0.053467005],"genre_scores_gemma":[0.72387534,0.002400458,0.2629527,0.001203012,0.0004132334,0.000544758,0.0011050175,0.00090571726,0.006599781],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9961792,0.0010831343,0.00026721242,0.0009207384,0.0012891871,0.0002605345],"domain_scores_gemma":[0.9659475,0.021247124,0.00069534744,0.008539135,0.0029272821,0.0006436312],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032005082,0.0015916957,0.0020583135,0.0017078992,0.0008928993,0.0027442572,0.0040817168,0.0021640963,0.028981091],"category_scores_gemma":[0.04074112,0.000781142,0.0010842269,0.0030743014,0.0016634499,0.01235488,0.0050320555,0.00309815,0.0057240343],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000652251,0.0011370883,0.0027634988,0.00091086596,0.00021478771,0.00020854342,0.00045769603,0.11147424,0.015895767,0.0581622,0.021370748,0.7867523],"study_design_scores_gemma":[0.0002721092,0.00032811536,0.00084049173,0.00006804761,0.00010451299,0.00025384492,0.00032493836,0.76836306,0.00873557,0.20636694,0.014298182,0.000044115877],"about_ca_topic_score_codex":0.0025916004,"about_ca_topic_score_gemma":0.0019553495,"teacher_disagreement_score":0.028981091,"about_ca_system_score_codex":0.0015210363,"about_ca_system_score_gemma":0.0014921302,"threshold_uncertainty_score":0.096951425},"labels":[],"label_agreement":null},{"id":"W1702922849","doi":"10.1007/s10590-015-9172-5","title":"Complexity of alignment and decoding problems: restrictions and approximations","year":2015,"lang":"en","type":"article","venue":"Machine Translation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Memorial University of Newfoundland","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Decoding methods; Viterbi algorithm; Computer science; Parameterized complexity; Computational complexity theory; Sentence; Sequential decoding; Algorithm; Word (group theory); Polynomial; Iterative Viterbi decoding; List decoding; Theoretical computer science; Mathematics; Artificial intelligence; Block code","score_opus":0.08034618598023999,"score_gpt":0.2992751910534103,"score_spread":0.21892900507317034,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1702922849","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08368531,0.004751166,0.86830163,0.013505113,0.0005752726,0.00025378235,0.002042134,0.001004671,0.025880868],"genre_scores_gemma":[0.6980473,0.0049906382,0.26934397,0.0019098696,0.0033899527,0.0008609836,0.004942419,0.0017690953,0.014745825],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9847564,0.0069535784,0.00090588804,0.0023471448,0.0032993853,0.0017375649],"domain_scores_gemma":[0.7760564,0.20664836,0.0033016724,0.008820168,0.0036340645,0.0015392145],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010068183,0.0018915866,0.003931131,0.0032184431,0.0024053426,0.009407007,0.0049419,0.0042814086,0.013222063],"category_scores_gemma":[0.100043185,0.0021052144,0.003775869,0.0055103637,0.004954841,0.02423318,0.0061806557,0.010156227,0.001979402],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013347461,0.0006514283,0.004574554,0.0012579004,0.00023648868,0.00051195285,0.0011708842,0.40436214,0.0015888755,0.43367434,0.03366266,0.116973944],"study_design_scores_gemma":[0.00006436505,0.00003850956,0.00051450904,0.000057512036,0.000039063547,0.00015926993,0.00014807076,0.4812514,0.000597914,0.5154575,0.0016346058,0.00003734607],"about_ca_topic_score_codex":0.005218279,"about_ca_topic_score_gemma":0.005499694,"teacher_disagreement_score":0.013222063,"about_ca_system_score_codex":0.00520528,"about_ca_system_score_gemma":0.004695951,"threshold_uncertainty_score":0.05324626},"labels":[],"label_agreement":null},{"id":"W1709985024","doi":"","title":"Automatically Assessing Whether a Text Is Cliched, with Applications to Literary Analysis","year":2013,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Style (visual arts); Computer science; Text generation; Natural language processing; Artificial intelligence; Literature; Psychology; Art","score_opus":0.009963230175129095,"score_gpt":0.2962544717027327,"score_spread":0.28629124152760366,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1709985024","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8534724,0.002596979,0.11176652,0.0010343288,0.00033959062,0.0005270624,0.00729043,0.004027602,0.018945005],"genre_scores_gemma":[0.8962543,0.0007241394,0.09332303,0.0001210283,0.0002132766,0.00020952396,0.0062562153,0.0003768902,0.0025215598],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9956462,0.0010381549,0.0007176176,0.0012103538,0.0012227895,0.00016488967],"domain_scores_gemma":[0.96207315,0.019970542,0.006178266,0.002673892,0.008191494,0.0009127667],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003046324,0.0007509714,0.0008492428,0.015624071,0.0019181416,0.0052772723,0.00085870177,0.0012393496,0.0035634604],"category_scores_gemma":[0.045306176,0.00033535555,0.00034659676,0.009187247,0.0014371421,0.0045135897,0.002933766,0.0012625451,0.0029531366],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009871521,0.00034792357,0.24072044,0.002661222,0.00021211928,0.0012506454,0.013473483,0.0020102777,0.06344962,0.007996569,0.01681343,0.6500772],"study_design_scores_gemma":[0.00013829651,0.0005731684,0.66155547,0.0009895532,0.00031612642,0.006731478,0.028987052,0.11995541,0.050592396,0.04480116,0.08489372,0.00046618885],"about_ca_topic_score_codex":0.0017860306,"about_ca_topic_score_gemma":0.003213877,"teacher_disagreement_score":0.015624071,"about_ca_system_score_codex":0.0009703657,"about_ca_system_score_gemma":0.0007045426,"threshold_uncertainty_score":0.016110659},"labels":[],"label_agreement":null},{"id":"W1715930141","doi":"10.1109/icsmc.2001.971943","title":"Extracting meaningful semantic information with EMATISE: an HPSG-based Internet search engine parser","year":2002,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Natural language processing; Parsing; Head-driven phrase structure grammar; The Internet; Artificial intelligence; Natural language; Programming language; Information retrieval; World Wide Web; Generative grammar","score_opus":0.020611542249157185,"score_gpt":0.25065305817072925,"score_spread":0.23004151592157207,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1715930141","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024673942,0.00026571407,0.85919994,0.0005522456,0.0000921887,0.0003848616,0.0047346726,0.10228901,0.0078073065],"genre_scores_gemma":[0.11664889,0.00020623422,0.8594746,0.00050343,0.000056818844,0.00030039457,0.011631818,0.0038548848,0.0073228874],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99915767,0.0002624861,0.00008478372,0.0002121819,0.0002330411,0.000049714698],"domain_scores_gemma":[0.99779785,0.0014413445,0.0001086311,0.0002764941,0.0003254841,0.000050257277],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024245935,0.0007201559,0.0009058517,0.0020009142,0.0005958499,0.0018802977,0.0013045253,0.0012117525,0.009295602],"category_scores_gemma":[0.0071553937,0.00078848033,0.00064821384,0.0013465341,0.00092031894,0.0048443037,0.0018414933,0.0014990438,0.0038515693],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011028724,0.000866183,0.0054126275,0.001517691,0.0002056541,0.0013269614,0.003684834,0.014826053,0.0923422,0.097105175,0.1317188,0.64989084],"study_design_scores_gemma":[0.0005133582,0.0004139195,0.0064835185,0.00021776387,0.00025195026,0.0030420518,0.00097290525,0.41147172,0.17821969,0.08996765,0.30813554,0.00030998696],"about_ca_topic_score_codex":0.0017077804,"about_ca_topic_score_gemma":0.002182053,"teacher_disagreement_score":0.009295602,"about_ca_system_score_codex":0.0005828342,"about_ca_system_score_gemma":0.0011757436,"threshold_uncertainty_score":0.031096935},"labels":[],"label_agreement":null},{"id":"W1726342440","doi":"","title":"Identifying Parallel Documents from a Large Bilingual Collection of Texts: Application to Parallel Article Extraction in Wikipedia.","year":2011,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":26,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Parallel corpora; Baseline (sea); Natural language processing; Relation (database); Machine translation; Information retrieval; Artificial intelligence; Relationship extraction; Resource (disambiguation); Order (exchange); Bilingual dictionary; Information extraction; Data mining","score_opus":0.021871585835434738,"score_gpt":0.3141396691065393,"score_spread":0.29226808327110454,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1726342440","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.573408,0.007271396,0.3612464,0.0014589954,0.00059730245,0.0018382715,0.015079014,0.024101999,0.014998668],"genre_scores_gemma":[0.36069268,0.0010297642,0.6087334,0.00018013654,0.00021768145,0.0004428834,0.02241112,0.0006968883,0.0055955015],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99841666,0.00034073187,0.00017291079,0.0005948298,0.00039195936,0.000082954655],"domain_scores_gemma":[0.99559516,0.0020390747,0.00043406413,0.0005698987,0.0010736788,0.000288108],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001481726,0.0010634435,0.0007031789,0.0055835987,0.002048007,0.0013859029,0.0008126577,0.0012197417,0.0021178185],"category_scores_gemma":[0.007838533,0.0005193551,0.00061195326,0.0044246837,0.00062426616,0.0016785628,0.002029283,0.00068723265,0.0021778129],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000682818,0.0006879503,0.022873575,0.002501244,0.00046536975,0.003916318,0.004285162,0.005479425,0.132628,0.0026398452,0.03671784,0.78712237],"study_design_scores_gemma":[0.0004915125,0.0011088068,0.09793739,0.00032882916,0.00082495174,0.01625491,0.010883414,0.30091667,0.30902877,0.019283174,0.24251433,0.00042723204],"about_ca_topic_score_codex":0.006789675,"about_ca_topic_score_gemma":0.01433119,"teacher_disagreement_score":0.006789675,"about_ca_system_score_codex":0.0005624981,"about_ca_system_score_gemma":0.0017081617,"threshold_uncertainty_score":0.013500333},"labels":[],"label_agreement":null},{"id":"W172656279","doi":"","title":"RALI: Automatic Weighting of Text Window Distances","year":2010,"lang":"en","type":"article","venue":"Meeting of the Association for Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Window (computing); Weighting; Word (group theory); Computer science; SemEval; Artificial intelligence; Task (project management); Word-sense disambiguation; Natural language processing; Limit (mathematics); Pattern recognition (psychology); Mathematics","score_opus":0.007342829639935821,"score_gpt":0.2629071066691153,"score_spread":0.2555642770291795,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W172656279","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.026656264,0.0012140092,0.906124,0.00013859603,0.00030894106,0.0004341748,0.0021715965,0.061735228,0.0012171015],"genre_scores_gemma":[0.11563235,0.00035284652,0.8737032,0.00008963068,0.000118555676,0.0005541122,0.003967041,0.0028764994,0.0027058728],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9977822,0.0005263804,0.0002233024,0.00092876184,0.00042783935,0.00011160186],"domain_scores_gemma":[0.99552417,0.0020524287,0.0005056604,0.0010254506,0.00072458136,0.00016768056],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028533102,0.0022907876,0.0015239163,0.004078322,0.0006956268,0.0015617425,0.0031786105,0.0012259942,0.004682442],"category_scores_gemma":[0.011748752,0.00093595946,0.0011223519,0.0022796893,0.0004621179,0.0043453104,0.002195255,0.001860302,0.0041922173],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010121058,0.00024116637,0.0029005324,0.000771912,0.00028729104,0.00013613158,0.00045166677,0.011573001,0.058807224,0.005057437,0.02420236,0.89455926],"study_design_scores_gemma":[0.0002498229,0.0003893123,0.004641346,0.00007726997,0.000200696,0.00039644772,0.00022493483,0.8376523,0.11169055,0.014052149,0.030250473,0.00017470129],"about_ca_topic_score_codex":0.0017493622,"about_ca_topic_score_gemma":0.0035968428,"teacher_disagreement_score":0.004682442,"about_ca_system_score_codex":0.00069770153,"about_ca_system_score_gemma":0.0008316221,"threshold_uncertainty_score":0.015664339},"labels":[],"label_agreement":null},{"id":"W173756214","doi":"","title":"Near-Synonym Choice in an Intelligent Thesaurus","year":2007,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Thesaurus; Synonym (taxonomy); Focus (optics); Task (project management); Context (archaeology); Information retrieval; Coherence (philosophical gambling strategy); Natural language processing; Artificial intelligence; Mathematics","score_opus":0.022483611121602513,"score_gpt":0.32069812854921537,"score_spread":0.29821451742761285,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W173756214","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.18459068,0.00088599505,0.8021505,0.00043708383,0.0001150027,0.00067671144,0.00074640434,0.003880779,0.0065168203],"genre_scores_gemma":[0.3011611,0.00018546675,0.69504964,0.00010918512,0.00004430463,0.00022598194,0.0009357923,0.00022164235,0.0020667831],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99349314,0.0027642108,0.00092334184,0.001273464,0.0014345482,0.000111295434],"domain_scores_gemma":[0.99324036,0.0034630892,0.00084707356,0.0009688878,0.001235976,0.0002446165],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005076599,0.0007161365,0.0012870076,0.0066801826,0.0019542594,0.0025859615,0.0017244911,0.0017123243,0.0036257398],"category_scores_gemma":[0.017797487,0.0008136756,0.00097457576,0.0038460754,0.0011311948,0.0063517294,0.0024915792,0.0009236267,0.002234728],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009630645,0.00080030144,0.015028262,0.0014238154,0.00048202503,0.0010097206,0.004324499,0.02150315,0.095837265,0.03475638,0.0137489,0.8101226],"study_design_scores_gemma":[0.00043582733,0.0011063439,0.020607162,0.0005416954,0.0007108503,0.005401697,0.0038265507,0.61836994,0.14900088,0.10924157,0.090135746,0.0006217596],"about_ca_topic_score_codex":0.0014502669,"about_ca_topic_score_gemma":0.0035430472,"teacher_disagreement_score":0.0066801826,"about_ca_system_score_codex":0.0008415494,"about_ca_system_score_gemma":0.0018529898,"threshold_uncertainty_score":0.026847959},"labels":[],"label_agreement":null},{"id":"W1740463855","doi":"","title":"Web-scale N-gram models for lexical disambiguation","year":2009,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":84,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Natural language processing; Context (archaeology); n-gram; Artificial intelligence; Selection (genetic algorithm); Spelling; Scale (ratio); Information retrieval; Language model; Linguistics","score_opus":0.02128184406940061,"score_gpt":0.2996087455678126,"score_spread":0.278326901498412,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1740463855","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03361065,0.0016807631,0.9504364,0.00067458744,0.00027039208,0.00011681571,0.0020127483,0.0075504156,0.0036471665],"genre_scores_gemma":[0.5385631,0.0013569267,0.44368246,0.0005502943,0.0004936825,0.00042857084,0.0057483763,0.0009567511,0.008219905],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99886775,0.0005641904,0.00007881922,0.00022300749,0.00019235048,0.00007393795],"domain_scores_gemma":[0.9972017,0.0016117354,0.00018121058,0.00047450935,0.00044036206,0.000090628746],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016005277,0.0011633714,0.0011353343,0.0021097823,0.0011098402,0.0013582322,0.0014111794,0.0014253266,0.003955182],"category_scores_gemma":[0.006129113,0.00057414256,0.00085733604,0.002752352,0.00055164675,0.003817637,0.0013329122,0.002171028,0.006163536],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010478286,0.0004620215,0.005870433,0.00037232222,0.0003117976,0.0004635839,0.00044716685,0.349241,0.011168402,0.040113714,0.032977585,0.5575242],"study_design_scores_gemma":[0.000010534711,0.000016731123,0.00026180866,0.0000096935055,0.000009765472,0.000038352126,0.000019276442,0.97902083,0.00080230547,0.0184483,0.0013471821,0.000015127223],"about_ca_topic_score_codex":0.0070107733,"about_ca_topic_score_gemma":0.017411383,"teacher_disagreement_score":0.0070107733,"about_ca_system_score_codex":0.000838376,"about_ca_system_score_gemma":0.0011124474,"threshold_uncertainty_score":0.013939977},"labels":[],"label_agreement":null},{"id":"W1748071216","doi":"10.3968/j.css.1923669720141001.4219","title":"An Overview of Researches on Biolinguistics","year":2014,"lang":"en","type":"article","venue":"Canadian social science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Automatic summarization; Status quo; Generative grammar; Transformational grammar; Transformational leadership; Computer science; Epistemology; Grammar; Linguistics; Engineering ethics; Sociology; Political science; Philosophy; Artificial intelligence; Law; Public relations; Engineering","score_opus":0.0724060929841819,"score_gpt":0.3833504047068563,"score_spread":0.3109443117226744,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1748071216","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0013018444,0.9715127,0.00438782,0.0029364384,0.0007883264,0.000012776886,0.000108496904,0.000043388933,0.018908156],"genre_scores_gemma":[0.014345983,0.973067,0.00505892,0.0012155062,0.0013211558,0.000038615002,0.0002430166,0.000035601875,0.0046741897],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9992906,0.00020026781,0.00009370782,0.00016372808,0.0001856162,0.00006600001],"domain_scores_gemma":[0.99845517,0.0010869153,0.000093441,0.000048612426,0.00025049143,0.00006528635],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001110038,0.0009908321,0.00083489134,0.011053911,0.0017584773,0.0034994986,0.000903495,0.0017152481,0.005147263],"category_scores_gemma":[0.0019257106,0.0005023451,0.0005431991,0.012597018,0.002497653,0.0049745124,0.0013555423,0.0017950399,0.0022665737],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010389384,0.000087012275,0.0025513892,0.0117189195,0.000104633626,0.0006239629,0.0038373533,0.00072492944,0.0023399123,0.2505886,0.048058446,0.6792609],"study_design_scores_gemma":[0.0000038665567,0.000033606957,0.0029727542,0.0031575076,0.000033575754,0.00075061247,0.00087565713,0.00026543072,0.0004822841,0.054324575,0.93706703,0.000033011547],"about_ca_topic_score_codex":0.00557783,"about_ca_topic_score_gemma":0.005591497,"teacher_disagreement_score":0.011053911,"about_ca_system_score_codex":0.0027187387,"about_ca_system_score_gemma":0.0024695941,"threshold_uncertainty_score":0.01972586},"labels":[],"label_agreement":null},{"id":"W1751837507","doi":"10.5281/zenodo.8100420","title":"Evaluating the meaning of answers to reading comprehension questions: A semantics-based approach","year":2012,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":32,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Meaning (existential); Linguistics; Comprehension; Semantics (computer science); Reading (process); Reading comprehension; Computer science; Psychology; Natural language processing; Epistemology; Philosophy; Programming language","score_opus":0.06708645963828096,"score_gpt":0.36696391995857264,"score_spread":0.29987746032029167,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1751837507","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.39105976,0.0066965558,0.5567459,0.005674251,0.00044074425,0.001063016,0.0037668035,0.0026799394,0.031873085],"genre_scores_gemma":[0.85435015,0.0007080592,0.14079438,0.00024401334,0.00017232884,0.00030833133,0.0021491533,0.00022698606,0.0010466037],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9866985,0.0069987276,0.0013891361,0.0014921807,0.0030463557,0.0003751154],"domain_scores_gemma":[0.956444,0.030888958,0.003294568,0.0017211449,0.0068770405,0.0007742583],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010400478,0.0013379004,0.0012456255,0.009028949,0.0013139528,0.0064379713,0.0019769403,0.003143837,0.0047147945],"category_scores_gemma":[0.07072907,0.0007592381,0.0016859367,0.0031807474,0.0025155128,0.010433845,0.0038352988,0.0017681614,0.000917974],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004248939,0.0013643736,0.052726753,0.004799092,0.0011717206,0.0015924711,0.022839412,0.019188143,0.06935935,0.14863911,0.012585458,0.6614852],"study_design_scores_gemma":[0.0007635633,0.0019711966,0.062183622,0.0011964182,0.001794374,0.0020155571,0.017871074,0.24692512,0.032459576,0.60820067,0.024228087,0.00039065647],"about_ca_topic_score_codex":0.002004883,"about_ca_topic_score_gemma":0.0022334526,"teacher_disagreement_score":0.010400478,"about_ca_system_score_codex":0.0019341651,"about_ca_system_score_gemma":0.0017813264,"threshold_uncertainty_score":0.055003643},"labels":[],"label_agreement":null},{"id":"W1753594312","doi":"10.3968/5831","title":"A Quantitative Study on Tao Te Ching Based on a Comparable Corpus","year":2014,"lang":"en","type":"article","venue":"Studies in sociology of science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Sentence; Criticism; Translation (biology); Linguistics; Literature; Philosophy; Art; Chemistry","score_opus":0.07602516455840597,"score_gpt":0.4141923400351989,"score_spread":0.3381671754767929,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1753594312","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9871991,0.00021567015,0.0019435161,0.0002934325,0.000028924149,0.00075391727,0.0006784045,0.00001980198,0.008867261],"genre_scores_gemma":[0.98934096,0.00029764452,0.0054022125,0.00010951965,0.000037657002,0.0014695473,0.0015058644,0.000040247705,0.0017964117],"study_design_codex":"qualitative","study_design_gemma":"not_applicable","domain_scores_codex":[0.9931676,0.004292947,0.00071113306,0.00069525116,0.00092972803,0.00020342355],"domain_scores_gemma":[0.96246773,0.02540516,0.00252416,0.0018491559,0.006924726,0.00082903967],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007609159,0.00037961898,0.0004781521,0.005497071,0.003991122,0.0016096238,0.00059553876,0.0005486301,0.0029967856],"category_scores_gemma":[0.031046744,0.00022842825,0.00026818478,0.011254495,0.0030627935,0.0019393122,0.0016083308,0.00081765465,0.00024686495],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00044285491,0.0010494944,0.057722416,0.0019687705,0.00005734617,0.0034622203,0.84600353,0.0004631277,0.021214804,0.009387617,0.004948967,0.05327888],"study_design_scores_gemma":[0.00012835623,0.00142315,0.27317464,0.00074738055,0.00013992298,0.001548655,0.6364078,0.0023026597,0.010817589,0.0023242256,0.070869274,0.00011638285],"about_ca_topic_score_codex":0.009059296,"about_ca_topic_score_gemma":0.011940772,"teacher_disagreement_score":0.009059296,"about_ca_system_score_codex":0.0034986841,"about_ca_system_score_gemma":0.0025658999,"threshold_uncertainty_score":0.04024154},"labels":[],"label_agreement":null},{"id":"W1758800344","doi":"10.1017/cbo9781139136310.010","title":"Summarization","year":2012,"lang":"en","type":"book-chapter","venue":"Cambridge University Press eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Automatic summarization; Computer science; Field (mathematics); Linguistics; Natural language processing; Philosophy","score_opus":0.017617657429319762,"score_gpt":0.20415057652009824,"score_spread":0.1865329190907785,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1758800344","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0055604577,0.018729284,0.6829326,0.0033515068,0.0028658293,0.0017174062,0.00918931,0.022154853,0.2534988],"genre_scores_gemma":[0.08376062,0.020944739,0.4984738,0.0023407193,0.0028687795,0.0013831842,0.03788217,0.0062157693,0.34613034],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99823076,0.00038126283,0.00019116,0.0005610639,0.00052425795,0.00011153922],"domain_scores_gemma":[0.9980338,0.00047573535,0.0001363258,0.00047371298,0.00081285444,0.00006755771],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015382092,0.0017065785,0.0010490977,0.0035785509,0.0012122962,0.0042039184,0.002264253,0.001227802,0.07853574],"category_scores_gemma":[0.004704142,0.00044734505,0.0011447143,0.003198791,0.0006724176,0.0042233435,0.0025621867,0.0012409827,0.056760088],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000112195725,0.000048204914,0.00022541292,0.0020615582,0.00008723973,0.00021522655,0.0006125892,0.0022774725,0.010955977,0.046464473,0.19754474,0.73939496],"study_design_scores_gemma":[0.000022785236,0.00007541789,0.00038742195,0.00027101822,0.0000740366,0.00042313416,0.00025425717,0.004232379,0.009837419,0.034171574,0.95020574,0.000044867968],"about_ca_topic_score_codex":0.0007499516,"about_ca_topic_score_gemma":0.00091520214,"teacher_disagreement_score":0.07853574,"about_ca_system_score_codex":0.0009878117,"about_ca_system_score_gemma":0.0010498241,"threshold_uncertainty_score":0.2627282},"labels":[],"label_agreement":null},{"id":"W1759747647","doi":"","title":"Unsupervised Detection of Downward-Entailing Operators By Maximizing Classification Certainty","year":2012,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Initialization; Computer science; Spurious relationship; Bootstrapping (finance); Artificial intelligence; Pattern recognition (psychology); Distillation; Polarity (international relations); Bayesian probability; Data mining; Machine learning; Mathematics","score_opus":0.019435193288745302,"score_gpt":0.2641606494085521,"score_spread":0.2447254561198068,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1759747647","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.075436674,0.0002933846,0.92026716,0.000309102,0.000022212766,0.0001131284,0.0002172694,0.0012717282,0.0020694588],"genre_scores_gemma":[0.54924023,0.00018041946,0.4477186,0.0001489349,0.0000994996,0.00017495795,0.00086308154,0.00024269274,0.0013315819],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9960341,0.00119858,0.0003211798,0.0012552574,0.00096286385,0.00022808013],"domain_scores_gemma":[0.9881285,0.0072873984,0.0012498148,0.0012816295,0.001800554,0.00025206956],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0049837967,0.0011194577,0.0013276,0.0029645425,0.00090386,0.0020181474,0.002445268,0.001445514,0.0017744781],"category_scores_gemma":[0.020546352,0.00061702065,0.00092046446,0.0013826914,0.001558447,0.0035048355,0.0024372763,0.0023677815,0.000692886],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008185442,0.000371654,0.022485409,0.000575506,0.00021617464,0.00048369012,0.001343826,0.04376831,0.07185877,0.0458918,0.007410918,0.80477536],"study_design_scores_gemma":[0.00006269976,0.00010076849,0.00569272,0.000057348465,0.00007288477,0.00030592075,0.00016313409,0.8904824,0.03539052,0.06322112,0.004390221,0.0000602854],"about_ca_topic_score_codex":0.0014944708,"about_ca_topic_score_gemma":0.0028934237,"teacher_disagreement_score":0.0049837967,"about_ca_system_score_codex":0.00092703156,"about_ca_system_score_gemma":0.0014181844,"threshold_uncertainty_score":0.026357174},"labels":[],"label_agreement":null},{"id":"W17659133","doi":"10.1017/s1481803500004127","title":"Joint Parsing and Alignment with Weakly Synchronized Grammars","year":2010,"lang":"en","type":"article","venue":"North American Chapter of the Association for Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":62,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Parsing; Computer science; Natural language processing; Artificial intelligence; Treebank; Word (group theory); Bottom-up parsing; Machine translation; Top-down parsing; Rule-based machine translation; Discriminative model; Speech recognition; Linguistics","score_opus":0.007642669732457285,"score_gpt":0.2343343053735263,"score_spread":0.22669163564106903,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W17659133","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00924834,0.00035114956,0.9699931,0.00045893327,0.00013038573,0.000100419114,0.0011106922,0.009720474,0.00888657],"genre_scores_gemma":[0.20096622,0.0006552625,0.77303886,0.0002416966,0.00015204879,0.00017321463,0.0069744317,0.0046378076,0.013160364],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9963085,0.0015954772,0.0002378718,0.0007998294,0.0006823632,0.0003758777],"domain_scores_gemma":[0.9945398,0.00285449,0.000210426,0.0013606437,0.00090199924,0.00013268844],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036321122,0.0012178981,0.001432959,0.0022436492,0.0022715773,0.004170787,0.0026073137,0.0015085111,0.008892339],"category_scores_gemma":[0.011504049,0.0015619404,0.0017667623,0.0040401206,0.0023841714,0.004769397,0.004125091,0.0023686276,0.0051763207],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005676992,0.00013774031,0.0024994924,0.0005544507,0.00021715063,0.00073766283,0.0016363682,0.10902376,0.013837941,0.3323434,0.044745494,0.4936989],"study_design_scores_gemma":[0.00008462347,0.000083751795,0.0014505632,0.0001008806,0.00015195676,0.00029267397,0.00036451645,0.4979049,0.018395716,0.4346687,0.046375565,0.00012613887],"about_ca_topic_score_codex":0.049614403,"about_ca_topic_score_gemma":0.089722276,"teacher_disagreement_score":0.049614403,"about_ca_system_score_codex":0.0021585175,"about_ca_system_score_gemma":0.00739217,"threshold_uncertainty_score":0.09865123},"labels":[],"label_agreement":null},{"id":"W1776431790","doi":"","title":"Similarity patterns in words","year":2012,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Morpheme; Similarity (geometry); Sketch; Computer science; Natural language processing; Presentation (obstetrics); Linguistics; Artificial intelligence; Natural (archaeology); History; Philosophy","score_opus":0.016965553387721474,"score_gpt":0.2860389681800025,"score_spread":0.26907341479228103,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1776431790","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.44435063,0.010134548,0.44797105,0.0029689972,0.0008893165,0.00078168476,0.0079958495,0.0021943306,0.08271357],"genre_scores_gemma":[0.82418346,0.0019440101,0.15435202,0.0004954016,0.0003280437,0.00043571112,0.0049741934,0.00029174367,0.012995242],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9974874,0.00046436483,0.00038047493,0.0007963101,0.0006920204,0.00017931788],"domain_scores_gemma":[0.99701464,0.0011351323,0.00054514396,0.00052810356,0.00063598296,0.00014097967],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006447463,0.00039527833,0.0006241772,0.005362464,0.0013053636,0.003961004,0.00086432236,0.0010450643,0.009259278],"category_scores_gemma":[0.010139893,0.00035277076,0.0004648017,0.0069301627,0.0022569087,0.008028485,0.0022723672,0.0007519224,0.0024568255],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005220241,0.00011943731,0.016378747,0.0013127327,0.00013549492,0.0015024857,0.005998945,0.0032726307,0.020588985,0.5518796,0.012450224,0.3858388],"study_design_scores_gemma":[0.000041296782,0.00016651528,0.016920011,0.00022435175,0.00007186335,0.0026796465,0.0033303578,0.0141391,0.0060527395,0.8836985,0.07259339,0.0000821576],"about_ca_topic_score_codex":0.000884239,"about_ca_topic_score_gemma":0.0011474333,"teacher_disagreement_score":0.009259278,"about_ca_system_score_codex":0.00082942366,"about_ca_system_score_gemma":0.0005491899,"threshold_uncertainty_score":0.030975401},"labels":[],"label_agreement":null},{"id":"W1779067148","doi":"","title":"Monolingual Corpus-based MT using chunks","year":2005,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Lexicon; Computer science; Lemma (botany); Natural language processing; Artificial intelligence; Machine translation; Sentence; Translation (biology); Matching (statistics); Mathematics","score_opus":0.020994137742121007,"score_gpt":0.2966103952994104,"score_spread":0.2756162575572894,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1779067148","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023297729,0.0005678102,0.95490867,0.00020962326,0.0001711646,0.00036089422,0.0006221796,0.006294356,0.013567586],"genre_scores_gemma":[0.28169158,0.00045941662,0.6994397,0.00015391993,0.000121666664,0.00069852266,0.0026517063,0.0011721515,0.0136112375],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99859375,0.00067076297,0.00013216201,0.00030535113,0.00023957573,0.00005838003],"domain_scores_gemma":[0.9973168,0.0012133603,0.0001319165,0.0007692431,0.0005089107,0.000059624894],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016289373,0.00063745945,0.00076300104,0.0014527176,0.0008405926,0.0015475508,0.0011197029,0.00065579446,0.008178985],"category_scores_gemma":[0.0049605127,0.00049667346,0.00037030134,0.0022030147,0.00063376175,0.0039641187,0.0019493344,0.00067069876,0.0044419244],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011389623,0.00021002773,0.0022064364,0.0014186455,0.0002809273,0.0012547652,0.00303208,0.04346682,0.102937356,0.10618281,0.0212846,0.7165865],"study_design_scores_gemma":[0.0002336559,0.000579937,0.0021899142,0.00029272184,0.00031024922,0.0016233836,0.0012511847,0.5082593,0.15002008,0.13678479,0.19827327,0.00018153217],"about_ca_topic_score_codex":0.0019657754,"about_ca_topic_score_gemma":0.0025478278,"teacher_disagreement_score":0.008178985,"about_ca_system_score_codex":0.0005715221,"about_ca_system_score_gemma":0.00096144277,"threshold_uncertainty_score":0.027361393},"labels":[],"label_agreement":null},{"id":"W1779767490","doi":"10.3968/j.ccc.1923670020100604.008","title":"A Study of Chinese-English Code-switching in Chinese Sports News Reports","year":2011,"lang":"en","type":"article","venue":"Cross-cultural communication","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Code-switching; Markedness; Linguistics; Humanities; China; Art; History; Philosophy","score_opus":0.0251255109000642,"score_gpt":0.3413522114265649,"score_spread":0.3162267005265007,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1779767490","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99811125,0.00006205775,0.00009483447,0.000051193667,0.0000045647657,0.0000166041,0.000057816498,0.0000026228897,0.0015992044],"genre_scores_gemma":[0.9985096,0.00012458985,0.00021330683,0.000030458625,0.000011320578,0.000022763037,0.0001805837,0.000005453155,0.00090189354],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9986241,0.00039263794,0.00013892287,0.00022860843,0.00042781874,0.00018789936],"domain_scores_gemma":[0.9907194,0.00472741,0.0020242545,0.00043432132,0.0017032478,0.000391362],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018348667,0.0002898896,0.00020198745,0.0028471996,0.0017426154,0.0014069236,0.00049778895,0.00035480704,0.001541101],"category_scores_gemma":[0.0083521595,0.00022219741,0.00019411375,0.0038237188,0.00156492,0.0010912503,0.000958154,0.00056149426,0.00014997362],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024095862,0.00012968098,0.34309608,0.00036003912,0.000036468653,0.0020754994,0.6006602,0.00007781775,0.015371048,0.001450232,0.0006324318,0.03586955],"study_design_scores_gemma":[0.000011174923,0.00012364377,0.81788427,0.000048953174,0.00004191945,0.00061751617,0.16853353,0.0005773132,0.0039229863,0.00012650603,0.008065994,0.000046062745],"about_ca_topic_score_codex":0.08858809,"about_ca_topic_score_gemma":0.097406514,"teacher_disagreement_score":0.08858809,"about_ca_system_score_codex":0.0024821751,"about_ca_system_score_gemma":0.0017939148,"threshold_uncertainty_score":0.1761449},"labels":[],"label_agreement":null},{"id":"W178136429","doi":"","title":"Studies in Inuktitut grammar","year":2012,"lang":"en","type":"dissertation","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Grammar; Linguistics; Philosophy","score_opus":0.038777145765384595,"score_gpt":0.36637789569581036,"score_spread":0.32760074993042576,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W178136429","genre_codex":"empirical","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7256979,0.004120863,0.0011444854,0.0051149377,0.00012511053,0.000060043232,0.00013687878,0.000057166155,0.26354265],"genre_scores_gemma":[0.98826665,0.0009561928,0.0004708173,0.00027202157,0.000010638188,0.000017121118,0.00008303812,0.00005058809,0.00987299],"study_design_codex":"qualitative","study_design_gemma":"not_applicable","domain_scores_codex":[0.99776065,0.00050512014,0.00009030543,0.00037106738,0.00040097814,0.00087192986],"domain_scores_gemma":[0.99822634,0.00075950276,0.00016283111,0.00013417275,0.00042851668,0.00028870584],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016751505,0.00056834996,0.0007149159,0.00331418,0.019331824,0.0070243734,0.0024274348,0.0012700314,0.008760654],"category_scores_gemma":[0.0039568916,0.00044051453,0.00036295754,0.0065486734,0.012084464,0.0041982033,0.0040936433,0.0024246282,0.0005179708],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00004552236,0.0000679617,0.016706983,0.00015518151,0.000015320802,0.0010112679,0.49592188,0.00018183513,0.0006911524,0.4665822,0.0030420597,0.01557863],"study_design_scores_gemma":[0.000020378953,0.000043276534,0.08593068,0.00048731558,0.00006496398,0.0013943582,0.61101806,0.001557676,0.001136153,0.041870322,0.25634798,0.00012883457],"about_ca_topic_score_codex":0.84883356,"about_ca_topic_score_gemma":0.9147,"teacher_disagreement_score":0.84883356,"about_ca_system_score_codex":0.06859307,"about_ca_system_score_gemma":0.021176912,"threshold_uncertainty_score":0.4976799},"labels":[],"label_agreement":null},{"id":"W1782466391","doi":"10.24908/pceea.v0i0.4784","title":"AN AUTOMATED COURSE-SPECIFIC VOCABULARY IDENTIFICATION PROGRAM","year":2013,"lang":"en","type":"article","venue":"Proceedings of the Canadian Engineering Education Association (CEEA)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Course (navigation); Identification (biology); Vocabulary; Computer science; Natural language processing; Mathematics education; Engineering; Psychology; Linguistics; Biology","score_opus":0.004769322285108539,"score_gpt":0.2424800423436486,"score_spread":0.23771072005854005,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1782466391","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.109409094,0.00038987934,0.7632897,0.0003375753,0.0001879024,0.00071599014,0.0119381035,0.102475554,0.011256165],"genre_scores_gemma":[0.235459,0.00021010614,0.7097298,0.00017577439,0.00009049014,0.0006627627,0.031835403,0.0021356742,0.019700946],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993771,0.000062914,0.000047760495,0.00031339744,0.00014017444,0.00005869383],"domain_scores_gemma":[0.9990361,0.00031343903,0.00008149835,0.0001281682,0.00036838852,0.00007252997],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00046412542,0.0006807743,0.00070866715,0.0034727692,0.0008636374,0.0010627923,0.0010137782,0.0004655218,0.013283384],"category_scores_gemma":[0.002335347,0.00039909303,0.0006548354,0.0016378906,0.00020334494,0.001711156,0.0013806387,0.00063988596,0.0071010697],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002978389,0.00030901426,0.006847702,0.00022135364,0.000053890497,0.00015594209,0.00017042944,0.0023465909,0.063235864,0.0032587699,0.041146148,0.8819566],"study_design_scores_gemma":[0.00038419472,0.00052369334,0.027585004,0.00010834121,0.0002661409,0.0009448332,0.001133128,0.59952915,0.2330069,0.013990166,0.12240603,0.00012250902],"about_ca_topic_score_codex":0.0059679756,"about_ca_topic_score_gemma":0.009860605,"teacher_disagreement_score":0.013283384,"about_ca_system_score_codex":0.0006955864,"about_ca_system_score_gemma":0.0019058888,"threshold_uncertainty_score":0.04443729},"labels":[],"label_agreement":null},{"id":"W178463821","doi":"","title":"Improving Query Translation for CLIR Using Statistical Models","year":2017,"lang":"en","type":"article","venue":"International ACM SIGIR Conference on Research and Development in Information Retrieval","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Machine translation; Cross-language information retrieval; Translation (biology); Principle of maximum entropy; Cohesion (chemistry); Word (group theory); Phrase; Query expansion; Rule-based machine translation; Information retrieval; Linguistics","score_opus":0.21237798673519043,"score_gpt":0.42379239057131485,"score_spread":0.21141440383612442,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W178463821","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01353519,0.0017923476,0.97463834,0.0004733885,0.00015033843,0.00021995504,0.00037290345,0.006886255,0.0019312989],"genre_scores_gemma":[0.23555152,0.001963333,0.75297296,0.0008003571,0.00038993312,0.00054409256,0.0028142468,0.0011041822,0.0038593996],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9953832,0.002277506,0.00036055315,0.00073124317,0.0010261472,0.00022125951],"domain_scores_gemma":[0.9931166,0.003840061,0.00035033323,0.0009994896,0.0015916851,0.00010173105],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004340593,0.0018272393,0.0020579444,0.0030655188,0.0009881081,0.0018271105,0.0018438393,0.0013783396,0.004056699],"category_scores_gemma":[0.01367289,0.00075741636,0.0018775915,0.0044335146,0.0007797951,0.0037184535,0.001455295,0.0017789613,0.0064409613],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006263531,0.0006724112,0.0025937015,0.001138852,0.00031442236,0.00029956704,0.00047456942,0.13441154,0.050305497,0.01622867,0.025985967,0.7669484],"study_design_scores_gemma":[0.00006273175,0.0001662173,0.00041492804,0.000018079843,0.00006926767,0.00017883825,0.000098819175,0.97684866,0.009801519,0.00717249,0.005126138,0.000042385185],"about_ca_topic_score_codex":0.009847313,"about_ca_topic_score_gemma":0.012750944,"teacher_disagreement_score":0.009847313,"about_ca_system_score_codex":0.0012459358,"about_ca_system_score_gemma":0.0024008981,"threshold_uncertainty_score":0.022955477},"labels":[],"label_agreement":null},{"id":"W1789152958","doi":"10.1007/3-540-36456-0_29","title":"Augmenting WordNet’s Structure Using LDOCE","year":2003,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"WordNet; Computer science; Artificial intelligence; Information retrieval; Natural language processing","score_opus":0.017050046321081168,"score_gpt":0.26547428046679383,"score_spread":0.24842423414571266,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1789152958","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.119195975,0.0014573911,0.7829776,0.0014888996,0.0012453707,0.000455608,0.010909813,0.031106008,0.051163353],"genre_scores_gemma":[0.3885991,0.0011704993,0.5697265,0.0005647033,0.0002874111,0.0002928987,0.02165963,0.0024268185,0.015272421],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995931,0.00010244038,0.000045688124,0.0001168395,0.00010242582,0.0000395515],"domain_scores_gemma":[0.997733,0.0008931081,0.00009185165,0.0006824752,0.00051905523,0.00008049131],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00066302594,0.0008074555,0.00085653376,0.0031582427,0.000699974,0.0014950032,0.00080865977,0.0006042617,0.01386213],"category_scores_gemma":[0.0043552364,0.0005438982,0.00066636974,0.0027196023,0.00053145643,0.005342075,0.0024324385,0.0011831524,0.006350989],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00045673866,0.00032260496,0.0027141045,0.00095163967,0.00009464834,0.0007177447,0.0007705977,0.012410698,0.02920681,0.04009456,0.032177962,0.8800819],"study_design_scores_gemma":[0.00022395741,0.00046422117,0.0037578687,0.0006227213,0.000458358,0.0010331734,0.0016254394,0.31943133,0.06581292,0.24640904,0.36001718,0.00014370105],"about_ca_topic_score_codex":0.0037092576,"about_ca_topic_score_gemma":0.010392506,"teacher_disagreement_score":0.01386213,"about_ca_system_score_codex":0.00045880463,"about_ca_system_score_gemma":0.0010848478,"threshold_uncertainty_score":0.046373427},"labels":[],"label_agreement":null},{"id":"W179314280","doi":"","title":"Discriminative Reranking for Machine Translation","year":2004,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":207,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Machine translation; Discriminative model; NIST; Computer science; Artificial intelligence; Metric (unit); Evaluation of machine translation; Natural language processing; Baseline (sea); Translation (biology); BLEU; Sentence; Language model; Machine learning; Example-based machine translation; Machine translation software usability","score_opus":0.025688929787255024,"score_gpt":0.30283514733615213,"score_spread":0.2771462175488971,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W179314280","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0024889682,0.0015999957,0.9936591,0.00016176522,0.00010405266,0.000031008727,0.00006571405,0.0009137604,0.0009755775],"genre_scores_gemma":[0.15013194,0.0027156041,0.8389713,0.00026261405,0.0008632711,0.00025659814,0.0009915043,0.00030610655,0.005500982],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99776673,0.0011533904,0.00011328453,0.00037285162,0.0005090286,0.00008476605],"domain_scores_gemma":[0.9957783,0.0025222925,0.00034815056,0.0007077409,0.00056348694,0.000079987236],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021975436,0.00116044,0.0015387381,0.001607817,0.000749138,0.0010805967,0.0012943742,0.0010835289,0.0031393706],"category_scores_gemma":[0.008533552,0.00039573718,0.0004943817,0.0031818151,0.0009679261,0.0023079123,0.00089835434,0.0017100142,0.002725045],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014445678,0.00012416214,0.0004900495,0.0005610146,0.000080528225,0.00014798602,0.00013503326,0.107424,0.011178084,0.07019607,0.014410872,0.7951078],"study_design_scores_gemma":[0.00007816669,0.00015074715,0.00042797907,0.00003421875,0.000037111247,0.00025770802,0.000026836764,0.82965165,0.007926566,0.14272198,0.018639475,0.00004756615],"about_ca_topic_score_codex":0.0016438008,"about_ca_topic_score_gemma":0.0028926216,"teacher_disagreement_score":0.0031393706,"about_ca_system_score_codex":0.0006617381,"about_ca_system_score_gemma":0.0009966285,"threshold_uncertainty_score":0.011621892},"labels":[],"label_agreement":null},{"id":"W1793630844","doi":"","title":"Complete Complimentary Results Report of the MARF's NLP Approach to the DEFT 2010 Competition","year":2010,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Competition (biology); Identification (biology); Artificial intelligence; Natural language processing; Biology","score_opus":0.061150943746997054,"score_gpt":0.20095438715371444,"score_spread":0.13980344340671738,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1793630844","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.107293166,0.009096927,0.04586711,0.012021667,0.017090425,0.0027383529,0.52166903,0.06583195,0.21839136],"genre_scores_gemma":[0.094385564,0.0008591676,0.036492255,0.0023422237,0.0013365151,0.0018661693,0.77627885,0.009386208,0.07705303],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9730397,0.008186777,0.0013340097,0.0038447944,0.011219922,0.0023748656],"domain_scores_gemma":[0.9673209,0.008917196,0.0006592888,0.007914121,0.0116650835,0.0035233432],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.019842247,0.004966388,0.0032730089,0.006099464,0.004491181,0.0069613196,0.0031972162,0.003738763,0.062411975],"category_scores_gemma":[0.039875984,0.0011231658,0.0024411795,0.00420137,0.0017891072,0.005390291,0.006366484,0.0041128835,0.07387547],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010162735,0.0008623812,0.0017471764,0.0006861304,0.00030115334,0.00031605252,0.0002913105,0.0036291936,0.004591011,0.0014058264,0.9266995,0.05845396],"study_design_scores_gemma":[0.001548233,0.0016599135,0.029672204,0.00025882764,0.00041568757,0.0015866446,0.0014364806,0.037251025,0.023510534,0.007548912,0.89455795,0.00055365975],"about_ca_topic_score_codex":0.047976393,"about_ca_topic_score_gemma":0.074031025,"teacher_disagreement_score":0.062411975,"about_ca_system_score_codex":0.0042257886,"about_ca_system_score_gemma":0.0054593757,"threshold_uncertainty_score":0.20878887},"labels":[],"label_agreement":null},{"id":"W1794039122","doi":"10.1162/tacl_a_00080","title":"Learning to Understand Phrases by Embedding the Dictionary","year":2016,"lang":"en","type":"article","venue":"Transactions of the Association for Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":161,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs; Compute Canada","keywords":"Computer science; Natural language processing; Word embedding; Artificial intelligence; Bridging (networking); Embedding; Lexical semantics; Semantics (computer science); Word (group theory); Task (project management); Linguistics; Lexical item","score_opus":0.011370769647575057,"score_gpt":0.2784432967756493,"score_spread":0.2670725271280742,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1794039122","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.049868435,0.00041955593,0.94227546,0.00088109035,0.00012624558,0.00009159806,0.00065734383,0.0015324283,0.004147825],"genre_scores_gemma":[0.5329585,0.0012616996,0.45155883,0.0005519855,0.00012451001,0.000252118,0.0034711102,0.00041412996,0.009407132],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9997166,0.00009447097,0.000017935623,0.00011956871,0.000033149245,0.000018324217],"domain_scores_gemma":[0.99899954,0.0006065253,0.00008055145,0.00017682787,0.00010636727,0.000030307934],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00054015906,0.00075889815,0.00030664538,0.00064974936,0.00022575489,0.0009960396,0.0008028085,0.0008797898,0.005467405],"category_scores_gemma":[0.0039061792,0.00043316258,0.00067419535,0.000615159,0.0006714675,0.0055837473,0.0013560137,0.002013062,0.0023778828],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019021008,0.00017404326,0.004908606,0.00061604736,0.00014428198,0.00032834624,0.0011919372,0.08137792,0.027119959,0.088985935,0.015148487,0.7798142],"study_design_scores_gemma":[0.000029749282,0.00012410925,0.0011325981,0.00009771817,0.00005069407,0.00029957105,0.00038635978,0.807824,0.0090639265,0.16547418,0.015486087,0.000030971056],"about_ca_topic_score_codex":0.0010278564,"about_ca_topic_score_gemma":0.0020932974,"teacher_disagreement_score":0.005467405,"about_ca_system_score_codex":0.00037008413,"about_ca_system_score_gemma":0.00046914464,"threshold_uncertainty_score":0.01829034},"labels":[],"label_agreement":null},{"id":"W1798186798","doi":"","title":"Transliteration Experiments on Chinese and Arabic","year":2012,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Transliteration; Pinyin; Computer science; Natural language processing; Arabic; Artificial intelligence; Context (archaeology); Romanization; Syllable; Linguistics; Representation (politics); Speech recognition; Chinese characters; History","score_opus":0.01252840988619606,"score_gpt":0.2999732166485898,"score_spread":0.2874448067623937,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1798186798","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.921971,0.0007437654,0.04745148,0.0004124589,0.00031263218,0.0008678871,0.002887786,0.0048449296,0.020507991],"genre_scores_gemma":[0.8886859,0.0004977045,0.09153333,0.00029630485,0.00006967595,0.0006879089,0.0050581372,0.0014662307,0.011704732],"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99855965,0.00055775384,0.00022652055,0.00027802517,0.00025843788,0.00011963667],"domain_scores_gemma":[0.98919934,0.0070411237,0.00036346712,0.0012776009,0.0018982808,0.00022021103],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011336731,0.0012523851,0.0010250102,0.0005733794,0.0009869932,0.00073932705,0.0010351697,0.0008088501,0.013782591],"category_scores_gemma":[0.012006671,0.0003472349,0.00049728766,0.0013748837,0.00052969326,0.0015603557,0.0010837498,0.0009933431,0.0033060452],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0092937,0.0021005273,0.0031882722,0.004869632,0.00032823416,0.004022121,0.014340433,0.059962146,0.4398436,0.006287329,0.020380097,0.4353839],"study_design_scores_gemma":[0.001392777,0.0035783288,0.009260746,0.00020433971,0.00041564537,0.0014880045,0.00474593,0.15060036,0.7768424,0.004306384,0.046773683,0.0003913287],"about_ca_topic_score_codex":0.0056124125,"about_ca_topic_score_gemma":0.003728304,"teacher_disagreement_score":0.013782591,"about_ca_system_score_codex":0.00051651127,"about_ca_system_score_gemma":0.00051196915,"threshold_uncertainty_score":0.04610735},"labels":[],"label_agreement":null},{"id":"W1806995473","doi":"10.1007/11799573_42","title":"Learning Semantic Parsers: A Constraint Handling Rule Approach","year":2006,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Parsing; Natural language processing; Artificial intelligence; Semantic compression; Semantic role labeling; Natural language; Semantic computing; Top-down parsing; Bottom-up parsing; Natural language understanding; Semantic analysis (machine learning); Frame (networking); Process (computing); S-attributed grammar; Word (group theory); Programming language; Semantic Web; Semantic technology; Linguistics","score_opus":0.013997348473425222,"score_gpt":0.24235615697399668,"score_spread":0.22835880850057144,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1806995473","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0054439665,0.00022066355,0.985535,0.0003586929,0.000051682004,0.0001469384,0.0007500236,0.004934599,0.0025583913],"genre_scores_gemma":[0.0515501,0.00032158048,0.9393959,0.00026934716,0.00008229068,0.00022107773,0.0045205136,0.0008321791,0.0028071173],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9961926,0.0009295369,0.00038669142,0.0012649081,0.001006086,0.00022022426],"domain_scores_gemma":[0.9868466,0.009431124,0.00041331927,0.0016328443,0.0014691018,0.00020704205],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0041723833,0.0019430694,0.0024999613,0.0032048938,0.001421131,0.004144214,0.008884546,0.0030948427,0.014299526],"category_scores_gemma":[0.017959775,0.0022857406,0.003932702,0.005187478,0.0018757265,0.009979146,0.0037633698,0.005698678,0.005707631],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000204426,0.0004601518,0.0017797388,0.0004263059,0.00026046883,0.0005259989,0.00026894826,0.06462914,0.0042686993,0.049533132,0.026666373,0.85097665],"study_design_scores_gemma":[0.0000957935,0.00008638422,0.00051917945,0.00010929928,0.00026056138,0.00028737888,0.00019120798,0.78854656,0.009739836,0.18654186,0.013544208,0.0000777345],"about_ca_topic_score_codex":0.004541285,"about_ca_topic_score_gemma":0.008032367,"teacher_disagreement_score":0.014299526,"about_ca_system_score_codex":0.0012670845,"about_ca_system_score_gemma":0.0025761225,"threshold_uncertainty_score":0.04783666},"labels":[],"label_agreement":null},{"id":"W180865882","doi":"","title":"Combining Lexical and Syntactic Features for Supervised Word Sense Disambiguation","year":2004,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":29,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Word (group theory); Word-sense disambiguation; SemEval; Feature (linguistics); Context (archaeology); Part of speech; Lexical analysis; Linguistics; WordNet","score_opus":0.01729562365414278,"score_gpt":0.28532037099738894,"score_spread":0.26802474734324616,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W180865882","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.37417498,0.0032796208,0.607722,0.0009158286,0.00038859196,0.0003132665,0.00068707165,0.0060075372,0.0065110372],"genre_scores_gemma":[0.7719494,0.00039841238,0.22491814,0.00021479139,0.00023346244,0.00017642714,0.0010472725,0.00021097228,0.00085108145],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9965835,0.0016442175,0.00027163932,0.00052445586,0.00082083925,0.00015547725],"domain_scores_gemma":[0.9914112,0.0058656735,0.00050294446,0.00067784963,0.0012930826,0.0002492112],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00477004,0.0012349555,0.0014519106,0.003411341,0.0011918566,0.0014392643,0.000977967,0.0012322406,0.0010984349],"category_scores_gemma":[0.011343823,0.000580233,0.0009761044,0.0022438192,0.0009201799,0.0039552497,0.0019329627,0.0012318428,0.0011101924],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010087556,0.00089587586,0.01544524,0.00046662186,0.0008125133,0.00032500358,0.00035814242,0.054494668,0.04676992,0.002132494,0.0054393667,0.8718514],"study_design_scores_gemma":[0.00017455024,0.00079004414,0.014308895,0.000086819826,0.0007873409,0.00050512893,0.00033727745,0.90428287,0.042584382,0.031934068,0.004005995,0.0002026522],"about_ca_topic_score_codex":0.0012766734,"about_ca_topic_score_gemma":0.004166377,"teacher_disagreement_score":0.00477004,"about_ca_system_score_codex":0.00041801613,"about_ca_system_score_gemma":0.0007135981,"threshold_uncertainty_score":0.025226712},"labels":[],"label_agreement":null},{"id":"W1822485886","doi":"10.3968/j.ccc.1923670020060204.012","title":"Corpus in Foreign Language Teaching and Research","year":2010,"lang":"en","type":"article","venue":"Cross-cultural communication","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Corpus linguistics; Humanities; Linguistics; Foreign language; Language education; Applied linguistics; Art; Philosophy","score_opus":0.033126452160450195,"score_gpt":0.4088849710136319,"score_spread":0.37575851885318173,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1822485886","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.027900081,0.1927968,0.29652086,0.08119479,0.010041575,0.0019315254,0.005029851,0.0019141926,0.3826703],"genre_scores_gemma":[0.4559325,0.11136606,0.3228718,0.010435202,0.00699035,0.0057608937,0.0064846654,0.0026802795,0.07747823],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.97548234,0.018218013,0.0013221562,0.00213661,0.0024343184,0.00040645362],"domain_scores_gemma":[0.9121053,0.06839444,0.002672536,0.0090262685,0.0062773866,0.001524067],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.026484713,0.0007045312,0.0012473601,0.012250607,0.0050036325,0.014054882,0.002351033,0.0025330274,0.021979311],"category_scores_gemma":[0.056372616,0.0006245896,0.0006359196,0.01919558,0.01232736,0.017859265,0.008241182,0.0040186895,0.0034340562],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000074916206,0.00005480694,0.001583246,0.002211719,0.0000453149,0.000434919,0.014460876,0.00081847,0.0005658314,0.7146429,0.037496615,0.22761048],"study_design_scores_gemma":[0.000028197122,0.000049346527,0.002001364,0.005198109,0.000032131873,0.0004228457,0.010709399,0.0015749793,0.0012691029,0.23294863,0.74570054,0.00006539554],"about_ca_topic_score_codex":0.0068946187,"about_ca_topic_score_gemma":0.005895337,"teacher_disagreement_score":0.026484713,"about_ca_system_score_codex":0.007767472,"about_ca_system_score_gemma":0.0116619,"threshold_uncertainty_score":0.1400662},"labels":[],"label_agreement":null},{"id":"W1824515224","doi":"10.1007/978-3-540-30498-2_32","title":"Coordination Revisited – A Constraint Handling Rule Approach","year":2004,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Grammar; Programming language; Constraint (computer-aided design); Semantics (computer science); Theoretical computer science; Artificial intelligence; Linguistics; Mathematics","score_opus":0.014253147288650478,"score_gpt":0.25316424316613245,"score_spread":0.23891109587748197,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1824515224","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00522213,0.0009527699,0.8537078,0.0037904838,0.0012083112,0.00017952776,0.00034228512,0.0007467145,0.13385002],"genre_scores_gemma":[0.18337403,0.0017068994,0.74208784,0.0018505085,0.0011101663,0.00035761087,0.0008973749,0.0015128535,0.06710263],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9943461,0.0013217825,0.0004636394,0.0014809968,0.0019767869,0.00041069728],"domain_scores_gemma":[0.9907319,0.004154005,0.00039516858,0.0029583413,0.0014823346,0.00027822086],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004150406,0.0010545941,0.0015624543,0.0024749872,0.002561105,0.007220786,0.011655065,0.0033345877,0.030105293],"category_scores_gemma":[0.015627613,0.0013379921,0.0026669835,0.0039873095,0.003909136,0.013631468,0.0039201244,0.0065329173,0.004779156],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005677232,0.000059069975,0.00017088748,0.00018395319,0.0000464199,0.00063163583,0.0003772571,0.0044553285,0.0012438513,0.9137376,0.014036138,0.0650012],"study_design_scores_gemma":[0.00006722802,0.000035874302,0.00016666444,0.00019487424,0.00013705716,0.0007069994,0.00033971408,0.07428965,0.006078138,0.8289087,0.0889884,0.000086752385],"about_ca_topic_score_codex":0.00850305,"about_ca_topic_score_gemma":0.0066141705,"teacher_disagreement_score":0.030105293,"about_ca_system_score_codex":0.0015680097,"about_ca_system_score_gemma":0.0024555917,"threshold_uncertainty_score":0.10071224},"labels":[],"label_agreement":null},{"id":"W1827836154","doi":"","title":"JIL: The Response","year":2013,"lang":"en","type":"article","venue":"Workplace","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science","score_opus":0.007176802010256527,"score_gpt":0.24241994813133927,"score_spread":0.23524314612108274,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1827836154","genre_codex":"commentary","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009573985,0.0037700343,0.06299829,0.4454668,0.07183863,0.0003530836,0.0040431293,0.035071414,0.36688468],"genre_scores_gemma":[0.12112778,0.0015201678,0.017424097,0.14556946,0.016626444,0.00047466205,0.0039510536,0.008259615,0.68504673],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99286747,0.0029125565,0.00031055688,0.0007453311,0.0021263107,0.0010377327],"domain_scores_gemma":[0.9837833,0.004891771,0.0007479276,0.0030290303,0.0043030633,0.003244946],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0062093874,0.0010975066,0.0006786353,0.0016268502,0.003943968,0.008361537,0.0020562997,0.006262426,0.11190544],"category_scores_gemma":[0.026675662,0.0004919373,0.00075172784,0.0010905365,0.002451486,0.007353665,0.008404096,0.008340849,0.07619525],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011259775,0.000053097756,0.00055713597,0.00010763762,0.0000054310167,0.00012860858,0.00094516046,0.00006832897,0.00092839974,0.016705634,0.93577063,0.044617355],"study_design_scores_gemma":[0.000025630412,0.00005738523,0.00058387924,0.00010906911,0.0000068447594,0.00012710143,0.0019226577,0.00069574476,0.0012280437,0.010794924,0.9844126,0.000036246576],"about_ca_topic_score_codex":0.0019968287,"about_ca_topic_score_gemma":0.0025049103,"teacher_disagreement_score":0.11190544,"about_ca_system_score_codex":0.0019007553,"about_ca_system_score_gemma":0.0025940384,"threshold_uncertainty_score":0.37436098},"labels":[],"label_agreement":null},{"id":"W18278838","doi":"","title":"Automatic Extraction of Synonymy Information","year":2006,"lang":"en","type":"article","venue":"The Journal of Rheumatology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Parsing; Information extraction; Natural language processing; Information retrieval; Artificial intelligence; Quality (philosophy)","score_opus":0.005241832304746508,"score_gpt":0.24820453474682852,"score_spread":0.242962702442082,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W18278838","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12340007,0.010298724,0.69664997,0.003664546,0.0020616471,0.002372249,0.11965121,0.022000069,0.01990163],"genre_scores_gemma":[0.23037371,0.00294554,0.6501672,0.00048164497,0.0006758657,0.0012142423,0.10871089,0.0010441446,0.0043867095],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99526864,0.00094974024,0.0010189901,0.0013471043,0.0011760435,0.00023940639],"domain_scores_gemma":[0.98728406,0.006549617,0.0013849337,0.0012566423,0.003173106,0.00035163466],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024596867,0.0023958087,0.002092429,0.02015198,0.0017164504,0.0032282376,0.0016784271,0.0019783052,0.008435115],"category_scores_gemma":[0.01658988,0.0007945215,0.0021316549,0.011344572,0.0008293138,0.0065260534,0.0034649034,0.0020833127,0.0055767167],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00089151296,0.0003169825,0.007681805,0.0071711875,0.00048666573,0.004154674,0.0036642214,0.0034108697,0.092839025,0.02559885,0.09077977,0.7630044],"study_design_scores_gemma":[0.000601199,0.0008366228,0.04434777,0.0023432558,0.0015557568,0.013616291,0.009219035,0.18002847,0.092303544,0.14293581,0.5114185,0.0007938022],"about_ca_topic_score_codex":0.0023329929,"about_ca_topic_score_gemma":0.0028489504,"teacher_disagreement_score":0.02015198,"about_ca_system_score_codex":0.001165782,"about_ca_system_score_gemma":0.0032559433,"threshold_uncertainty_score":0.02821821},"labels":[],"label_agreement":null},{"id":"W1828773426","doi":"10.1007/978-3-642-21043-3_23","title":"Correcting Different Types of Errors in Texts","year":2011,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Spelling; Computer science; Punctuation; Natural language processing; Artificial intelligence; Word (group theory); Verb; Error detection and correction; Speech recognition; Linguistics; Algorithm","score_opus":0.018046131009018113,"score_gpt":0.2580295709324431,"score_spread":0.239983439923425,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1828773426","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12493015,0.0049580364,0.81214654,0.0034650504,0.003032646,0.00048101036,0.003149095,0.02652788,0.021309525],"genre_scores_gemma":[0.39816165,0.002890842,0.5376267,0.00073186244,0.001115375,0.00020465495,0.005059033,0.008769213,0.04544069],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9899877,0.001809357,0.0013025166,0.002511787,0.0038236403,0.0005651106],"domain_scores_gemma":[0.94145966,0.028626256,0.0043698796,0.015514842,0.0094767595,0.0005525587],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0039778994,0.001921792,0.0014781485,0.0048898472,0.0014322611,0.0034017882,0.0035158477,0.0034023442,0.011001751],"category_scores_gemma":[0.052776873,0.0009764024,0.0013884915,0.004380479,0.0019302085,0.008025542,0.0041776374,0.002351731,0.005535632],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007215537,0.00016652729,0.004845657,0.0015074398,0.00017089896,0.0010180401,0.0022023655,0.0084743695,0.025385434,0.023783451,0.020311639,0.9114126],"study_design_scores_gemma":[0.00021786321,0.0007492895,0.007932932,0.0018025644,0.0015587542,0.006602221,0.0021742962,0.14387715,0.4724812,0.19059995,0.1715314,0.0004724487],"about_ca_topic_score_codex":0.0009564962,"about_ca_topic_score_gemma":0.0011574565,"teacher_disagreement_score":0.011001751,"about_ca_system_score_codex":0.00085706444,"about_ca_system_score_gemma":0.0012127918,"threshold_uncertainty_score":0.036804497},"labels":[],"label_agreement":null},{"id":"W1829370927","doi":"10.1007/978-3-540-74628-7_11","title":"Disambiguating Hypernym Relations for Roget’s Thesaurus","year":2007,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Thesaurus; Computer science; Natural language processing; Relation (database); Information retrieval; Artificial intelligence; Semantic relation; Psychology; Data mining; Neuroscience","score_opus":0.03043405292626822,"score_gpt":0.2972423155458985,"score_spread":0.26680826261963025,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1829370927","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.35613334,0.00676079,0.57480675,0.0018972057,0.0013297774,0.00063323206,0.0074973,0.0067561436,0.044185437],"genre_scores_gemma":[0.5164001,0.0015777039,0.4599619,0.00027563627,0.00023405021,0.00017185434,0.012924962,0.0010163225,0.0074374652],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.998262,0.00047993308,0.00032009027,0.00045849633,0.00035614477,0.00012335872],"domain_scores_gemma":[0.99808997,0.0009024521,0.00016540996,0.00033686348,0.00037894797,0.00012639255],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001453222,0.00073338836,0.0009612683,0.010351745,0.0022019132,0.003066574,0.0011379578,0.0014161236,0.0066886237],"category_scores_gemma":[0.005784191,0.00076845405,0.0011695386,0.0049823713,0.001128252,0.006979863,0.004575308,0.0015602517,0.0033018512],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00085779343,0.00035360805,0.008909469,0.0021468643,0.00032617123,0.004291156,0.013700365,0.003423313,0.096612774,0.22333746,0.0430162,0.60302484],"study_design_scores_gemma":[0.00028236993,0.00031882332,0.019168505,0.0016098233,0.0006461013,0.009808038,0.012138826,0.07944135,0.070795015,0.33421037,0.47117937,0.00040136746],"about_ca_topic_score_codex":0.002112591,"about_ca_topic_score_gemma":0.0047261794,"teacher_disagreement_score":0.010351745,"about_ca_system_score_codex":0.0008853273,"about_ca_system_score_gemma":0.0012006093,"threshold_uncertainty_score":0.022375703},"labels":[],"label_agreement":null},{"id":"W1836529743","doi":"10.1007/978-3-642-13059-5_27","title":"Phrase-Based Statistical Machine Translation for a Low-Density Language Pair","year":2010,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Phrase; Bengali; Machine translation; Transliteration; Natural language processing; Artificial intelligence; Evaluation of machine translation; Component (thermodynamics); Set (abstract data type); Translation (biology); BLEU; Test set; Machine translation software usability; Example-based machine translation; Programming language","score_opus":0.012263877094776969,"score_gpt":0.27085807556316316,"score_spread":0.2585941984683862,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1836529743","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.027699137,0.0006911585,0.9553019,0.000725019,0.00055839843,0.00017953465,0.0016696943,0.007367571,0.005807514],"genre_scores_gemma":[0.24787895,0.0005856581,0.7367581,0.0003367335,0.00040771833,0.00032962018,0.0056094,0.0012926603,0.0068011074],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99838245,0.00055699464,0.0001514465,0.00039078013,0.0003877903,0.00013058941],"domain_scores_gemma":[0.9962685,0.0018229203,0.00017153993,0.0006287231,0.0010195732,0.000088683235],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001373229,0.0010341874,0.001384318,0.0014847966,0.0012514485,0.0021435155,0.0011792628,0.0016706071,0.012077903],"category_scores_gemma":[0.0058002686,0.00068627996,0.0012088203,0.0022629155,0.0005924942,0.0025049425,0.0021397125,0.0018951936,0.012789345],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008109024,0.00037728323,0.0011878301,0.0011610723,0.00020801241,0.0019505463,0.00070191576,0.028293904,0.10425995,0.06224312,0.0416434,0.75716203],"study_design_scores_gemma":[0.00016296134,0.0005603044,0.0012673319,0.00009297033,0.00022395542,0.0020099948,0.00044781112,0.78729355,0.08080306,0.09934193,0.027656527,0.00013961918],"about_ca_topic_score_codex":0.0010657472,"about_ca_topic_score_gemma":0.0021200548,"teacher_disagreement_score":0.012077903,"about_ca_system_score_codex":0.0004777949,"about_ca_system_score_gemma":0.0015493939,"threshold_uncertainty_score":0.040404618},"labels":[],"label_agreement":null},{"id":"W1840193652","doi":"10.1007/978-3-642-21043-3_42","title":"Cross-Lingual Word Sense Disambiguation for Languages with Scarce Resources","year":2011,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Computer science; Persian; Natural language processing; Artificial intelligence; Context (archaeology); Word (group theory); Meaning (existential); Word-sense disambiguation; Resource (disambiguation); SemEval; Linguistics; WordNet","score_opus":0.017577510790669167,"score_gpt":0.2883880051316756,"score_spread":0.27081049434100646,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1840193652","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.107076176,0.014129452,0.83112097,0.0017134261,0.0020502214,0.00021417222,0.007029885,0.0136694005,0.02299617],"genre_scores_gemma":[0.43604696,0.003809999,0.523636,0.00069337053,0.00034333838,0.00020768418,0.021619087,0.002449935,0.01119363],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99803346,0.00044236446,0.0004300254,0.0005484982,0.0003436865,0.00020193539],"domain_scores_gemma":[0.9976445,0.0010870554,0.00015706931,0.0005047141,0.0004879345,0.00011869393],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021994286,0.0013356826,0.001691242,0.0063707656,0.001993199,0.0039414866,0.0017602467,0.0012155911,0.006741445],"category_scores_gemma":[0.0047980007,0.0009686295,0.0015351526,0.00690153,0.0010965405,0.009110761,0.006347803,0.0014172212,0.0051455367],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005389009,0.00016335267,0.004849537,0.0016280314,0.00039063845,0.0015500361,0.0022625267,0.005165112,0.03804173,0.028722765,0.04881488,0.8678725],"study_design_scores_gemma":[0.00020156664,0.00033486824,0.01971807,0.000980112,0.00094942586,0.0074553317,0.013228843,0.22586441,0.13819143,0.31725317,0.27523136,0.0005913154],"about_ca_topic_score_codex":0.0025642859,"about_ca_topic_score_gemma":0.0072945077,"teacher_disagreement_score":0.006741445,"about_ca_system_score_codex":0.0005920203,"about_ca_system_score_gemma":0.0018185041,"threshold_uncertainty_score":0.022552371},"labels":[],"label_agreement":null},{"id":"W1843172357","doi":"10.1007/978-3-540-73078-1_41","title":"Conceptualizing Student Models for ICALL","year":2007,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":32,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"","keywords":"Computer science; Perspective (graphical); Language acquisition; Representation (politics); Language model; Artificial intelligence; Second-language acquisition; Natural language processing; Human–computer interaction; Linguistics; Mathematics education; Psychology","score_opus":0.04332132384124821,"score_gpt":0.3269412944556134,"score_spread":0.28361997061436517,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1843172357","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015577675,0.0002576504,0.840805,0.0037581432,0.0001231788,0.00013704396,0.0006299962,0.0017666009,0.13694473],"genre_scores_gemma":[0.6838124,0.0004517801,0.2620934,0.00077490916,0.00017707399,0.00043608167,0.0024314055,0.00066861894,0.04915438],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99862075,0.00056799303,0.00007628352,0.0002607952,0.00027771806,0.0001963744],"domain_scores_gemma":[0.99709237,0.0013214,0.00021385877,0.00052517635,0.00055405236,0.0002931981],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018965472,0.00073895475,0.00033646185,0.0012831661,0.0014613221,0.005911313,0.0020280078,0.0020826457,0.014410665],"category_scores_gemma":[0.0065760855,0.0005292912,0.0012239272,0.0011566988,0.0025267643,0.01068827,0.0028200115,0.0030304529,0.0035158058],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000018386505,0.000033170756,0.00082176976,0.00003264301,0.000008928205,0.00007084801,0.0014746516,0.0065384125,0.00021474948,0.97007006,0.0050181854,0.015698168],"study_design_scores_gemma":[0.000018179364,0.000023299484,0.00029500466,0.00008217484,0.000028110899,0.00016702389,0.001260156,0.10975766,0.0011253515,0.82089484,0.06632332,0.00002487219],"about_ca_topic_score_codex":0.0070857955,"about_ca_topic_score_gemma":0.009761297,"teacher_disagreement_score":0.014410665,"about_ca_system_score_codex":0.0028853607,"about_ca_system_score_gemma":0.0019545793,"threshold_uncertainty_score":0.048208535},"labels":[],"label_agreement":null},{"id":"W1846534677","doi":"10.7287/peerj.preprints.1459v1","title":"Using machine translation for converting <i>Python</i> <i>2</i> to <i>Python</i> <i>3</i> code","year":2015,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Python (programming language); Programming language; Computer science; Machine translation; Artificial intelligence; Natural language processing","score_opus":0.10222477695372965,"score_gpt":0.3466878108172054,"score_spread":0.24446303386347573,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1846534677","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13698573,0.00070632366,0.7086783,0.001753292,0.0015777688,0.0016021436,0.015602085,0.11154632,0.021548001],"genre_scores_gemma":[0.27654526,0.00038160098,0.66517574,0.0006071781,0.000110061716,0.0012327398,0.03382822,0.013933447,0.008185706],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9894609,0.0049968283,0.001224112,0.0017763568,0.002041451,0.0005003535],"domain_scores_gemma":[0.9725112,0.010023537,0.0017501885,0.00765922,0.0076518976,0.0004039865],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0069878786,0.0017921823,0.0008386903,0.003050634,0.0017234259,0.002968628,0.0009857547,0.0010211573,0.009793827],"category_scores_gemma":[0.045834523,0.000814504,0.0011429078,0.0036622938,0.0008531046,0.0039029988,0.002545903,0.0028102468,0.010901505],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007750139,0.00069538114,0.017152704,0.00252781,0.00043447877,0.0008322551,0.0025891585,0.013902543,0.050988458,0.008721095,0.07454775,0.82683337],"study_design_scores_gemma":[0.00032775511,0.0013201329,0.031421434,0.00086601404,0.0003901937,0.0027818433,0.0030919907,0.2626178,0.36253926,0.03759079,0.29638684,0.00066592987],"about_ca_topic_score_codex":0.0033986613,"about_ca_topic_score_gemma":0.005688782,"teacher_disagreement_score":0.009793827,"about_ca_system_score_codex":0.00089222915,"about_ca_system_score_gemma":0.0029012247,"threshold_uncertainty_score":0.036955893},"labels":[],"label_agreement":null},{"id":"W1848260265","doi":"","title":"Discriminative Instance Weighting for Domain Adaptation in Statistical Machine Translation","year":2010,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":211,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Discriminative model; Weighting; Computer science; Domain adaptation; Machine translation; Artificial intelligence; Relevance (law); Granularity; Phrase; Natural language processing; Domain (mathematical analysis); Machine learning; Range (aeronautics); Adaptation (eye); Translation (biology); Pattern recognition (psychology); Mathematics; Engineering","score_opus":0.018349085065491492,"score_gpt":0.29006638940078716,"score_spread":0.2717173043352957,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1848260265","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0054994617,0.0004119468,0.9916489,0.00009135697,0.00005222323,0.000043049935,0.000088894034,0.0014313003,0.0007329109],"genre_scores_gemma":[0.26858753,0.00080685643,0.7247429,0.00031756877,0.00030045712,0.00028439166,0.0011052071,0.00075635326,0.0030987835],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9982185,0.0009706583,0.00008846035,0.0003350178,0.00028856815,0.000098872486],"domain_scores_gemma":[0.9972121,0.0014791068,0.00016645537,0.0008140229,0.00024175525,0.00008650998],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027547472,0.00096902833,0.0012152616,0.0012973845,0.0005009207,0.0009891014,0.0018659437,0.0010863421,0.003015335],"category_scores_gemma":[0.008819309,0.00056982506,0.00076605,0.0026918205,0.0008518529,0.0031163252,0.0020299165,0.002578774,0.0023228752],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00034096985,0.00025919886,0.0013709932,0.00023815749,0.00018210587,0.0001385576,0.00018268336,0.18486777,0.021321915,0.03148202,0.008839818,0.7507758],"study_design_scores_gemma":[0.00002262174,0.000057813373,0.00031475286,0.000012638114,0.000027762211,0.00010272475,0.000023277968,0.94891965,0.006706495,0.03888143,0.00490518,0.000025732474],"about_ca_topic_score_codex":0.0016023681,"about_ca_topic_score_gemma":0.002718308,"teacher_disagreement_score":0.003015335,"about_ca_system_score_codex":0.0005895585,"about_ca_system_score_gemma":0.000628259,"threshold_uncertainty_score":0.0145686865},"labels":[],"label_agreement":null},{"id":"W1850097402","doi":"","title":"El enfoque basado en conocimiento para la extracción automática de palabras clave","year":2013,"lang":"es","type":"article","venue":"Computación y Sistemas","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Computer Research Institute of Montréal; Polytechnique Montréal","funders":"","keywords":"Humanities; Philosophy","score_opus":0.014297193705301013,"score_gpt":0.2998249585096376,"score_spread":0.28552776480433656,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1850097402","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11632585,0.0050761155,0.8055489,0.003869303,0.001769085,0.00046131693,0.0016669349,0.021619612,0.04366282],"genre_scores_gemma":[0.39421707,0.0028561982,0.5575758,0.0008456588,0.00036741645,0.00028131547,0.002730397,0.0024225323,0.038703606],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.997531,0.0006933758,0.00014895377,0.00059322827,0.0007957462,0.00023770123],"domain_scores_gemma":[0.99326926,0.0024098605,0.00023426469,0.0012679468,0.0026303204,0.00018846238],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031725017,0.0014394624,0.0008525549,0.002463812,0.0014603212,0.0038728956,0.001076553,0.0017824145,0.016391557],"category_scores_gemma":[0.015549669,0.00051740435,0.0008021264,0.0017298161,0.00087574054,0.0031567435,0.0014716912,0.0015552855,0.0055737826],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009889286,0.00024024802,0.010340811,0.0010920824,0.00008873025,0.00062582793,0.0015624004,0.0041775075,0.08975573,0.014629092,0.016764147,0.85973436],"study_design_scores_gemma":[0.00022652383,0.0009007795,0.039408702,0.0014308866,0.0011860395,0.0035232212,0.0053571425,0.16047591,0.36718097,0.048928604,0.37105477,0.0003264573],"about_ca_topic_score_codex":0.008319208,"about_ca_topic_score_gemma":0.009716902,"teacher_disagreement_score":0.016391557,"about_ca_system_score_codex":0.0009778421,"about_ca_system_score_gemma":0.0018032376,"threshold_uncertainty_score":0.0548352},"labels":[],"label_agreement":null},{"id":"W185181089","doi":"10.7202/1030015ar","title":"Multilinguisme et langages documentaires : le projet MACS en contexte européen","year":2015,"lang":"fr","type":"article","venue":"Documentation et bibliothèques","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Political science; Art","score_opus":0.046035099291230556,"score_gpt":0.37659023349747106,"score_spread":0.3305551342062405,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W185181089","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5548453,0.0054157884,0.070738025,0.010747116,0.00034346816,0.0006163118,0.0012268249,0.0014822034,0.354585],"genre_scores_gemma":[0.81100726,0.002321603,0.084986344,0.00052014185,0.00009983542,0.00040927468,0.0017042275,0.0004264233,0.098524965],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99378294,0.0032993816,0.00047745218,0.00082276075,0.0013199792,0.00029756626],"domain_scores_gemma":[0.9911628,0.0037884265,0.0005239138,0.001286101,0.0025437193,0.0006950538],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009097705,0.0003159603,0.00033067816,0.0021982472,0.0022858807,0.006666304,0.0007985956,0.0010209598,0.008030365],"category_scores_gemma":[0.010439587,0.00030151216,0.0003296272,0.0034658154,0.0026478637,0.0052526365,0.003795135,0.0010597702,0.0016633566],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00038330557,0.00027099447,0.018593045,0.0008708637,0.00005622448,0.0012270927,0.20752606,0.0018489431,0.016949475,0.23336054,0.012927124,0.5059864],"study_design_scores_gemma":[0.000048356636,0.00017468451,0.019442847,0.0004670152,0.000040980805,0.00069816585,0.055859264,0.0011577029,0.009617219,0.015131791,0.89731693,0.00004507158],"about_ca_topic_score_codex":0.03400454,"about_ca_topic_score_gemma":0.028843798,"teacher_disagreement_score":0.03400454,"about_ca_system_score_codex":0.00436788,"about_ca_system_score_gemma":0.008145227,"threshold_uncertainty_score":0.067613244},"labels":[],"label_agreement":null},{"id":"W1852496846","doi":"10.1111/tops.12211","title":"The Latent Structure of Dictionaries","year":2016,"lang":"en","type":"article","venue":"Topics in Cognitive Science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":40,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal; University of Ottawa","funders":"","keywords":"Rest (music); Word (group theory); Computer science; Core (optical fiber); Set (abstract data type); Categorization; Vertex (graph theory); Artificial intelligence; Natural language processing; Graph; Mathematics; Theoretical computer science","score_opus":0.017829470852915353,"score_gpt":0.3011776556249314,"score_spread":0.283348184772016,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1852496846","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.28152633,0.0013461361,0.64927036,0.0037822495,0.00018307441,0.00017596342,0.004420019,0.0017641995,0.057531714],"genre_scores_gemma":[0.9021036,0.0005239994,0.08587225,0.00031559978,0.00008155983,0.00019964304,0.0034244622,0.00041114812,0.0070678457],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99730504,0.0006687423,0.00025090468,0.0010189918,0.0005397015,0.00021661124],"domain_scores_gemma":[0.9919883,0.0033899664,0.0009131919,0.0021632605,0.0010920546,0.0004532352],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010831254,0.00045190516,0.0006305832,0.0027332897,0.0013443896,0.0051140036,0.0015340325,0.0010362368,0.010732479],"category_scores_gemma":[0.015089429,0.00087273173,0.001056789,0.0019993074,0.005070048,0.012204142,0.0037243122,0.0016081616,0.0020491097],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017460185,0.00006253123,0.0075636394,0.00041032632,0.0000884937,0.00023198301,0.004010972,0.0053924625,0.005751299,0.90678555,0.0030025172,0.066525586],"study_design_scores_gemma":[0.000039613697,0.00006448157,0.0047806213,0.00012858323,0.00009049773,0.00033125115,0.0012110142,0.04182143,0.0031220454,0.93310463,0.015248689,0.000057151803],"about_ca_topic_score_codex":0.003526358,"about_ca_topic_score_gemma":0.0032671124,"teacher_disagreement_score":0.010732479,"about_ca_system_score_codex":0.0018754042,"about_ca_system_score_gemma":0.0013189517,"threshold_uncertainty_score":0.035903692},"labels":[],"label_agreement":null},{"id":"W1853379672","doi":"10.1007/978-3-540-24840-8_52","title":"Two Set-Theoretic Approaches to the Semantics of Adjective-Noun Combinations","year":2004,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Windsor","funders":"","keywords":"Adjective; Noun; Syntax; Linguistics; Semantics (computer science); Hierarchy; Computer science; Set (abstract data type); Parallels; Simplicity; Natural language processing; Simple (philosophy); Artificial intelligence; Mathematics; Philosophy; Epistemology; Programming language","score_opus":0.03453039172299875,"score_gpt":0.265354845730622,"score_spread":0.23082445400762325,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1853379672","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012385342,0.0028675883,0.9217639,0.0051898607,0.0006968479,0.00013246563,0.00044718708,0.00040053265,0.056116287],"genre_scores_gemma":[0.42188835,0.002251498,0.5529362,0.0019338139,0.0019881425,0.000773824,0.0012024363,0.0004732475,0.016552515],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99377126,0.0021919543,0.00066306867,0.0010072283,0.0018674276,0.0004991646],"domain_scores_gemma":[0.99403834,0.0031733818,0.0004939133,0.0009697051,0.00088681595,0.0004378083],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004427365,0.0017471986,0.0020703883,0.007127675,0.0041240975,0.012411569,0.006432709,0.0044876155,0.011046898],"category_scores_gemma":[0.0102327205,0.0023737801,0.005484164,0.00648606,0.014425437,0.028277153,0.008623914,0.008093907,0.0013898562],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00001513451,0.000016079412,0.00004542707,0.00004163247,0.0000100820935,0.0000375976,0.0003732884,0.00035203827,0.00019480662,0.99477684,0.00049762294,0.0036393146],"study_design_scores_gemma":[0.000008914274,0.000010643167,0.00006939098,0.000021573976,0.000013349369,0.00008193358,0.00014532903,0.0027379296,0.00023231246,0.9930601,0.0036033608,0.000015137468],"about_ca_topic_score_codex":0.0016728585,"about_ca_topic_score_gemma":0.0021921583,"teacher_disagreement_score":0.012411569,"about_ca_system_score_codex":0.00409579,"about_ca_system_score_gemma":0.0018902968,"threshold_uncertainty_score":0.036955535},"labels":[],"label_agreement":null},{"id":"W1859439652","doi":"10.1558/cj.v20i2.227-244","title":"Bug Diagnosis By String Matching","year":2003,"lang":"en","type":"article","venue":"CALICO Journal","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Northern British Columbia","funders":"","keywords":"Computer science; Robustness (evolution); Artificial intelligence; String searching algorithm; Natural language processing; Theoretical computer science; Matching (statistics); Asynchronous communication; Edit distance; Sentence; Machine learning; Pattern matching","score_opus":0.010523460334949095,"score_gpt":0.2617802084241759,"score_spread":0.25125674808922677,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1859439652","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03558021,0.00018060833,0.9505365,0.00016724039,0.000046763325,0.00013226473,0.00050245685,0.011321655,0.0015323493],"genre_scores_gemma":[0.29551283,0.00020901898,0.6980832,0.00012820882,0.000024869325,0.0001597669,0.001873392,0.00083313923,0.0031755029],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99701273,0.00064627023,0.00041440479,0.0009151021,0.000854122,0.00015732834],"domain_scores_gemma":[0.99493706,0.0019868,0.00060649536,0.0014369456,0.00094449206,0.00008811324],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016063933,0.0008932222,0.0011659369,0.0035866022,0.00062105386,0.0018612419,0.0023610191,0.0013047694,0.0039657457],"category_scores_gemma":[0.0108565,0.00047218648,0.0012436528,0.002486511,0.0009463992,0.0031794794,0.0017996097,0.00068760995,0.0015833133],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00043145014,0.00015225851,0.013243857,0.0005156978,0.0001740549,0.0007949671,0.000651419,0.04683048,0.026240248,0.024540883,0.0056214635,0.8808032],"study_design_scores_gemma":[0.00007400114,0.00031072323,0.002510824,0.000074661526,0.00020278507,0.0011129209,0.0003000073,0.8424867,0.072470516,0.06456394,0.015810037,0.000082897895],"about_ca_topic_score_codex":0.0026412634,"about_ca_topic_score_gemma":0.002718003,"teacher_disagreement_score":0.0039657457,"about_ca_system_score_codex":0.00068314484,"about_ca_system_score_gemma":0.0013075611,"threshold_uncertainty_score":0.013266683},"labels":[],"label_agreement":null},{"id":"W1862411264","doi":"","title":"Mercure: Towards an Automatic E-mail Follow-up System","year":2003,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Corporation; Style (visual arts); Order (exchange); Computer science; Point (geometry); World Wide Web; Geography; Business; Political science; Law; Archaeology","score_opus":0.014196295768164767,"score_gpt":0.2663812810957288,"score_spread":0.252184985327564,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1862411264","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03957726,0.00073101564,0.60288244,0.0015097638,0.00018030782,0.00088946015,0.0023295535,0.34385023,0.008049988],"genre_scores_gemma":[0.13497347,0.00030765438,0.83798796,0.0009322437,0.0002798564,0.00044682345,0.008687856,0.0023892606,0.013994969],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9975418,0.000489301,0.00026129943,0.00064337184,0.00093731855,0.0001270449],"domain_scores_gemma":[0.99471277,0.0020297333,0.0007325872,0.0011198855,0.001045105,0.00035990303],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034983251,0.00089470425,0.0011963065,0.0034384325,0.001479256,0.0029752122,0.0034934396,0.0034707992,0.0073110927],"category_scores_gemma":[0.009094851,0.00086093385,0.0012219533,0.0012806869,0.00064822345,0.0060540177,0.0023604904,0.002508256,0.0068382896],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017163501,0.0012911631,0.011981604,0.00076058635,0.00018919261,0.0016011024,0.0016643266,0.007133731,0.060974438,0.021183643,0.20281814,0.6886858],"study_design_scores_gemma":[0.000627936,0.0009340475,0.014849141,0.00037620886,0.0004171594,0.002529704,0.00065690064,0.62129587,0.10727959,0.028067704,0.22260892,0.0003568916],"about_ca_topic_score_codex":0.0030906543,"about_ca_topic_score_gemma":0.0031432137,"teacher_disagreement_score":0.0073110927,"about_ca_system_score_codex":0.0010192717,"about_ca_system_score_gemma":0.0012893436,"threshold_uncertainty_score":0.02445811},"labels":[],"label_agreement":null},{"id":"W1862977469","doi":"10.1558/cj.v20i3.533-548","title":"Multiple Learner Errors and Meaningful Feedback","year":2003,"lang":"en","type":"article","venue":"CALICO Journal","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":51,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Focus (optics); Parsing; Vocabulary; Sentence; Natural language processing; Domain (mathematical analysis); Grammar; Artificial intelligence; Queue; Phrase; Rule-based machine translation; Linguistics; Programming language","score_opus":0.014920340237406234,"score_gpt":0.2540821634711794,"score_spread":0.23916182323377316,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1862977469","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.37042558,0.0014995455,0.56861895,0.0063922727,0.00036961003,0.0004017252,0.00042964914,0.01601931,0.035843343],"genre_scores_gemma":[0.91839397,0.0003772747,0.0653268,0.00069734146,0.00021209764,0.00013932884,0.00032382229,0.0012418984,0.013287379],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.97829276,0.007331589,0.0013919008,0.0019411221,0.010214406,0.00082825427],"domain_scores_gemma":[0.88715607,0.080944344,0.0087485025,0.008122296,0.012014574,0.0030143186],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008009072,0.0011127987,0.0010696031,0.0011602262,0.00092623517,0.0039091767,0.0015031062,0.00258324,0.007846885],"category_scores_gemma":[0.08378934,0.00043190483,0.00045255307,0.00061038753,0.0012844263,0.0053469725,0.004487019,0.0021296313,0.0025250532],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00241555,0.0012483936,0.050189022,0.0015516866,0.000113273454,0.007076301,0.018633137,0.013072482,0.023857927,0.017357867,0.019080522,0.8454038],"study_design_scores_gemma":[0.0006587514,0.004700827,0.050672,0.0017974233,0.0005947859,0.03650714,0.019529197,0.25248975,0.23499724,0.16826221,0.22871359,0.0010770732],"about_ca_topic_score_codex":0.00057849655,"about_ca_topic_score_gemma":0.00088110904,"teacher_disagreement_score":0.008009072,"about_ca_system_score_codex":0.000967301,"about_ca_system_score_gemma":0.00152659,"threshold_uncertainty_score":0.04235655},"labels":[],"label_agreement":null},{"id":"W1875686838","doi":"10.16995/dscn.139","title":"Linking Fancy unto Fancy: Towards a Semantic Codex","year":2009,"lang":"fr","type":"article","venue":"Digital Studies / Le champ numérique","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Meaning (existential); Computer science; Automatic summarization; Hypertext; Reading (process); Representation (politics); Interpretation (philosophy); Dimension (graph theory); Set (abstract data type); Linguistics; Artificial intelligence; Epistemology; Philosophy; World Wide Web; Mathematics","score_opus":0.03232175263403574,"score_gpt":0.30173998544897657,"score_spread":0.2694182328149408,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1875686838","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020654144,0.0004134138,0.9170339,0.005313893,0.00048091903,0.0001910774,0.0011940327,0.0056234784,0.049095195],"genre_scores_gemma":[0.38453776,0.00065898214,0.5786793,0.0013914577,0.00036178424,0.00026328533,0.003167677,0.002990337,0.027949555],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99692386,0.0010777471,0.00026910624,0.0007653395,0.000773187,0.00019076941],"domain_scores_gemma":[0.9952958,0.0015716987,0.0003654335,0.0012036401,0.0013229919,0.0002403871],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031128265,0.0008712574,0.0009000911,0.0053810775,0.0034708753,0.0074067535,0.0019692585,0.0026596184,0.010969231],"category_scores_gemma":[0.010895681,0.0006978811,0.0010369428,0.0037491198,0.005857489,0.02130225,0.0073541277,0.0030305653,0.0022444616],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000058363275,0.000041770178,0.0019375132,0.0001852974,0.000019489902,0.00018200948,0.0031343743,0.00097505987,0.0018890756,0.9030211,0.011701304,0.07685464],"study_design_scores_gemma":[0.000021027616,0.000031627787,0.0013885576,0.0003891533,0.00008065469,0.00044317907,0.003237493,0.031808525,0.0074064108,0.59237903,0.36273807,0.00007620461],"about_ca_topic_score_codex":0.008283811,"about_ca_topic_score_gemma":0.0056482386,"teacher_disagreement_score":0.010969231,"about_ca_system_score_codex":0.0021285415,"about_ca_system_score_gemma":0.0028351168,"threshold_uncertainty_score":0.03669572},"labels":[],"label_agreement":null},{"id":"W187572437","doi":"10.1007/978-94-015-9696-1_5","title":"Relationships in Multilingual Thesauri","year":2001,"lang":"en","type":"book-chapter","venue":"Information science and knowledge management","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Linguistics; Thesaurus; Relation (database); Context (archaeology); Information structure; Information retrieval; Natural language processing; World Wide Web; History; Philosophy","score_opus":0.03188928000121715,"score_gpt":0.2877159163677354,"score_spread":0.25582663636651826,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W187572437","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.056071263,0.01713416,0.6638031,0.0034767292,0.0007262425,0.00020987456,0.0010514043,0.0022770122,0.25525028],"genre_scores_gemma":[0.44851607,0.014007604,0.41131514,0.0008139472,0.0006140329,0.0003751319,0.0033216644,0.0016725785,0.119363755],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99854636,0.0006789718,0.00018751153,0.00021364146,0.00029254338,0.00008096119],"domain_scores_gemma":[0.9979188,0.0012086043,0.00014814365,0.0004150663,0.00024835646,0.00006104985],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013970208,0.00041164985,0.00049028615,0.0024025363,0.001996952,0.00564912,0.0009218467,0.0010630069,0.010492714],"category_scores_gemma":[0.0057452517,0.0009143645,0.0007522337,0.0052666124,0.002010628,0.015521395,0.0027318394,0.0016878261,0.0029645944],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000044672513,0.000037951122,0.00050820963,0.00028716432,0.000024160961,0.00035148024,0.0029764809,0.0017505473,0.0014971062,0.8517436,0.008508897,0.13226976],"study_design_scores_gemma":[0.000016403515,0.00003264541,0.000658358,0.00034641204,0.00008671354,0.000876894,0.0015003739,0.012501007,0.00344609,0.65822905,0.32226804,0.00003799682],"about_ca_topic_score_codex":0.004989381,"about_ca_topic_score_gemma":0.0053191716,"teacher_disagreement_score":0.010492714,"about_ca_system_score_codex":0.0015558974,"about_ca_system_score_gemma":0.0011244719,"threshold_uncertainty_score":0.035101593},"labels":[],"label_agreement":null},{"id":"W1878190030","doi":"","title":"Spoken Among the Trees","year":2010,"lang":"en","type":"article","venue":"Journal of Mennonite studies","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Linguistics; Philosophy","score_opus":0.015510951460425615,"score_gpt":0.30657266763349444,"score_spread":0.2910617161730688,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1878190030","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19793405,0.0017654676,0.1132765,0.012224836,0.0027151909,0.00016026005,0.0023198358,0.0026397335,0.66696405],"genre_scores_gemma":[0.86561114,0.00044086852,0.013489883,0.0011237495,0.00032067372,0.000037157246,0.0009871273,0.000789999,0.11719938],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99903905,0.00039371208,0.000042505213,0.00023590587,0.00022136212,0.00006748729],"domain_scores_gemma":[0.9970983,0.0015003824,0.000099645265,0.0005103942,0.0006255318,0.00016569032],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00081042945,0.00030679567,0.00036242988,0.0007491726,0.0020515898,0.00385802,0.000542596,0.0009102989,0.042016003],"category_scores_gemma":[0.00761528,0.00029029974,0.00020633018,0.0006182961,0.0012204167,0.004317885,0.0013748064,0.0020185825,0.006605154],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005556556,0.00020639577,0.011586811,0.00040692373,0.00006588129,0.0010842092,0.06807159,0.0011658699,0.031293478,0.45705378,0.10172132,0.32678804],"study_design_scores_gemma":[0.00008044774,0.00022532775,0.016984139,0.00029547422,0.00011718558,0.002009698,0.054505106,0.019528056,0.020002019,0.16778938,0.71828747,0.00017570984],"about_ca_topic_score_codex":0.004640526,"about_ca_topic_score_gemma":0.0067313174,"teacher_disagreement_score":0.042016003,"about_ca_system_score_codex":0.000678743,"about_ca_system_score_gemma":0.00079256814,"threshold_uncertainty_score":0.14055753},"labels":[],"label_agreement":null},{"id":"W1878813","doi":"10.1139/m91-053","title":"A Remark on Natural Language Processing from the Biolinguistic Perspective","year":2007,"lang":"en","type":"article","venue":"New Trends in Software Methodologies, Tools and Techniques","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Computer science; Perspective (graphical); Deep linguistic processing; Universal Networking Language; Language identification; Linguistics; Natural language; Natural (archaeology); Natural language processing; Language structure; Natural language programming; Object language; Language technology; Artificial intelligence; Cognitive science; Comprehension approach; Psychology; History","score_opus":0.07642038018737468,"score_gpt":0.39285066376176714,"score_spread":0.31643028357439246,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1878813","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014451824,0.029564843,0.65089405,0.17510812,0.008554795,0.00009063367,0.001052331,0.0020183416,0.118265115],"genre_scores_gemma":[0.48384023,0.01787325,0.3855605,0.04366736,0.019472659,0.0008573998,0.0011391633,0.0016796472,0.045909822],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99523515,0.0021487945,0.00038719267,0.0011067988,0.000846979,0.00027509683],"domain_scores_gemma":[0.9890027,0.008492873,0.00030336928,0.0010381563,0.0009784041,0.00018453859],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005540327,0.0007888001,0.00092342735,0.0024662053,0.0033990168,0.007057718,0.0024037918,0.0029864088,0.0078629535],"category_scores_gemma":[0.011063418,0.0004907299,0.0016918732,0.0018039535,0.017052677,0.013539356,0.0034750206,0.0070480504,0.0028900895],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009017211,0.000014812091,0.00031628873,0.00047907475,0.00003303986,0.00040453096,0.003384513,0.00042625866,0.0016688932,0.95023555,0.016917612,0.02602935],"study_design_scores_gemma":[0.00002796418,0.000049863473,0.0005421076,0.00014959321,0.000027199361,0.00073329074,0.0009175921,0.0027863237,0.0016790595,0.81254214,0.18049201,0.000052899337],"about_ca_topic_score_codex":0.0024114994,"about_ca_topic_score_gemma":0.0013668584,"teacher_disagreement_score":0.0078629535,"about_ca_system_score_codex":0.0012822315,"about_ca_system_score_gemma":0.0008560116,"threshold_uncertainty_score":0.029300392},"labels":[],"label_agreement":null},{"id":"W1880198177","doi":"10.1109/icassp.2001.940900","title":"A dynamic semantic model for re-scoring recognition hypotheses","year":2002,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"OceanWorks International (Canada)","funders":"","keywords":"Computer science; Dialog box; Artificial intelligence; Speech recognition; Natural language processing; Machine learning; World Wide Web","score_opus":0.06300285210089923,"score_gpt":0.2811839903988858,"score_spread":0.21818113829798658,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1880198177","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0032255785,0.00013402427,0.99342936,0.00033633085,0.000056150642,0.000065703374,0.00027667958,0.00069395936,0.0017822545],"genre_scores_gemma":[0.30810517,0.0006354625,0.6767933,0.000822497,0.00022122012,0.0005450634,0.0024650863,0.00040823343,0.010003986],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99798656,0.0006232337,0.0001385687,0.0005804885,0.0005320419,0.00013904407],"domain_scores_gemma":[0.99751425,0.0013553873,0.00021105453,0.00039565368,0.00044937927,0.000074263255],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025978622,0.0012470387,0.00081887253,0.0024582515,0.0009645871,0.0017303918,0.0028998165,0.0017852911,0.0051393355],"category_scores_gemma":[0.0074811797,0.0009597007,0.0016302933,0.0016664411,0.0010251796,0.003867252,0.00152271,0.0022963688,0.0023695852],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037953292,0.00030508035,0.0031145227,0.00025192005,0.00035071129,0.0004190557,0.0004756019,0.3737728,0.0079970285,0.09023152,0.012028558,0.51067364],"study_design_scores_gemma":[0.000016617805,0.000041170475,0.00044742087,0.000024916535,0.000052242867,0.00010219842,0.000034785127,0.9474662,0.0019266434,0.043929696,0.0059115035,0.000046534326],"about_ca_topic_score_codex":0.014684089,"about_ca_topic_score_gemma":0.020394808,"teacher_disagreement_score":0.014684089,"about_ca_system_score_codex":0.0018617228,"about_ca_system_score_gemma":0.0014901133,"threshold_uncertainty_score":0.029197216},"labels":[],"label_agreement":null},{"id":"W188613399","doi":"","title":"Information Processing Redux","year":2010,"lang":"en","type":"article","venue":"New Trends in Software Methodologies, Tools and Techniques","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Computer science; Information extraction; Information processing; Information retrieval; Natural language processing; Natural language; Information system; Artificial intelligence; Process (computing); Question answering; Point (geometry); Programming language; Engineering","score_opus":0.07073178938678519,"score_gpt":0.3634253553560891,"score_spread":0.2926935659693039,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W188613399","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006756345,0.0222625,0.14530337,0.06929319,0.012979811,0.00022425133,0.0018070063,0.008292549,0.7330811],"genre_scores_gemma":[0.1006329,0.027839605,0.14106524,0.021628346,0.008535849,0.00038328164,0.003061772,0.0030534198,0.6937996],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9953317,0.0011647558,0.000257985,0.0008205278,0.002156229,0.00026881078],"domain_scores_gemma":[0.9889346,0.0032408335,0.00035193897,0.0045484323,0.0024624155,0.00046188856],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0044678645,0.0010535523,0.00072376936,0.00356511,0.0025206448,0.010812608,0.0020644472,0.0020450826,0.08032217],"category_scores_gemma":[0.012209092,0.00056253,0.0011098773,0.0034679605,0.0028816056,0.01244136,0.00551361,0.004435681,0.03831361],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014976024,0.00005662934,0.00034752706,0.00032073987,0.000027637363,0.00020511441,0.0004646283,0.0006547162,0.0016454873,0.41598985,0.18272077,0.39741716],"study_design_scores_gemma":[0.00001612386,0.000023466144,0.00025991342,0.00014822617,0.000013185489,0.00025150768,0.000120076256,0.0018381306,0.0014280454,0.08020034,0.9156832,0.00001772274],"about_ca_topic_score_codex":0.0031321994,"about_ca_topic_score_gemma":0.0026517438,"teacher_disagreement_score":0.08032217,"about_ca_system_score_codex":0.003165284,"about_ca_system_score_gemma":0.0034539846,"threshold_uncertainty_score":0.26870447},"labels":[],"label_agreement":null},{"id":"W1887914154","doi":"10.1558/cj.v20i3.561-578","title":"A New Template-Template-enhanced ICALL System for a Second Language Composition Course","year":2003,"lang":"en","type":"article","venue":"CALICO Journal","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Northern British Columbia","funders":"","keywords":"Computer science; Bottleneck; Automaton; Artificial intelligence; Simple (philosophy); Finite-state machine; Scheme (mathematics); Natural language processing; Composition (language); Natural language; Theoretical computer science; Programming language","score_opus":0.010975231054222455,"score_gpt":0.2811540294831161,"score_spread":0.27017879842889364,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1887914154","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012778667,0.00006549956,0.85531324,0.0001860949,0.00016339394,0.00031194306,0.000581683,0.12295269,0.0076467507],"genre_scores_gemma":[0.12182709,0.000054354277,0.84864706,0.00030302018,0.00011195091,0.0004620925,0.0023004103,0.0031023112,0.023191702],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9992042,0.0001114156,0.00006133187,0.00033396334,0.00022855829,0.000060527214],"domain_scores_gemma":[0.99871266,0.0003452276,0.00007159209,0.00036271557,0.00032621343,0.00018169606],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010360043,0.00077191123,0.00077891315,0.00072875153,0.00076808105,0.0016325953,0.0024510347,0.0012278581,0.018238394],"category_scores_gemma":[0.0026199243,0.0004936946,0.00064487226,0.0005099572,0.0005836981,0.0018861676,0.0011797644,0.001225484,0.009267238],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013620008,0.00068950374,0.002441465,0.00048207844,0.00008812376,0.0013980313,0.0010946046,0.01902384,0.16791774,0.021101475,0.05874077,0.7256604],"study_design_scores_gemma":[0.0003528666,0.00042253794,0.0012241113,0.000044899833,0.00013900813,0.001178462,0.00014214958,0.61349744,0.19548427,0.009838633,0.17749621,0.00017947315],"about_ca_topic_score_codex":0.0030821108,"about_ca_topic_score_gemma":0.0033828525,"teacher_disagreement_score":0.018238394,"about_ca_system_score_codex":0.0011738669,"about_ca_system_score_gemma":0.0019106325,"threshold_uncertainty_score":0.06101352},"labels":[],"label_agreement":null},{"id":"W1891960889","doi":"10.1109/icsmc.2001.971944","title":"Modular HPSG [Head-driven Phrase Structure Grammar]","year":2002,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Head-driven phrase structure grammar; Computer science; Phrase structure grammar; Phrase structure rules; Grammar; Programming language; Natural language processing; Generalized phrase structure grammar; Modular design; Modularity (biology); Artificial intelligence; Rule-based machine translation; Context-free grammar; Generative grammar; Linguistics; Emergent grammar; Relational grammar","score_opus":0.013782773198430657,"score_gpt":0.24620532936578918,"score_spread":0.23242255616735852,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1891960889","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.001589231,0.00020984461,0.97990453,0.00028836002,0.00010073929,0.00020944336,0.0008199499,0.0059815296,0.010896313],"genre_scores_gemma":[0.08208996,0.00059727917,0.90072984,0.0006213268,0.0001529222,0.0004191837,0.0024327391,0.001876121,0.011080654],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.999203,0.00020926191,0.00006835671,0.00018813385,0.00025505098,0.00007625786],"domain_scores_gemma":[0.99942183,0.00014439375,0.00006563887,0.00017863252,0.00015141617,0.0000381465],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010418235,0.0008088975,0.00055997504,0.0008400172,0.0005416163,0.0012283952,0.002146268,0.0010099121,0.011732182],"category_scores_gemma":[0.002055437,0.0007055041,0.0010363879,0.0014476929,0.0019517611,0.0026061446,0.0019868824,0.0015048204,0.0058270292],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000050569484,0.00003844809,0.00034244792,0.00037338005,0.00005028004,0.00058286224,0.0005888424,0.0101125045,0.008987383,0.8561918,0.02522107,0.0974603],"study_design_scores_gemma":[0.00005292156,0.0000676446,0.00038520724,0.00011145863,0.00006707116,0.0014580673,0.00005533896,0.06293962,0.009217902,0.6187407,0.3068275,0.0000766114],"about_ca_topic_score_codex":0.0016972318,"about_ca_topic_score_gemma":0.002074388,"teacher_disagreement_score":0.011732182,"about_ca_system_score_codex":0.0005725171,"about_ca_system_score_gemma":0.0013606015,"threshold_uncertainty_score":0.03924811},"labels":[],"label_agreement":null},{"id":"W1894075015","doi":"10.1162/coli_a_00226","title":"CODRA: A Novel Discriminative Framework for Rhetorical Analysis","year":2015,"lang":"en","type":"article","venue":"Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":186,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Parsing; Computer science; Natural language processing; Rhetorical question; Discriminative model; Artificial intelligence; Coherence (philosophical gambling strategy); Probabilistic logic; Parse tree; Classifier (UML); Linguistics; Margin (machine learning); Machine learning; Mathematics","score_opus":0.07786810095225595,"score_gpt":0.37519876686569953,"score_spread":0.2973306659134436,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1894075015","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.003956604,0.0005497567,0.9885156,0.00028875857,0.000048191923,0.00011115683,0.00095971377,0.0038667172,0.00170348],"genre_scores_gemma":[0.17848986,0.00051393866,0.80902547,0.00040364754,0.00029711658,0.0005195447,0.0054091667,0.0012945102,0.004046726],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99633443,0.0014186562,0.00019468025,0.0011136064,0.00071882,0.0002197965],"domain_scores_gemma":[0.99228066,0.0043906807,0.00075755664,0.0014392374,0.00086086476,0.0002709751],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0043811863,0.0017611989,0.0015589945,0.0069141574,0.0017376158,0.0028412247,0.0038734989,0.0021213496,0.005917545],"category_scores_gemma":[0.012827323,0.001136831,0.0017011046,0.004333582,0.0022253226,0.004854862,0.0033675919,0.003643182,0.0033482776],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002632594,0.0002956001,0.0052366294,0.00076266774,0.00020144435,0.00035136068,0.0009168882,0.046637055,0.015375774,0.17852832,0.026256088,0.72517484],"study_design_scores_gemma":[0.00004108062,0.00006886266,0.0020230599,0.00007312797,0.000059912683,0.0004856064,0.00013740738,0.77746105,0.005572089,0.18205494,0.03194838,0.00007450148],"about_ca_topic_score_codex":0.0066156816,"about_ca_topic_score_gemma":0.013948265,"teacher_disagreement_score":0.0069141574,"about_ca_system_score_codex":0.0016078675,"about_ca_system_score_gemma":0.0028933615,"threshold_uncertainty_score":0.023170173},"labels":[],"label_agreement":null},{"id":"W1896493377","doi":"10.21248/hpsg.2008.11","title":"Memory management for unification-based processing of typed feature structures","year":2008,"lang":"en","type":"article","venue":"Proceedings of the International Conference on Head-Driven Phrase Structure Grammar","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Unification; Parsing; Feature (linguistics); Computer science; Rank (graph theory); Grammar; Value (mathematics); Signature (topology); Artificial intelligence; Natural language processing; Theoretical computer science; Programming language; Machine learning; Linguistics; Mathematics","score_opus":0.030763836882820715,"score_gpt":0.2916918817946063,"score_spread":0.2609280449117856,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1896493377","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.69586325,0.000856843,0.2901101,0.0016158508,0.00014053001,0.00026303733,0.00019945114,0.007457424,0.003493518],"genre_scores_gemma":[0.92249787,0.00021618711,0.07501826,0.00026127775,0.000050776172,0.00015133918,0.00016734091,0.00036858863,0.0012682537],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9972128,0.0009319609,0.00031333728,0.00062042795,0.00047534987,0.00044602895],"domain_scores_gemma":[0.96918744,0.017284015,0.0027165397,0.008539115,0.0016196097,0.0006533563],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005796794,0.00077522924,0.0010553825,0.0013720472,0.0019064022,0.0045224,0.0034602683,0.0018133741,0.0032223817],"category_scores_gemma":[0.031582024,0.00067635864,0.0006905049,0.0019715775,0.002109999,0.012517518,0.002250227,0.002458683,0.00080848875],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004703521,0.0011855707,0.02597583,0.0006928146,0.00024524104,0.0009626101,0.00368518,0.069485195,0.12553525,0.094072334,0.0050455444,0.66841084],"study_design_scores_gemma":[0.00035325545,0.0016513172,0.0058349865,0.00011570131,0.00048052636,0.0007658721,0.0014034208,0.55357724,0.22206716,0.20886436,0.0045825285,0.00030367926],"about_ca_topic_score_codex":0.002514448,"about_ca_topic_score_gemma":0.0030139422,"teacher_disagreement_score":0.005796794,"about_ca_system_score_codex":0.002012332,"about_ca_system_score_gemma":0.002793323,"threshold_uncertainty_score":0.030656695},"labels":[],"label_agreement":null},{"id":"W1902044186","doi":"10.3917/i2d.152.0070","title":"L’extraction d’entités nommées : une opportunité pour le secteur culturel ?","year":2015,"lang":"fr","type":"article","venue":"I2D - Information données & documents","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Humanities; Political science; Philosophy","score_opus":0.04627228685610568,"score_gpt":0.30643242697891415,"score_spread":0.26016014012280847,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1902044186","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19136907,0.0055272644,0.662344,0.01886772,0.00054902065,0.0007103907,0.0029540136,0.0039505237,0.11372792],"genre_scores_gemma":[0.45316428,0.0040249852,0.47122866,0.0020177786,0.00012425856,0.00041509108,0.0047446745,0.0017260833,0.06255415],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9950777,0.0019648413,0.0003461459,0.0006402617,0.0017302975,0.00024061193],"domain_scores_gemma":[0.9858548,0.005506787,0.0006660811,0.0026953926,0.00505143,0.00022554495],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008309878,0.0008598091,0.00059420266,0.0032827582,0.0033608498,0.008255018,0.0017257516,0.0015485436,0.008575985],"category_scores_gemma":[0.017488783,0.0006389776,0.0009202268,0.0044387733,0.003418018,0.011780696,0.0029750396,0.0017719453,0.0029629527],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025355857,0.00009268858,0.023956759,0.0022316868,0.00018930898,0.001774044,0.080509655,0.0029912665,0.042060338,0.14638326,0.027439578,0.6721179],"study_design_scores_gemma":[0.000027418542,0.00010476177,0.025775468,0.001289293,0.0001892715,0.0021816297,0.041083414,0.012660047,0.043527383,0.051942583,0.8209918,0.00022684937],"about_ca_topic_score_codex":0.093388185,"about_ca_topic_score_gemma":0.12527688,"teacher_disagreement_score":0.093388185,"about_ca_system_score_codex":0.005622158,"about_ca_system_score_gemma":0.0075414935,"threshold_uncertainty_score":0.18568921},"labels":[],"label_agreement":null},{"id":"W1904294760","doi":"","title":"Toward plWordNet 2.0","year":2010,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"WordNet; Computer science; Variety (cybernetics); Artificial intelligence; Natural language processing; Plan (archaeology); Information retrieval; History","score_opus":0.013237800301918741,"score_gpt":0.2664916909758271,"score_spread":0.25325389067390836,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1904294760","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0062672105,0.0016890939,0.86968225,0.0046203188,0.0013162809,0.00051207247,0.0057738894,0.07833065,0.031808145],"genre_scores_gemma":[0.028219676,0.0022662594,0.87324417,0.002206427,0.0004835879,0.0011338665,0.05179833,0.014284638,0.026363043],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99751306,0.00072634337,0.00021866157,0.0006511912,0.0006965938,0.00019419326],"domain_scores_gemma":[0.9964449,0.0008229455,0.00020538272,0.000745908,0.0013425781,0.0004382375],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010036447,0.0022606407,0.0010979186,0.0047808774,0.0011326586,0.0058769523,0.0037018743,0.0018580386,0.015036392],"category_scores_gemma":[0.011700028,0.0019062164,0.0010951948,0.0031616646,0.001044002,0.014148855,0.0051714615,0.0046356767,0.023991136],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00055655686,0.00035368436,0.002090035,0.0012575655,0.00014647588,0.00034376516,0.0009472038,0.008720628,0.014221382,0.35111833,0.23307973,0.38716468],"study_design_scores_gemma":[0.00009219172,0.00015613339,0.00068651605,0.00047853697,0.0000640629,0.0003003116,0.00031733146,0.06271812,0.010637689,0.17602244,0.74843866,0.00008800824],"about_ca_topic_score_codex":0.002552886,"about_ca_topic_score_gemma":0.0033350568,"teacher_disagreement_score":0.015036392,"about_ca_system_score_codex":0.0013611601,"about_ca_system_score_gemma":0.0035664942,"threshold_uncertainty_score":0.053078473},"labels":[],"label_agreement":null},{"id":"W190911302","doi":"10.1007/978-3-642-54903-8_46","title":"Evaluation of Sentence Compression Techniques against Human Performance","year":2014,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Automatic summarization; Computer science; Sentence; Natural language processing; Artificial intelligence; Syntax; Compression (physics); Word (group theory); Context (archaeology); Feature (linguistics); Speech recognition; Linguistics","score_opus":0.02513418462569633,"score_gpt":0.297930327783601,"score_spread":0.2727961431579047,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W190911302","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.87424874,0.011911378,0.08117521,0.00042549628,0.00095351425,0.0006244792,0.0069766087,0.01448308,0.009201586],"genre_scores_gemma":[0.86486065,0.0029412506,0.099052265,0.00017739202,0.00036898864,0.00029015163,0.020961482,0.0007477113,0.010600066],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9972939,0.00094595144,0.0003191148,0.0004848239,0.0008200781,0.0001361846],"domain_scores_gemma":[0.98738074,0.00900427,0.00048659707,0.0008512739,0.0020427466,0.00023431117],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002514527,0.0014830148,0.00077847426,0.0021438072,0.00041426616,0.0008741049,0.0011160142,0.0010804016,0.0050226036],"category_scores_gemma":[0.010163305,0.00018801563,0.00047479587,0.0016341801,0.0003383584,0.0010551106,0.00079987774,0.0005438518,0.0026493866],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0063790902,0.0013541064,0.0064072404,0.002114805,0.00067754433,0.00067300873,0.0004182103,0.026927076,0.084769696,0.00057125406,0.015877089,0.8538309],"study_design_scores_gemma":[0.00090754044,0.015889458,0.059650246,0.00028290876,0.001692889,0.0034707482,0.0009235238,0.53841996,0.36088255,0.0018816622,0.015762856,0.00023558471],"about_ca_topic_score_codex":0.0024748186,"about_ca_topic_score_gemma":0.002339765,"teacher_disagreement_score":0.0050226036,"about_ca_system_score_codex":0.00038118925,"about_ca_system_score_gemma":0.00038458346,"threshold_uncertainty_score":0.016802311},"labels":[],"label_agreement":null},{"id":"W1911155832","doi":"","title":"Expression of uncertainty in linguistic data","year":2008,"lang":"en","type":"article","venue":"International Conference on Information Fusion","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":30,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Defence Research and Development Canada","funders":"","keywords":"Ambiguity; Computer science; Certainty; Expression (computer science); Utterance; Natural language; Deep linguistic processing; Natural language processing; Interpretation (philosophy); Linguistics; Artificial intelligence; Sine qua non; Point (geometry); Natural (archaeology); Mathematics","score_opus":0.056087336018732926,"score_gpt":0.3233109329352256,"score_spread":0.26722359691649267,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1911155832","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025858615,0.0033094762,0.95501137,0.0068101576,0.00029462122,0.00015062884,0.0007159272,0.00020068404,0.007648517],"genre_scores_gemma":[0.5736148,0.0034559937,0.41710716,0.0012072448,0.00092959445,0.00061966217,0.0007981947,0.00014183269,0.002125504],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.97014093,0.011474096,0.0039800704,0.002801317,0.010632191,0.00097142626],"domain_scores_gemma":[0.9123062,0.06587419,0.008983385,0.006453807,0.005757179,0.0006252715],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.021590408,0.00097173854,0.0014682813,0.0069729555,0.0027407787,0.011256266,0.0024850836,0.0027519593,0.0014637379],"category_scores_gemma":[0.09165405,0.0009825175,0.0015619335,0.0063616717,0.009542524,0.017478315,0.006769065,0.0036482674,0.000317212],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000095688854,0.000026261812,0.002727542,0.00063775084,0.00014829713,0.0015807186,0.0060852696,0.021558143,0.0027652269,0.9145459,0.0012072021,0.048622027],"study_design_scores_gemma":[0.000008783767,0.000026841773,0.00064940995,0.00027319946,0.000053118234,0.00055408885,0.001311512,0.032574616,0.0016775221,0.9540008,0.008780623,0.00008957076],"about_ca_topic_score_codex":0.0025336407,"about_ca_topic_score_gemma":0.0012757235,"teacher_disagreement_score":0.021590408,"about_ca_system_score_codex":0.0037203794,"about_ca_system_score_gemma":0.0020160482,"threshold_uncertainty_score":0.11418235},"labels":[],"label_agreement":null},{"id":"W19132006","doi":"10.1038/embor.2008.243","title":"Harnessing Unlabeled Examples through Iterative Application of Dynamic Markov Modeling","year":2006,"lang":"en","type":"article","venue":"EMBO Reports","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Markov chain; Classifier (UML); Pattern recognition (psychology); Hidden Markov model; Test set; Artificial intelligence; Markov process; Markov model; Mathematics; Machine learning; Statistics","score_opus":0.010920251025065006,"score_gpt":0.2747627759233041,"score_spread":0.2638425248982391,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W19132006","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01496995,0.00015403678,0.9812152,0.0002560093,0.000020104128,0.000081297956,0.00009714404,0.0008268196,0.0023794312],"genre_scores_gemma":[0.37643638,0.00028270172,0.6173841,0.00018761221,0.00007304091,0.0006172621,0.00072179805,0.00033614298,0.0039609247],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9984908,0.0007974996,0.00007152324,0.00028132167,0.00022374738,0.00013506562],"domain_scores_gemma":[0.9901044,0.008258237,0.0003313233,0.0006546477,0.00047002098,0.00018135879],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029899818,0.0012768996,0.0014364772,0.00207417,0.0012400752,0.0014890636,0.003084425,0.0019667484,0.005809887],"category_scores_gemma":[0.012490664,0.0011309976,0.0016964993,0.0014544554,0.0014524485,0.0030323148,0.0033732525,0.0018111141,0.0011237855],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013297563,0.000073459945,0.001089272,0.00013343475,0.00006762163,0.00013186484,0.0001725763,0.8775855,0.00094602077,0.045639873,0.0011782434,0.07284922],"study_design_scores_gemma":[0.0000053365943,0.000007209416,0.00001888297,0.000004764097,0.0000046524456,0.000007950849,0.000007144579,0.98314697,0.00021436761,0.01627228,0.00030753927,0.0000030050996],"about_ca_topic_score_codex":0.006354742,"about_ca_topic_score_gemma":0.013710026,"teacher_disagreement_score":0.006354742,"about_ca_system_score_codex":0.0015017678,"about_ca_system_score_gemma":0.0016125775,"threshold_uncertainty_score":0.019435942},"labels":[],"label_agreement":null},{"id":"W1921471220","doi":"","title":"Analysis on the Semantics of Word Trip","year":2011,"lang":"en","type":"article","venue":"Studies in literature and language","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Semantics (computer science); Word (group theory); Perspective (graphical); Computer science; Linguistics; Generalization; Natural language processing; Key (lock); Artificial intelligence; Computational semantics; Operational semantics; Mathematics; Programming language; Philosophy","score_opus":0.026379899440651833,"score_gpt":0.30421212279812415,"score_spread":0.2778322233574723,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1921471220","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.21323128,0.002024618,0.696706,0.0027788181,0.00024876322,0.0001595874,0.0024228648,0.0009641311,0.08146398],"genre_scores_gemma":[0.88700473,0.0010103065,0.10298432,0.00016706735,0.00015462199,0.0001488157,0.0017608937,0.00019911179,0.006570119],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"qualitative","domain_scores_codex":[0.99933064,0.00019475391,0.00007643478,0.00017227045,0.0001659134,0.000059965285],"domain_scores_gemma":[0.9987024,0.0005164948,0.00017061431,0.00016221593,0.00039166232,0.000056517638],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00056883285,0.00042961125,0.00033506795,0.005224976,0.0014848892,0.002442736,0.0005598134,0.00041562275,0.006145462],"category_scores_gemma":[0.0030750453,0.000186216,0.0007884815,0.00451792,0.0027333093,0.0075602676,0.0015334929,0.00082075014,0.0007260044],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000050662573,0.000011134865,0.0013117244,0.00020485594,0.000022732982,0.00023960228,0.0023420362,0.0019165514,0.0025179093,0.9574304,0.002369163,0.031583257],"study_design_scores_gemma":[0.000008209965,0.000033315268,0.0020786005,0.00010407235,0.000046239114,0.00054425845,0.0031574846,0.017556397,0.0029597974,0.92626053,0.047221117,0.000029948606],"about_ca_topic_score_codex":0.0021390782,"about_ca_topic_score_gemma":0.0014192767,"teacher_disagreement_score":0.006145462,"about_ca_system_score_codex":0.0011561309,"about_ca_system_score_gemma":0.0008175919,"threshold_uncertainty_score":0.020558596},"labels":[],"label_agreement":null},{"id":"W1923455183","doi":"","title":"Generating and validating abstracts of meeting conversations: a user study","year":2010,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":66,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Usability; Formative assessment; Information retrieval; Ontology; World Wide Web; Coherence (philosophical gambling strategy); Representation (politics); Preference; Selection (genetic algorithm); Natural language processing; Human–computer interaction; Artificial intelligence","score_opus":0.013466438542500294,"score_gpt":0.28081686011364626,"score_spread":0.26735042157114597,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1923455183","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9280803,0.00071969396,0.06275027,0.0004103877,0.00007670076,0.0010321786,0.0011835679,0.004015685,0.0017312474],"genre_scores_gemma":[0.8553051,0.0004664832,0.13493872,0.00038207616,0.000069374306,0.001126725,0.00442219,0.00077945646,0.0025099558],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9701082,0.022138732,0.0025303909,0.0019192968,0.002954667,0.00034869308],"domain_scores_gemma":[0.743602,0.22100732,0.0042282813,0.015523361,0.014164611,0.0014744316],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.029756851,0.0017640583,0.0014627826,0.001800566,0.0013857228,0.0023603074,0.0019026363,0.0023086425,0.0031433066],"category_scores_gemma":[0.12475255,0.000667932,0.0010195281,0.0011194678,0.0008177606,0.0032240376,0.002141607,0.0013228738,0.0016359268],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.009311658,0.008621059,0.073468655,0.0074932566,0.0012189777,0.0070081595,0.104680024,0.017674511,0.14417405,0.003135548,0.020346275,0.60286784],"study_design_scores_gemma":[0.004019314,0.04363648,0.14368734,0.0014041881,0.0024176515,0.028501559,0.053604625,0.24738014,0.31727076,0.005945503,0.1505296,0.0016029795],"about_ca_topic_score_codex":0.0008940477,"about_ca_topic_score_gemma":0.0011721262,"teacher_disagreement_score":0.029756851,"about_ca_system_score_codex":0.000679374,"about_ca_system_score_gemma":0.0006333146,"threshold_uncertainty_score":0.15737116},"labels":[],"label_agreement":null},{"id":"W192459572","doi":"10.63317/32wvbo2cxiv3","title":"Can Syntactic and Logical Graphs help Word Sense Disambiguation?","year":2010,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; Simon Fraser University","funders":"","keywords":"Computer science; Natural language processing; Parsing; Sentence; Predicative expression; Artificial intelligence; Logical form; Context (archaeology); Dependency (UML); Word (group theory); Dependency grammar; Syntax; Grammar; Linguistics","score_opus":0.011364359614625997,"score_gpt":0.2664729018180604,"score_spread":0.2551085422034344,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W192459572","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13328865,0.0051508057,0.8324953,0.005581044,0.00066749705,0.00032496263,0.0025134082,0.007962761,0.0120156],"genre_scores_gemma":[0.4591476,0.002104285,0.53245044,0.0006718362,0.00016845013,0.00020657698,0.0031131422,0.0006800198,0.0014576332],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9969625,0.0015714385,0.00023582135,0.0006813147,0.00042054497,0.00012834107],"domain_scores_gemma":[0.9837427,0.012497709,0.0010501794,0.0013939774,0.0010613846,0.00025402804],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004976413,0.0012221338,0.0011519837,0.005844899,0.0014152351,0.0043408424,0.0014393602,0.0015497517,0.0068841064],"category_scores_gemma":[0.02549379,0.00091894413,0.0012531417,0.0046677804,0.0019189845,0.017097132,0.0023740286,0.0016318285,0.0028639254],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009313206,0.00029282726,0.011920477,0.0016289677,0.00034446872,0.00023641122,0.001482393,0.014047918,0.013940676,0.057570755,0.010852703,0.8867511],"study_design_scores_gemma":[0.00036762617,0.0004747918,0.024673866,0.0006659178,0.00086942455,0.0013456437,0.004885705,0.2333273,0.028110772,0.63213885,0.07261797,0.0005221286],"about_ca_topic_score_codex":0.0027356253,"about_ca_topic_score_gemma":0.0046407534,"teacher_disagreement_score":0.0068841064,"about_ca_system_score_codex":0.0010728488,"about_ca_system_score_gemma":0.0013723773,"threshold_uncertainty_score":0.026318073},"labels":[],"label_agreement":null},{"id":"W1926678298","doi":"","title":"A web application for filtering and annotating web speech data","year":2013,"lang":"en","type":"article","venue":"eCommons (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Social Sciences and Humanities Research Council of Canada; McGill University; National Science Foundation","keywords":"Computer science; World Wide Web; Web application; Information retrieval","score_opus":0.03677190477040208,"score_gpt":0.23257768979641374,"score_spread":0.19580578502601168,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1926678298","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0072983108,0.0003354589,0.55180883,0.0004273083,0.00028444923,0.0011017551,0.022112211,0.40644822,0.010183478],"genre_scores_gemma":[0.03855632,0.0003278039,0.8304092,0.00086818426,0.0002702666,0.0032487437,0.06194469,0.030981548,0.03339328],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9975266,0.00068846357,0.0002929333,0.0006644227,0.00069182186,0.00013585143],"domain_scores_gemma":[0.9929987,0.003823394,0.0003443477,0.0011925617,0.0013600675,0.00028082912],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035897763,0.002121083,0.0011113394,0.0041106497,0.0013973875,0.0015903921,0.0015002806,0.001621732,0.042119402],"category_scores_gemma":[0.008446042,0.0010417928,0.001192426,0.0022244216,0.00063664466,0.002713852,0.0030389025,0.0015292532,0.04162784],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012538931,0.00035405357,0.0033095677,0.0016309731,0.0002749898,0.0014521859,0.0020391112,0.0018207441,0.10533343,0.0038961722,0.36423576,0.51439905],"study_design_scores_gemma":[0.00037244594,0.00037883353,0.021347072,0.0004342209,0.00018535645,0.004051754,0.0011903809,0.105304904,0.1266639,0.018639902,0.72094065,0.00049060665],"about_ca_topic_score_codex":0.003031822,"about_ca_topic_score_gemma":0.005660067,"teacher_disagreement_score":0.042119402,"about_ca_system_score_codex":0.0005407386,"about_ca_system_score_gemma":0.0012152501,"threshold_uncertainty_score":0.14090347},"labels":[],"label_agreement":null},{"id":"W1926880687","doi":"","title":"Research into the Mental Lexicon Representation of Chinese English Learners Based on Spreading Activation Model","year":2011,"lang":"en","type":"article","venue":"Studies in literature and language","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Mental lexicon; Lexicon; Word Association; Association (psychology); Representation (politics); Relation (database); Computer science; Word (group theory); Natural language processing; Mental representation; Linguistics; Process (computing); Artificial intelligence; Psychology; Cognition","score_opus":0.04571520611608981,"score_gpt":0.38627886404977696,"score_spread":0.34056365793368715,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1926880687","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9646395,0.00019049579,0.020977013,0.0003093736,0.000013747361,0.00008805954,0.000045879504,0.000044104596,0.013691853],"genre_scores_gemma":[0.99459684,0.00014045443,0.003726243,0.000019233532,0.0000038450485,0.0000309805,0.000036924594,0.0000072818407,0.0014381619],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99956125,0.00014629411,0.000026059397,0.000085605825,0.00012441055,0.000056296565],"domain_scores_gemma":[0.998609,0.00082981196,0.00013370192,0.000095923046,0.0002533952,0.00007819147],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00082139566,0.00028531573,0.0002463317,0.00085952523,0.00031881037,0.0013343659,0.0005418262,0.00029581494,0.0035083513],"category_scores_gemma":[0.0051422985,0.00019597755,0.0002940048,0.0006014443,0.00073262834,0.0021063776,0.00039063592,0.0005405742,0.00020465144],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00042611387,0.00061090407,0.33102295,0.0007236844,0.00022761806,0.0018722325,0.08933969,0.012987439,0.1152192,0.10922726,0.002282514,0.33606043],"study_design_scores_gemma":[0.00012348068,0.0006894983,0.50611466,0.00016839092,0.00034309103,0.0021051618,0.032935303,0.3133357,0.02793321,0.105897784,0.010158641,0.0001950611],"about_ca_topic_score_codex":0.005490155,"about_ca_topic_score_gemma":0.0029502723,"teacher_disagreement_score":0.005490155,"about_ca_system_score_codex":0.00059438165,"about_ca_system_score_gemma":0.00068166043,"threshold_uncertainty_score":0.011736631},"labels":[],"label_agreement":null},{"id":"W193080678","doi":"","title":"Lessons from NRC's Portage System at WMT 2010","year":2010,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Computer science; Machine translation; Natural language processing; Translation (biology); Artificial intelligence","score_opus":0.011914044771457601,"score_gpt":0.2616577710372566,"score_spread":0.24974372626579902,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W193080678","genre_codex":"commentary","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10316295,0.00850462,0.16745573,0.571747,0.0039234217,0.00042529198,0.0039844317,0.019877013,0.120919526],"genre_scores_gemma":[0.50326794,0.0076159863,0.34424546,0.035163727,0.002591084,0.00035568248,0.0069038332,0.008595681,0.09126054],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99271166,0.0029571322,0.00045507893,0.0010758443,0.0024049592,0.0003953061],"domain_scores_gemma":[0.97998613,0.008357655,0.0002822988,0.0029033576,0.0072367936,0.0012337926],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012880755,0.0006693179,0.0007189929,0.0010554509,0.002079664,0.004321726,0.004200311,0.0036293336,0.0076676668],"category_scores_gemma":[0.04554673,0.00071311544,0.00032739853,0.0016177701,0.0024014567,0.017169287,0.002366465,0.0062651057,0.0078391265],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037014997,0.00042534218,0.0056395587,0.00064765144,0.000033026303,0.0014740895,0.006471283,0.0053247567,0.006497103,0.02999226,0.5123319,0.43079293],"study_design_scores_gemma":[0.000288744,0.00049278897,0.0069598285,0.00057211233,0.00005597925,0.0040245974,0.008183162,0.037926678,0.027397789,0.06958105,0.8441782,0.00033908497],"about_ca_topic_score_codex":0.03753202,"about_ca_topic_score_gemma":0.03968353,"teacher_disagreement_score":0.03753202,"about_ca_system_score_codex":0.0027366062,"about_ca_system_score_gemma":0.0026000733,"threshold_uncertainty_score":0.07462716},"labels":[],"label_agreement":null},{"id":"W1932710031","doi":"10.3968/j.ccc.1923670020060202.010","title":"On Semantic and Syntactical Selective Constraints among Multiple Words in Context","year":2010,"lang":"en","type":"article","venue":"Cross-cultural communication","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Linguistics; Context (archaeology); Sentence; Phrase; Humanities; Natural language processing; Artificial intelligence; Computer science; Philosophy; History","score_opus":0.013002766836556287,"score_gpt":0.321625399280752,"score_spread":0.30862263244419574,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1932710031","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13918641,0.021796694,0.75355065,0.005165672,0.00040618295,0.00045357147,0.0009877991,0.0005555212,0.077897504],"genre_scores_gemma":[0.7586514,0.012960211,0.21639606,0.0013414457,0.00059431535,0.0005775887,0.0015247897,0.0004687281,0.00748536],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9934302,0.002590337,0.0007046567,0.0014844446,0.001396654,0.00039379916],"domain_scores_gemma":[0.9893344,0.0074783633,0.00094272883,0.0008847127,0.001142986,0.00021684932],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0043103574,0.0013004069,0.0011832736,0.0046188184,0.0040389597,0.005555454,0.0016056601,0.0013869407,0.0063234787],"category_scores_gemma":[0.010941657,0.001615023,0.001763631,0.0053644665,0.009666139,0.019883981,0.0036582125,0.002680843,0.0010562923],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025511754,0.0000553337,0.0070202574,0.0015707565,0.0001813836,0.0021343646,0.009901882,0.0048707332,0.0127001805,0.8395004,0.002914349,0.11889526],"study_design_scores_gemma":[0.000043682863,0.000120468445,0.012004194,0.0010220602,0.00028348598,0.0023438348,0.0067688227,0.015463055,0.013536715,0.86769104,0.08046641,0.00025609115],"about_ca_topic_score_codex":0.009786994,"about_ca_topic_score_gemma":0.009848745,"teacher_disagreement_score":0.009786994,"about_ca_system_score_codex":0.0026946254,"about_ca_system_score_gemma":0.0031033885,"threshold_uncertainty_score":0.022795618},"labels":[],"label_agreement":null},{"id":"W1933502375","doi":"","title":"MSR SPLAT, a language analysis toolkit","year":2012,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":32,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Computer science; Parsing; Set (abstract data type); Dependency (UML); Programming language; Natural language processing; Artificial intelligence","score_opus":0.009082678841866323,"score_gpt":0.2786049867048803,"score_spread":0.269522307863014,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1933502375","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0020515237,0.00042417416,0.47740135,0.00037272653,0.0002045794,0.0004714467,0.041595563,0.4632616,0.01421711],"genre_scores_gemma":[0.031237036,0.00077172805,0.6726862,0.00088673766,0.00019123955,0.0011665899,0.14949763,0.113965265,0.029597534],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99753296,0.0004194098,0.00038469667,0.0005266089,0.0009626579,0.00017373024],"domain_scores_gemma":[0.9969121,0.0009978038,0.00031824363,0.0005904005,0.0010448707,0.00013655596],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017274085,0.0023053237,0.0013394456,0.0034796179,0.0011642283,0.004020654,0.0029045178,0.0009634514,0.059339583],"category_scores_gemma":[0.005930535,0.0022309835,0.0026411996,0.0019981766,0.00066909246,0.0061948798,0.0038227672,0.0034769138,0.06551396],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00064506265,0.00014128842,0.0018120444,0.0022849466,0.00032605216,0.0007575534,0.0013921892,0.0048247008,0.01677545,0.031356897,0.6531102,0.28657365],"study_design_scores_gemma":[0.00016825675,0.000092028946,0.0015140609,0.0003109759,0.00015440873,0.0015169124,0.0004205338,0.05728152,0.02222572,0.0383189,0.877748,0.0002486626],"about_ca_topic_score_codex":0.0068713203,"about_ca_topic_score_gemma":0.008225118,"teacher_disagreement_score":0.059339583,"about_ca_system_score_codex":0.0009960192,"about_ca_system_score_gemma":0.002785569,"threshold_uncertainty_score":0.1985107},"labels":[],"label_agreement":null},{"id":"W193484706","doi":"","title":"Use translation programs after entering handwriting into computers with a digital pen","year":2007,"lang":"en","type":"article","venue":"Annual Conference on Computers","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Handwriting; Computer science; Software; Machine translation; Translation (biology); Handwriting recognition; Multimedia; Natural language processing; Speech recognition; Artificial intelligence; Programming language; Feature extraction","score_opus":0.02716227720622748,"score_gpt":0.27091679605995356,"score_spread":0.24375451885372607,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W193484706","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.44544244,0.0012107475,0.46863616,0.0008384908,0.0004224266,0.0015629136,0.0009791009,0.018035883,0.06287187],"genre_scores_gemma":[0.5642675,0.00097553077,0.37372562,0.0005986945,0.00011585537,0.00067503896,0.0017270312,0.0036255284,0.05428915],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9988663,0.00032467872,0.0001023092,0.0002952582,0.00031741933,0.00009399295],"domain_scores_gemma":[0.9940942,0.0032534979,0.00040642434,0.0014547725,0.0006948905,0.00009616582],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012696492,0.0008781051,0.0003963575,0.0008034873,0.0005837117,0.0014848263,0.0007481239,0.00058330654,0.014533394],"category_scores_gemma":[0.009331768,0.00023746134,0.00032929957,0.0012538591,0.0006950671,0.0018279143,0.0009961477,0.000716944,0.005970567],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012152013,0.0003509454,0.0067104455,0.00074798195,0.0000642531,0.0021129444,0.0055451402,0.0012524955,0.091809824,0.004899067,0.006654574,0.87863725],"study_design_scores_gemma":[0.0002504077,0.0022739654,0.021831265,0.0005329538,0.0002876318,0.008280119,0.0034088017,0.013974281,0.5317778,0.010536612,0.4066711,0.00017510557],"about_ca_topic_score_codex":0.0005454505,"about_ca_topic_score_gemma":0.0007968449,"teacher_disagreement_score":0.014533394,"about_ca_system_score_codex":0.00023051506,"about_ca_system_score_gemma":0.00053462735,"threshold_uncertainty_score":0.04861909},"labels":[],"label_agreement":null},{"id":"W1936582270","doi":"","title":"Improving post-editing and automatic translation by the creation of phraseological databases: an experiment","year":2014,"lang":"en","type":"article","venue":"Digital Access to Libraries (Université catholique de Louvain (UCL), l'Université de Namur (UNamur) and the Université Saint-Louis (USL-B))","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Phraseology; Linguistics; Context (archaeology); Computer science; Lexicon; Machine translation; Corpus linguistics; Natural language processing; Artificial intelligence; Computational linguistics; Lexical database; Encyclopedia; WordNet; History; Philosophy","score_opus":0.011427410147930592,"score_gpt":0.22546893726939873,"score_spread":0.21404152712146815,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1936582270","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9080745,0.002471826,0.048337087,0.0006563806,0.0007177772,0.0012412147,0.0038703326,0.01897004,0.015660774],"genre_scores_gemma":[0.79253197,0.00092355127,0.18112776,0.00078623864,0.00020478327,0.0009499676,0.012416457,0.0018484831,0.009210643],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9965012,0.0012320746,0.00045985222,0.0010076015,0.0006346811,0.00016450408],"domain_scores_gemma":[0.98245275,0.010008247,0.00042835693,0.004654378,0.0019097666,0.000546557],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004364193,0.0011495316,0.0017165722,0.001174909,0.0010582343,0.002164014,0.0023872787,0.0023710658,0.007501026],"category_scores_gemma":[0.016439123,0.00049243314,0.0009238419,0.0021654137,0.0008238837,0.00379986,0.0022489917,0.0018288917,0.0066306493],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.007137689,0.011298882,0.006516903,0.003220035,0.00042916197,0.0023388974,0.0058028633,0.00974767,0.07580826,0.0032007496,0.029132063,0.84536684],"study_design_scores_gemma":[0.0070933886,0.023292253,0.048051424,0.0005967578,0.0022583667,0.006906235,0.009595158,0.2944913,0.36002082,0.017667387,0.229231,0.00079600146],"about_ca_topic_score_codex":0.0017865662,"about_ca_topic_score_gemma":0.0013682779,"teacher_disagreement_score":0.007501026,"about_ca_system_score_codex":0.0004080095,"about_ca_system_score_gemma":0.00086502475,"threshold_uncertainty_score":0.025093436},"labels":[],"label_agreement":null},{"id":"W1936999679","doi":"","title":"The Acadian Nool module: Automatic processing of a regional oral French","year":2013,"lang":"en","type":"article","venue":"The Journal of Macrodynamic Analysis (Memorial University of Newfoundland)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Université de Moncton","funders":"European Regional Development Fund; New Brunswick Innovation Foundation","keywords":"Computer science; Pronoun; Variety (cybernetics); Natural language processing; Linguistics; Artificial intelligence; Vernacular; Rule-based machine translation","score_opus":0.008185018520655016,"score_gpt":0.2239361385557385,"score_spread":0.21575112003508348,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1936999679","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13576505,0.00092259026,0.5647559,0.00064249086,0.00027756664,0.0010064227,0.03758704,0.21096432,0.048078526],"genre_scores_gemma":[0.29467568,0.0005901087,0.59816647,0.00033145928,0.00017585931,0.0011078807,0.057387665,0.016376033,0.031188903],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99945694,0.00008185226,0.000039520513,0.00024405828,0.00010336552,0.00007417564],"domain_scores_gemma":[0.9991897,0.0002542268,0.000054876735,0.00014030917,0.00031567397,0.000045260203],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008108037,0.0013629418,0.0005522765,0.0037089405,0.0010215353,0.0023825036,0.0006434721,0.00045052962,0.019171653],"category_scores_gemma":[0.0020609,0.00034093263,0.0006965466,0.0014114435,0.0005734559,0.0012208772,0.0011123,0.00045882125,0.007095505],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00046743374,0.000106786894,0.007598244,0.00069878076,0.00010901081,0.00084446877,0.0027004555,0.0016632305,0.06416837,0.007077751,0.083206244,0.8313592],"study_design_scores_gemma":[0.00014014143,0.00030642885,0.04305972,0.00021283007,0.00016458001,0.0017686747,0.0041017598,0.07140317,0.16542567,0.006361072,0.7068044,0.00025157086],"about_ca_topic_score_codex":0.022463089,"about_ca_topic_score_gemma":0.023473404,"teacher_disagreement_score":0.9775369,"about_ca_system_score_codex":0.0011822263,"about_ca_system_score_gemma":0.0017958627,"threshold_uncertainty_score":0.06413555},"labels":[],"label_agreement":null},{"id":"W1938147042","doi":"10.21248/hpsg.2004.24","title":"Lexical Resource Semantics: From theory to implementation","year":2004,"lang":"en","type":"article","venue":"Proceedings of the International Conference on Head-Driven Phrase Structure Grammar","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Semantics (computer science); Operational semantics; Formal semantics (linguistics); Programming language; Constraint (computer-aided design); Computational semantics; Extension (predicate logic); Resource (disambiguation); Feature (linguistics); Natural language processing; Artificial intelligence; Theoretical computer science; Linguistics; Mathematics","score_opus":0.020711946730825395,"score_gpt":0.3114052702362699,"score_spread":0.2906933235054445,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1938147042","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.003925883,0.0020200901,0.94996834,0.0050234175,0.00023686988,0.0001547305,0.00022068091,0.0027276757,0.035722315],"genre_scores_gemma":[0.20781292,0.003967441,0.76975,0.0016495211,0.0005973465,0.00083334045,0.00065458316,0.0016002357,0.013134602],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99678826,0.0012507312,0.00030286866,0.00052627973,0.000867766,0.00026403303],"domain_scores_gemma":[0.9973464,0.0011721216,0.000111885325,0.0008853631,0.00032417636,0.00016001909],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004517542,0.0008884189,0.0011278323,0.002588751,0.001414189,0.011179364,0.0044389586,0.0024179572,0.016195934],"category_scores_gemma":[0.008926431,0.0015680925,0.0014745088,0.0029187168,0.010131129,0.023882024,0.005827455,0.004372523,0.0042324495],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000014813309,0.000018613713,0.00007216522,0.000081900886,0.0000086369255,0.00003106574,0.00017248436,0.0014680325,0.00029283666,0.97129595,0.0016865236,0.024857022],"study_design_scores_gemma":[0.000019745228,0.000020841797,0.000038414004,0.000108548236,0.000011639601,0.00007769298,0.00015545674,0.013790921,0.0008093174,0.9520861,0.032861777,0.000019421244],"about_ca_topic_score_codex":0.0029786981,"about_ca_topic_score_gemma":0.0020331086,"teacher_disagreement_score":0.016195934,"about_ca_system_score_codex":0.0031883628,"about_ca_system_score_gemma":0.0030388678,"threshold_uncertainty_score":0.05418074},"labels":[],"label_agreement":null},{"id":"W1942226195","doi":"10.1007/978-3-540-70596-3_17","title":"Employing a Domain Specific Ontology to Perform Semantic Search","year":2008,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Computer science; Ontology; Information retrieval; Semantic search; Semantic Web; Domain (mathematical analysis); Ontology learning; Upper ontology; Metadata; World Wide Web; Natural language; Suggested Upper Merged Ontology; Natural language processing","score_opus":0.024952012683194403,"score_gpt":0.278435293524624,"score_spread":0.25348328084142957,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1942226195","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010226846,0.00019228678,0.97659177,0.00048117724,0.00010526714,0.00020208042,0.0007628596,0.0020665228,0.009371154],"genre_scores_gemma":[0.08319283,0.0006144644,0.9093612,0.00020861646,0.000033721368,0.000117592994,0.0027187562,0.00030195355,0.0034508193],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99879694,0.00025288662,0.00018990053,0.00020667916,0.00045889604,0.00009463707],"domain_scores_gemma":[0.998604,0.0004895932,0.00007470627,0.00048472223,0.00029021854,0.000056729084],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017362399,0.0004906754,0.00071133545,0.0033532237,0.0011444308,0.0029638375,0.0013108604,0.0011292697,0.0026760916],"category_scores_gemma":[0.0045582643,0.0004802205,0.0017457312,0.0039952467,0.000964572,0.0063485163,0.0026867024,0.0016741602,0.0016096437],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013966789,0.00040105663,0.0029697446,0.0010408541,0.0002474531,0.0008997445,0.0015297887,0.013792525,0.044702336,0.42761606,0.018882388,0.4877784],"study_design_scores_gemma":[0.000043781543,0.00008914432,0.0017412921,0.00043754018,0.00043275353,0.0016810423,0.0014662198,0.29282758,0.05549229,0.36205018,0.2836036,0.0001345695],"about_ca_topic_score_codex":0.007483081,"about_ca_topic_score_gemma":0.01467187,"teacher_disagreement_score":0.007483081,"about_ca_system_score_codex":0.001075134,"about_ca_system_score_gemma":0.0027548503,"threshold_uncertainty_score":0.014879048},"labels":[],"label_agreement":null},{"id":"W194491081","doi":"10.82308/16076","title":"Crosslingual implementation of linguistic taggers using parallel corpora","year":2008,"lang":"en","type":"book","venue":"eScholarship@McGill (McGill)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Natural language processing; Artificial intelligence; Sentence; Set (abstract data type); Part of speech; Linguistics; Programming language","score_opus":0.029896249771264885,"score_gpt":0.2952183447783031,"score_spread":0.26532209500703824,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W194491081","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.054584842,0.00053597,0.81321317,0.00045766297,0.0009800404,0.0008613308,0.00476838,0.10221532,0.022383263],"genre_scores_gemma":[0.14626706,0.00044836305,0.8109997,0.0003169971,0.00011698815,0.0008339798,0.015524696,0.0080061825,0.017486045],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99667895,0.00083112705,0.00048354405,0.0010874124,0.0007149879,0.00020396664],"domain_scores_gemma":[0.99345064,0.0021003357,0.000274262,0.0024641985,0.0015428277,0.0001677464],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0041202945,0.0011845388,0.0010622719,0.003077521,0.0017366814,0.0043547633,0.0024799246,0.001466244,0.014363597],"category_scores_gemma":[0.01100694,0.0017239732,0.0015817943,0.0032516676,0.00096936536,0.007909001,0.004051274,0.0019919085,0.012025652],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009457723,0.0008908957,0.007678552,0.0010749036,0.00038284715,0.0014406604,0.004078892,0.01442011,0.06861063,0.023047775,0.037580226,0.8398488],"study_design_scores_gemma":[0.0004933891,0.000582024,0.012372457,0.00030942308,0.00047059517,0.0018387893,0.0043396796,0.35227928,0.25482988,0.04598129,0.32604328,0.00045992207],"about_ca_topic_score_codex":0.0075662695,"about_ca_topic_score_gemma":0.011774586,"teacher_disagreement_score":0.014363597,"about_ca_system_score_codex":0.0012872803,"about_ca_system_score_gemma":0.002502048,"threshold_uncertainty_score":0.048051},"labels":[],"label_agreement":null},{"id":"W194948764","doi":"","title":"HCP with PSMA: a robust spoken language parser","year":2007,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba","funders":"","keywords":"Computer science; Parsing; Natural language processing; Spoken language; Artificial intelligence; Parser combinator; Phrase; Sentence; Grammar; Speech recognition; Natural language; Top-down parsing; Linguistics","score_opus":0.009845508503019063,"score_gpt":0.24986076972293564,"score_spread":0.24001526121991656,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W194948764","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0059773806,0.000063682084,0.98133343,0.00011224323,0.000045240686,0.00007973474,0.00013973213,0.011027893,0.001220648],"genre_scores_gemma":[0.12747961,0.000070171125,0.86697567,0.00017054584,0.00006446293,0.00017817519,0.00037338486,0.00082533964,0.0038625288],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99931943,0.00015455339,0.00004092399,0.00025439152,0.00019119681,0.000039430153],"domain_scores_gemma":[0.99908245,0.0004668416,0.000050494462,0.00017936704,0.00019137254,0.000029492621],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010655741,0.00063205644,0.0006541381,0.00064053823,0.00057944516,0.0009831754,0.0016330641,0.0010798429,0.0050950735],"category_scores_gemma":[0.0029407567,0.0006262509,0.00057154556,0.0006197273,0.0007339421,0.0016990798,0.0010772772,0.0014941599,0.0017672578],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00030227078,0.00017983839,0.000918925,0.00021356074,0.00016526159,0.000539693,0.00033594324,0.06847134,0.07055911,0.057108324,0.01974265,0.781463],"study_design_scores_gemma":[0.000037764068,0.000060521666,0.0004243639,0.000010439305,0.00004540136,0.00021202852,0.000027583446,0.926574,0.03406067,0.027857373,0.010657825,0.00003213338],"about_ca_topic_score_codex":0.0033798204,"about_ca_topic_score_gemma":0.0037513503,"teacher_disagreement_score":0.0050950735,"about_ca_system_score_codex":0.0005958249,"about_ca_system_score_gemma":0.0015753418,"threshold_uncertainty_score":0.017044663},"labels":[],"label_agreement":null},{"id":"W1950554513","doi":"","title":"Towards an optimal weighting of context words based on distance","year":2010,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Weighting; Computer science; Word (group theory); Artificial intelligence; Context (archaeology); Window (computing); Similarity (geometry); Schema (genetic algorithms); Natural language processing; Pattern recognition (psychology); Mathematics; Machine learning; Image (mathematics)","score_opus":0.009410203804900258,"score_gpt":0.2717018169574011,"score_spread":0.2622916131525008,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1950554513","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.034936216,0.0005282849,0.96328294,0.000097375036,0.00003929601,0.0000733031,0.00009896668,0.00045498327,0.0004886499],"genre_scores_gemma":[0.18834145,0.00033798,0.8097377,0.00008707329,0.000050601284,0.00018115556,0.0003694715,0.00031608887,0.0005784296],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99578226,0.001558229,0.00041749433,0.0012763593,0.0007862361,0.00017945594],"domain_scores_gemma":[0.9956241,0.0018899768,0.00038033002,0.0006933749,0.0012720024,0.0001403182],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003199571,0.0014609896,0.0021221268,0.0051820967,0.0009603533,0.0018548127,0.0018609248,0.0015508927,0.00081031007],"category_scores_gemma":[0.01431103,0.001043392,0.00091241935,0.0036082116,0.0012161069,0.0046530557,0.0025921168,0.0017515704,0.0006525186],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00054188597,0.00034725363,0.008637956,0.0005039018,0.00031913898,0.00015855474,0.0009800191,0.12404606,0.04553521,0.034896854,0.005795125,0.77823806],"study_design_scores_gemma":[0.000077895245,0.00010734177,0.0020302688,0.000052245246,0.00006750523,0.00016314385,0.00017696603,0.92780155,0.011468438,0.05514045,0.0028376281,0.00007661588],"about_ca_topic_score_codex":0.0043107197,"about_ca_topic_score_gemma":0.006450602,"teacher_disagreement_score":0.0051820967,"about_ca_system_score_codex":0.001272364,"about_ca_system_score_gemma":0.001862829,"threshold_uncertainty_score":0.016921163},"labels":[],"label_agreement":null},{"id":"W19546019","doi":"10.1016/j.jocd.2009.05.001","title":"Language Identification Strategies for Cross Language Information Retrieval.","year":2010,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Language identification; Identification (biology); Task (project management); Natural language; Information retrieval; Grammar; Language model; Metadata; Linguistics; World Wide Web","score_opus":0.007840132938790805,"score_gpt":0.3097372937516138,"score_spread":0.301897160812823,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W19546019","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01427447,0.0030441133,0.94354475,0.0023172423,0.00031969996,0.001925424,0.0050516445,0.019369109,0.010153516],"genre_scores_gemma":[0.11538402,0.0020564385,0.8564451,0.0010185007,0.00024399985,0.0017271236,0.01432587,0.0011333188,0.007665739],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99282736,0.0036626777,0.0010831384,0.0008947626,0.001128705,0.00040331477],"domain_scores_gemma":[0.97769964,0.015167433,0.0009601621,0.0019940997,0.0037838246,0.0003948455],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009779995,0.0018639261,0.0017858878,0.009635076,0.0018953569,0.0047413907,0.002602851,0.002372948,0.023959173],"category_scores_gemma":[0.03491482,0.0008524655,0.0021613592,0.006169723,0.0011339427,0.010926181,0.0060362564,0.0015790835,0.014675046],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006312827,0.00040368913,0.0024805581,0.0021547538,0.00026996434,0.0011087512,0.0029502816,0.00302404,0.010311838,0.027109394,0.048289932,0.9012654],"study_design_scores_gemma":[0.00048538568,0.0006231798,0.00549729,0.0015461902,0.0010037427,0.004149153,0.016253294,0.35338622,0.044439506,0.3205323,0.25164416,0.00043952971],"about_ca_topic_score_codex":0.0049306406,"about_ca_topic_score_gemma":0.0047849105,"teacher_disagreement_score":0.023959173,"about_ca_system_score_codex":0.0013296809,"about_ca_system_score_gemma":0.0032381,"threshold_uncertainty_score":0.08015138},"labels":[],"label_agreement":null},{"id":"W1956103381","doi":"","title":"Automatic Acquisition of Lexical Formality","year":2010,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":56,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Formality; Computer science; Word (group theory); Natural language processing; Artificial intelligence; Word Association; Metric (unit); Similarity (geometry); Association (psychology); Task (project management); Synonym (taxonomy); Linguistics; Speech recognition; Psychology","score_opus":0.0284122819659165,"score_gpt":0.3393629900576921,"score_spread":0.3109507080917756,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1956103381","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.31549624,0.0018253406,0.6265804,0.0007653699,0.00040077374,0.00057657313,0.0050962,0.032364693,0.0168944],"genre_scores_gemma":[0.66334444,0.00051767763,0.32234108,0.0001528582,0.0001473376,0.00027219663,0.008415918,0.0012969736,0.0035115369],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9976675,0.000461404,0.00030264448,0.00085575506,0.0005591194,0.00015351648],"domain_scores_gemma":[0.9901086,0.004382833,0.0010587602,0.0014617491,0.0026734597,0.0003145325],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021890511,0.0009597391,0.0012739331,0.0072820806,0.00080276985,0.0029582642,0.001291727,0.00061217614,0.0060697906],"category_scores_gemma":[0.015714016,0.0008203471,0.00070406037,0.002430234,0.00077885936,0.005937049,0.00400361,0.0014518676,0.0038987005],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029409584,0.00014376038,0.018412687,0.00089042174,0.00010956997,0.00055050413,0.0015796226,0.0014727585,0.115010835,0.011439104,0.009114193,0.84098244],"study_design_scores_gemma":[0.00034373943,0.0007875534,0.11593482,0.00083261187,0.00048638546,0.006344464,0.005297728,0.41881517,0.18925267,0.1408122,0.12052107,0.0005716207],"about_ca_topic_score_codex":0.0011920779,"about_ca_topic_score_gemma":0.0021416396,"teacher_disagreement_score":0.0072820806,"about_ca_system_score_codex":0.0006262375,"about_ca_system_score_gemma":0.0014796808,"threshold_uncertainty_score":0.020305455},"labels":[],"label_agreement":null},{"id":"W1960827080","doi":"10.1109/icdar.1997.620559","title":"HMM word recognition engine","year":2002,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":52,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Hidden Markov model; Computer science; Word (group theory); Speech recognition; Feature (linguistics); Scheme (mathematics); Artificial intelligence; Cheque; Natural language processing; Word recognition; Pattern recognition (psychology); World Wide Web; Linguistics","score_opus":0.024997555857519043,"score_gpt":0.23948095273907602,"score_spread":0.214483396881557,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1960827080","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007451539,0.0005401638,0.8075963,0.00012589314,0.0002990927,0.0005010141,0.014537591,0.15775706,0.011191307],"genre_scores_gemma":[0.09336834,0.0008439226,0.78397655,0.00047501328,0.00017093678,0.0010796029,0.06154609,0.005708649,0.05283085],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991462,0.00007637885,0.00009662174,0.00030528387,0.00029334176,0.000082092876],"domain_scores_gemma":[0.99872106,0.0003284137,0.00007319513,0.00027848198,0.00054138957,0.00005745277],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009090542,0.0010308851,0.0016330035,0.0017292793,0.0005043621,0.0014139062,0.0018858749,0.00097178906,0.045056965],"category_scores_gemma":[0.0023177345,0.0007650958,0.00092044426,0.0012507079,0.00025103547,0.0019798214,0.0009553094,0.0011218078,0.047147103],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00071088795,0.0003563908,0.0046693655,0.0009507145,0.0003466872,0.00057583,0.00027215783,0.014070476,0.081854634,0.008841257,0.1367615,0.7505901],"study_design_scores_gemma":[0.00021702862,0.00038325423,0.012664542,0.00019183748,0.0004761686,0.0017413422,0.00021127153,0.4031962,0.1921984,0.01580873,0.37254414,0.00036704683],"about_ca_topic_score_codex":0.005396612,"about_ca_topic_score_gemma":0.006088835,"teacher_disagreement_score":0.045056965,"about_ca_system_score_codex":0.00046102915,"about_ca_system_score_gemma":0.0011853301,"threshold_uncertainty_score":0.15073055},"labels":[],"label_agreement":null},{"id":"W1964829063","doi":"10.1037/h0087461","title":"Abstract analogies and positive transfer in artificial grammar learning.","year":2005,"lang":"en","type":"article","venue":"Canadian Journal of Experimental Psychology/Revue canadienne de psychologie expérimentale","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":29,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Lethbridge","funders":"","keywords":"Grammaticality; Psychology; Analogy; Similarity (geometry); Grammar; Linguistics; Vocabulary; Natural language processing; Artificial intelligence; Abstraction; Cognitive psychology; Computer science","score_opus":0.024371112078868735,"score_gpt":0.3005866132060053,"score_spread":0.2762155011271366,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1964829063","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8214442,0.00086898374,0.10719839,0.0022598547,0.00023297087,0.00021720016,0.00011387265,0.00039092,0.06727367],"genre_scores_gemma":[0.9903058,0.00013151736,0.0071423687,0.0001977562,0.00003417062,0.000077641365,0.000038096103,0.000022398744,0.0020502512],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.9980344,0.0007702115,0.00008347449,0.00038958582,0.0006316529,0.00009062427],"domain_scores_gemma":[0.98427725,0.010458038,0.0013789985,0.0028435148,0.0005217131,0.0005205194],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023446153,0.00045833868,0.0003564597,0.00058767124,0.00025058232,0.00091987354,0.0010094595,0.0008083169,0.0077406825],"category_scores_gemma":[0.028769787,0.00022617825,0.00045425852,0.00037696116,0.0041352934,0.0033312012,0.0025789838,0.002021094,0.00046342585],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012984516,0.002362696,0.016737359,0.0009880802,0.00031293606,0.0014560763,0.0027204957,0.020970639,0.077639796,0.5016489,0.0026585856,0.37120596],"study_design_scores_gemma":[0.00018397125,0.0011484232,0.015348568,0.00004248098,0.00006511412,0.0011657286,0.00015831176,0.020747585,0.016618716,0.94138604,0.0030982683,0.000036842914],"about_ca_topic_score_codex":0.00019627431,"about_ca_topic_score_gemma":0.00014834541,"teacher_disagreement_score":0.0077406825,"about_ca_system_score_codex":0.0005691932,"about_ca_system_score_gemma":0.00032188304,"threshold_uncertainty_score":0.025895119},"labels":[],"label_agreement":null},{"id":"W1965104780","doi":"10.1145/1992896.1992911","title":"Attempts to verify written English","year":2011,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Computer science; Correctness; GRASP; Grammar; Implementation; Programming language; Natural language processing; Natural language; Artificial intelligence; Natural (archaeology); Linguistics","score_opus":0.023219143036950985,"score_gpt":0.253226255208647,"score_spread":0.230007112171696,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1965104780","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06590309,0.00023699702,0.9221879,0.001251234,0.00022759724,0.00016834929,0.00031478677,0.0036498492,0.006060141],"genre_scores_gemma":[0.42495736,0.0003820337,0.56828135,0.0005038865,0.00012883711,0.00015780608,0.0011469923,0.001107039,0.0033346526],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.98725104,0.004700579,0.0012139514,0.0022468397,0.0038288615,0.00075861654],"domain_scores_gemma":[0.9185672,0.05328176,0.0032591582,0.013099139,0.011206954,0.0005857573],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006805356,0.0010226605,0.0010454899,0.0015543913,0.00233359,0.0049867183,0.0027508328,0.0019504627,0.0057406337],"category_scores_gemma":[0.088011175,0.0012901347,0.002205522,0.0011211876,0.0049186302,0.009932995,0.0040926947,0.0036339988,0.0013958226],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00061718083,0.00034873307,0.013623713,0.0024309533,0.00043608184,0.001844418,0.010806483,0.03674167,0.06677713,0.5633643,0.008456514,0.29455277],"study_design_scores_gemma":[0.0001961032,0.00045350124,0.0035497835,0.00053609663,0.00039343027,0.0013927259,0.0028281834,0.20955771,0.18784904,0.54800683,0.045034364,0.00020223018],"about_ca_topic_score_codex":0.0027631118,"about_ca_topic_score_gemma":0.003601301,"teacher_disagreement_score":0.006805356,"about_ca_system_score_codex":0.0012684102,"about_ca_system_score_gemma":0.0034148183,"threshold_uncertainty_score":0.035990596},"labels":[],"label_agreement":null},{"id":"W1965605789","doi":"10.1145/502512.502559","title":"DIRT @SBT@discovery of inference rules from text","year":2001,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":548,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Dirt; Computer science; Inference; Rule of inference; Artificial intelligence; Natural language processing; Information retrieval; Data science; Data mining; Cartography; Geography","score_opus":0.014375009322980968,"score_gpt":0.2750401360449637,"score_spread":0.2606651267219827,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1965605789","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0072057536,0.0004358421,0.97841746,0.0008671803,0.00015799674,0.00021758744,0.0022399207,0.0077221277,0.0027361175],"genre_scores_gemma":[0.05901876,0.00041672218,0.9221076,0.00058573764,0.00021185495,0.00034417046,0.011704686,0.0009613621,0.0046491735],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99346465,0.0016087673,0.00052985933,0.002126344,0.002071321,0.00019905675],"domain_scores_gemma":[0.9758351,0.016366428,0.0011833379,0.0041292896,0.0021016272,0.00038436853],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004447702,0.0010656111,0.0013664719,0.005606521,0.0015802642,0.0035669112,0.0035180235,0.0021195868,0.007314675],"category_scores_gemma":[0.029615955,0.0010080088,0.0020551328,0.0034867977,0.0018178687,0.005588131,0.002483728,0.0028860674,0.0080629485],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020389313,0.0003089711,0.007127674,0.0005940963,0.0002943019,0.0006927142,0.0005344043,0.014898482,0.011416723,0.03711486,0.036834903,0.88997906],"study_design_scores_gemma":[0.00012108806,0.00009646243,0.004032381,0.00017691923,0.00022676216,0.0011427836,0.00024455212,0.6611633,0.035050794,0.21412168,0.083507515,0.000115770476],"about_ca_topic_score_codex":0.0042844308,"about_ca_topic_score_gemma":0.008585417,"teacher_disagreement_score":0.007314675,"about_ca_system_score_codex":0.0011138062,"about_ca_system_score_gemma":0.0023642574,"threshold_uncertainty_score":0.024470031},"labels":[],"label_agreement":null},{"id":"W1965923765","doi":"10.3115/1596431.1596438","title":"Non-classical lexical semantic relations","year":2004,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":127,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"WordNet; Computer science; Natural language processing; Artificial intelligence; Context (archaeology); Linguistics; Lexical choice; Lexical item; Philosophy; History","score_opus":0.010902032469480164,"score_gpt":0.269521507400675,"score_spread":0.2586194749311948,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1965923765","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06351005,0.011664814,0.7111025,0.004907404,0.0008420014,0.0002281851,0.0012074913,0.0010582341,0.20547931],"genre_scores_gemma":[0.71486455,0.00553095,0.24404308,0.0014976916,0.001119618,0.00040000895,0.0014741999,0.00029696533,0.030772913],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9942532,0.0014663639,0.00056265347,0.0017938676,0.0016947552,0.00022913345],"domain_scores_gemma":[0.99269533,0.0044254884,0.0005409627,0.0012624658,0.0009395777,0.0001361595],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032923413,0.0008133979,0.0009185808,0.0038949375,0.0022018787,0.0063211387,0.0018940079,0.0019441609,0.0123125],"category_scores_gemma":[0.014248615,0.00054391014,0.0008905135,0.004254494,0.007816381,0.02109302,0.003817006,0.0024830846,0.0023601737],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00001676904,0.000013596712,0.00041559836,0.00020760656,0.000019404159,0.00015601356,0.0006044903,0.00046118733,0.00073781534,0.9755822,0.0012337464,0.020551566],"study_design_scores_gemma":[0.0000059565605,0.0000071132326,0.00045607385,0.00004702109,0.000017907923,0.0002479075,0.0002598972,0.0030252754,0.00056723563,0.97923595,0.016115487,0.000014149766],"about_ca_topic_score_codex":0.0009813388,"about_ca_topic_score_gemma":0.0016530969,"teacher_disagreement_score":0.0123125,"about_ca_system_score_codex":0.0021127537,"about_ca_system_score_gemma":0.0014086122,"threshold_uncertainty_score":0.041189432},"labels":[],"label_agreement":null},{"id":"W1965995772","doi":"10.1145/1458550.1458569","title":"Clustering the topics using TF-IDF for model fusion","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; tf–idf; Clef; Cluster analysis; Information retrieval; Task (project management); Variety (cybernetics); Question answering; Term (time); Data mining; Artificial intelligence; Machine learning","score_opus":0.05968255531486609,"score_gpt":0.3047592944829973,"score_spread":0.24507673916813122,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1965995772","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01141184,0.00029625953,0.9829444,0.00011514974,0.000074928794,0.0001706184,0.0005542035,0.004001114,0.0004314997],"genre_scores_gemma":[0.16168824,0.00035079266,0.83039683,0.00009196966,0.00013596618,0.00057409774,0.0045740297,0.00074413954,0.0014440035],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9955987,0.0014654608,0.00039013545,0.001207604,0.00087244174,0.00046572002],"domain_scores_gemma":[0.99552757,0.0018145044,0.00019929098,0.0010832478,0.0012609845,0.00011449296],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0065630213,0.0029154285,0.0027646467,0.007669288,0.0015360394,0.0023976294,0.0024792347,0.0025807237,0.0028343997],"category_scores_gemma":[0.018177826,0.000963021,0.004478784,0.0056007975,0.00079513644,0.003663378,0.0017672115,0.0030402967,0.0035346884],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00065003295,0.00044989574,0.004097215,0.0003396485,0.0005650046,0.00028476334,0.0010446004,0.22374478,0.02619169,0.005394094,0.011012023,0.7262263],"study_design_scores_gemma":[0.000032465014,0.00008072844,0.0011083866,0.000027419203,0.000107166605,0.00012106909,0.00016505062,0.97381747,0.010208671,0.010125649,0.00412134,0.00008457234],"about_ca_topic_score_codex":0.01964384,"about_ca_topic_score_gemma":0.012525484,"teacher_disagreement_score":0.01964384,"about_ca_system_score_codex":0.001919369,"about_ca_system_score_gemma":0.0019015562,"threshold_uncertainty_score":0.039058983},"labels":[],"label_agreement":null},{"id":"W1966216812","doi":"10.3115/1117755.1117757","title":"Adapting a synonym database to specific domains","year":2000,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":31,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Pruning; WordNet; Computer science; Synonym (taxonomy); Domain (mathematical analysis); Lexical database; Information retrieval; Artificial intelligence; Natural language processing; Database; Mathematics","score_opus":0.016652075933204197,"score_gpt":0.2667677192926828,"score_spread":0.25011564335947856,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1966216812","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015974162,0.00039996125,0.9731108,0.00024055221,0.00010542787,0.00035766442,0.0005995792,0.0055839913,0.0036278819],"genre_scores_gemma":[0.07020214,0.00049690984,0.9215037,0.00030076853,0.000053754,0.00033025778,0.0034114418,0.0010556333,0.0026453615],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99315083,0.0018888569,0.0008106443,0.0017376039,0.0021801724,0.00023183988],"domain_scores_gemma":[0.9881864,0.003475395,0.00046111894,0.0049111266,0.002711624,0.00025426992],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0049130484,0.0009862079,0.0016967246,0.007651176,0.0016283644,0.003648366,0.0042401673,0.001300999,0.003111189],"category_scores_gemma":[0.019044284,0.0011633442,0.0017478924,0.004590322,0.00086580607,0.006007179,0.0050676693,0.0021478718,0.0022371823],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002884178,0.00033408077,0.0050808974,0.00095968513,0.0004908017,0.0012899401,0.002348964,0.01475848,0.05435321,0.039678354,0.020165391,0.86025184],"study_design_scores_gemma":[0.00018657604,0.00031865548,0.0056665344,0.0006500522,0.00073534995,0.005657096,0.0029288873,0.29738238,0.10710703,0.11291736,0.46604624,0.00040379522],"about_ca_topic_score_codex":0.0021978626,"about_ca_topic_score_gemma":0.004806241,"teacher_disagreement_score":0.007651176,"about_ca_system_score_codex":0.0008443777,"about_ca_system_score_gemma":0.0016569842,"threshold_uncertainty_score":0.025982976},"labels":[],"label_agreement":null},{"id":"W1966682361","doi":"10.1075/term.8.1.02mar","title":"French patterns for expressing concept relations","year":2002,"lang":"en","type":"article","venue":"Terminology International Journal of Theoretical and Applied Issues in Specialized Communication","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":83,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Linguistics; Computer science; Function (biology); Natural language processing; Artificial intelligence; Philosophy","score_opus":0.018220220493002252,"score_gpt":0.3132039937137447,"score_spread":0.29498377322074243,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1966682361","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.039457053,0.0030550035,0.891602,0.0019969791,0.0003434677,0.00043830424,0.0061063976,0.003122213,0.053878654],"genre_scores_gemma":[0.15970495,0.0019574687,0.8188239,0.0005804716,0.00013064679,0.0006879203,0.0051351096,0.00044819532,0.012531378],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"qualitative","domain_scores_codex":[0.998159,0.0005588214,0.00036353566,0.00037897236,0.00038711788,0.00015257424],"domain_scores_gemma":[0.9982368,0.00066589797,0.00026012847,0.00040826495,0.0003737115,0.000055269782],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014886205,0.0011312487,0.00052650645,0.0053505087,0.0016208513,0.0035158156,0.0008992771,0.0011609647,0.008885303],"category_scores_gemma":[0.003911896,0.0005799073,0.00110326,0.0062713884,0.0016263631,0.004893317,0.00130397,0.0015387755,0.0022527003],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010925671,0.000041352705,0.0033303578,0.000533856,0.000053787193,0.0005358212,0.0040118126,0.0015608367,0.008425038,0.6885054,0.015356501,0.277536],"study_design_scores_gemma":[0.000041380143,0.00013089235,0.0037102667,0.00048909517,0.00010761901,0.0030587409,0.0022626622,0.013089773,0.012617048,0.30795485,0.65643245,0.000105253326],"about_ca_topic_score_codex":0.009264897,"about_ca_topic_score_gemma":0.012498066,"teacher_disagreement_score":0.009264897,"about_ca_system_score_codex":0.0015649169,"about_ca_system_score_gemma":0.00129041,"threshold_uncertainty_score":0.0297243},"labels":[],"label_agreement":null},{"id":"W1967740833","doi":"10.1109/icalt.2013.164","title":"Writing-Based Learning Analytics for Education","year":2013,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Athabasca University","funders":"Beijing Normal University; Athabasca University","keywords":"Analytics; Computer science; Data science; Field (mathematics); Software analytics; Learning analytics; Cultural analytics; The Internet; Web analytics; World Wide Web; Data analysis; Semantic analytics; Software; Web intelligence; Data mining; Software development","score_opus":0.01174043770616022,"score_gpt":0.28679892573166565,"score_spread":0.2750584880255054,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1967740833","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011680505,0.008606021,0.9166963,0.010130547,0.00046650064,0.0009117739,0.009644742,0.019359963,0.022503654],"genre_scores_gemma":[0.16835894,0.004675528,0.80364376,0.0008105505,0.0005051345,0.0011963862,0.013115289,0.00069396343,0.007000429],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9952726,0.0021820865,0.0005667801,0.0006284505,0.0011942283,0.00015593367],"domain_scores_gemma":[0.97956425,0.012299306,0.00146253,0.0032610854,0.0027864748,0.00062636216],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0055748895,0.0013477897,0.001240217,0.0058176015,0.0009849764,0.008678393,0.001775771,0.0014195761,0.009401463],"category_scores_gemma":[0.029542595,0.0003901571,0.00091122225,0.00797812,0.0012332675,0.00838037,0.003253974,0.0023634173,0.0054806937],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016575186,0.00035229072,0.0050223977,0.00096056383,0.000104275125,0.00012941382,0.0011140897,0.008678045,0.0018721139,0.10240641,0.046307433,0.8328871],"study_design_scores_gemma":[0.000051257703,0.00014781665,0.0051259757,0.0010702698,0.000056541674,0.00018473559,0.0018041274,0.15736808,0.004621941,0.658206,0.17125863,0.00010468863],"about_ca_topic_score_codex":0.002196906,"about_ca_topic_score_gemma":0.0019871357,"teacher_disagreement_score":0.009401463,"about_ca_system_score_codex":0.0015980751,"about_ca_system_score_gemma":0.0025416974,"threshold_uncertainty_score":0.031451046},"labels":[],"label_agreement":null},{"id":"W1967963033","doi":"10.7202/1006182ar","title":"Dutch Parallel Corpus: A Balanced Copyright-Cleared Parallel Corpus","year":2011,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":77,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Parallel corpora; Sentence; Corpus linguistics; Text corpus; Metadata; Information retrieval; Machine translation; World Wide Web","score_opus":0.04397547829965751,"score_gpt":0.26540036135321504,"score_spread":0.22142488305355754,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1967963033","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07467075,0.0049512116,0.13937385,0.0031138458,0.0027801606,0.0049821483,0.65751797,0.004987152,0.10762289],"genre_scores_gemma":[0.09460726,0.0016391512,0.12776713,0.00045190938,0.00038246906,0.008776954,0.7344234,0.005200575,0.026751183],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9943743,0.0017186594,0.0011402192,0.0010810414,0.0014773725,0.00020844454],"domain_scores_gemma":[0.98308855,0.005454346,0.0009148661,0.0021871526,0.0077549545,0.00060026],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005251894,0.001062571,0.0012502291,0.005861925,0.0027198573,0.0032050477,0.0019388922,0.0013315763,0.053142257],"category_scores_gemma":[0.022950334,0.0008695832,0.0005955972,0.009653918,0.0014415001,0.0030284738,0.0034665042,0.0017955729,0.018034566],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011109698,0.00035778675,0.0031748386,0.006663521,0.00017195682,0.002749316,0.0068747913,0.0038204486,0.029393084,0.03541341,0.6726194,0.23765047],"study_design_scores_gemma":[0.0003590038,0.000058515398,0.0076138363,0.0003810554,0.00008595819,0.0012391487,0.0011473757,0.002838818,0.0075349063,0.0048800097,0.97374433,0.00011702538],"about_ca_topic_score_codex":0.0217523,"about_ca_topic_score_gemma":0.021097083,"teacher_disagreement_score":0.053142257,"about_ca_system_score_codex":0.002209835,"about_ca_system_score_gemma":0.005527705,"threshold_uncertainty_score":0.17777854},"labels":[],"label_agreement":null},{"id":"W1967976762","doi":"10.1007/s10849-011-9138-9","title":"Introduction to the Special Issue on the Mathematics of Language","year":2011,"lang":"en","type":"article","venue":"Journal of Logic Language and Information","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Mathematics education; Computer science; Linguistics; Mathematics; Philosophy","score_opus":0.012244356999675212,"score_gpt":0.24500111767583185,"score_spread":0.23275676067615664,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1967976762","genre_codex":"editorial","genre_gemma":"editorial","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"editorial","genre_consensus":"editorial","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00082960963,0.08292676,0.033199463,0.08546128,0.67500854,0.000098092954,0.0011093393,0.0006861154,0.12068074],"genre_scores_gemma":[0.00636832,0.040101998,0.00915894,0.017276607,0.7423371,0.0001235015,0.0013693903,0.00090925925,0.18235491],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99872595,0.00025824757,0.00013578252,0.00030246843,0.0004723233,0.00010528063],"domain_scores_gemma":[0.9958948,0.0018961095,0.00020296167,0.0004338156,0.00095340377,0.0006188225],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018597196,0.0015677051,0.00191136,0.0040577203,0.001415446,0.0055703474,0.0016212172,0.0025141276,0.07635386],"category_scores_gemma":[0.0053211525,0.0006770678,0.001980728,0.0024342164,0.0020662507,0.007938437,0.002665916,0.00785864,0.031489607],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00002051831,0.000032369262,0.0001314438,0.00023281101,0.000018999914,0.00006537605,0.00005588983,0.00012128876,0.00041894117,0.032957774,0.9180915,0.047853082],"study_design_scores_gemma":[0.0000061017795,0.00002203881,0.00032335642,0.00011366098,0.000012840555,0.00018622249,0.000026060627,0.00023428409,0.000115631585,0.025196815,0.9737483,0.000014569251],"about_ca_topic_score_codex":0.000575377,"about_ca_topic_score_gemma":0.0013564748,"teacher_disagreement_score":0.07635386,"about_ca_system_score_codex":0.001556596,"about_ca_system_score_gemma":0.0014066463,"threshold_uncertainty_score":0.2554291},"labels":[],"label_agreement":null},{"id":"W1968267429","doi":"10.1177/0023830913484896","title":"Sidestepping the Combinatorial Explosion: An Explanation of <i>n</i> -gram Frequency Effects Based on Naive Discriminative Learning","year":2013,"lang":"en","type":"article","venue":"Language and Speech","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":109,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Alexander von Humboldt-Stiftung","keywords":"Discriminative model; Word lists by frequency; Word (group theory); Natural language processing; Computer science; Artificial intelligence; Frequency; n-gram; Key (lock); Speech recognition; Linguistics; Language model; Mathematics; Statistics","score_opus":0.007765609616439314,"score_gpt":0.24910885941243358,"score_spread":0.24134324979599425,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1968267429","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13477916,0.00075472315,0.839725,0.003455193,0.00020728917,0.00010265962,0.0005156094,0.001368021,0.019092185],"genre_scores_gemma":[0.9091408,0.0004365821,0.08122322,0.0013049674,0.00019380485,0.00018247007,0.00057121913,0.0005777789,0.0063691502],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987606,0.00038945474,0.000063998865,0.00035408995,0.00028125173,0.00015066011],"domain_scores_gemma":[0.99158365,0.005480412,0.0005032925,0.00170351,0.0005234997,0.00020555356],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022703186,0.00081664586,0.0011719201,0.0014187098,0.001158179,0.0021345224,0.0027079324,0.0014289402,0.01092327],"category_scores_gemma":[0.018303068,0.000915873,0.001595751,0.0013453503,0.0043065445,0.006930252,0.0027283344,0.0033597162,0.001541071],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00095421256,0.0003529109,0.017219856,0.00044486212,0.00021668673,0.0015404411,0.0016313439,0.12242336,0.014113387,0.6192797,0.010933228,0.21089],"study_design_scores_gemma":[0.000055440745,0.00008649309,0.0024789055,0.000024768307,0.000033373355,0.0004483383,0.00007298523,0.38074884,0.001928851,0.612345,0.0017302504,0.000046759305],"about_ca_topic_score_codex":0.0032501903,"about_ca_topic_score_gemma":0.0034047668,"teacher_disagreement_score":0.01092327,"about_ca_system_score_codex":0.0012408361,"about_ca_system_score_gemma":0.00068372255,"threshold_uncertainty_score":0.03654194},"labels":[],"label_agreement":null},{"id":"W1969081791","doi":"10.5842/43-0-202","title":"Produk teenoor proses tydens akademiese redigering: Opmerkings as aanduiders van redigeergerigtheid","year":2014,"lang":"nl","type":"article","venue":"Stellenbosch Papers in Linguistics Plus","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Theology; Psychology; Philosophy","score_opus":0.012718122081288271,"score_gpt":0.27268396153289187,"score_spread":0.2599658394516036,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1969081791","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.43777514,0.0073704068,0.004074947,0.008183098,0.0004915329,0.0001621467,0.0003499362,0.00013670251,0.54145616],"genre_scores_gemma":[0.7485193,0.0086829085,0.0044960966,0.0023504375,0.00010711117,0.00019735981,0.00072976836,0.00045476662,0.23446229],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9966545,0.0012983455,0.00013074238,0.00036792518,0.0011312842,0.00041723924],"domain_scores_gemma":[0.9961986,0.0013321349,0.0004868369,0.00042027625,0.000923552,0.00063859834],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004242703,0.00056428224,0.00038243335,0.00067608635,0.0032077283,0.009207473,0.0009717673,0.0011192193,0.038041398],"category_scores_gemma":[0.009383628,0.0005098549,0.0003220018,0.0010304713,0.001905887,0.008524528,0.0041699195,0.0032208702,0.010587294],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005259171,0.0010262101,0.023208955,0.0016019975,0.00006901216,0.0022690299,0.38200638,0.00018935128,0.005492129,0.13347632,0.07310504,0.3770296],"study_design_scores_gemma":[0.000038680668,0.0002508613,0.025048638,0.0009944881,0.000076530334,0.0007341327,0.16648,0.00014990859,0.0033770008,0.010346219,0.7924477,0.000055958015],"about_ca_topic_score_codex":0.011932561,"about_ca_topic_score_gemma":0.02147921,"teacher_disagreement_score":0.038041398,"about_ca_system_score_codex":0.0031775348,"about_ca_system_score_gemma":0.0050918283,"threshold_uncertainty_score":0.1272611},"labels":[],"label_agreement":null},{"id":"W1969359618","doi":"10.3758/s13423-015-0802-y","title":"Lexical stress assignment as a problem of probabilistic inference","year":2015,"lang":"en","type":"article","venue":"Psychonomic Bulletin & Review","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Inference; Stress (linguistics); Psychology; Conceptualization; Probabilistic logic; Bayesian inference; Natural language processing; Bayesian probability; Posterior probability; Word (group theory); Cognitive psychology; Artificial intelligence; Linguistics; Computer science","score_opus":0.03417122246720825,"score_gpt":0.32753565672955254,"score_spread":0.29336443426234426,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1969359618","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011182889,0.010582775,0.96514314,0.006705969,0.0005715451,0.000059485235,0.00029518644,0.00058720016,0.004871834],"genre_scores_gemma":[0.44230044,0.024474131,0.5160812,0.0020227758,0.0037228232,0.00031204053,0.0014381342,0.0004903607,0.009158043],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9963012,0.0014934705,0.00032484654,0.00094549847,0.0007879988,0.00014692897],"domain_scores_gemma":[0.98091805,0.016489204,0.0005668539,0.00084450195,0.0010247459,0.00015667819],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0068979263,0.000891117,0.001807118,0.0023859094,0.0011626906,0.005528816,0.004227073,0.0029455856,0.006147734],"category_scores_gemma":[0.032169435,0.0016192133,0.0019163624,0.0034902126,0.0040577915,0.011886364,0.0026375856,0.0046847165,0.0016889955],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017070954,0.000081340644,0.0024278958,0.0012573289,0.00031863313,0.0002496211,0.0005776847,0.027032483,0.0022764876,0.44839296,0.012475158,0.50473976],"study_design_scores_gemma":[0.000015852247,0.000009685721,0.0005485007,0.00007604596,0.000039503368,0.00011811738,0.00007238158,0.041569583,0.0005663134,0.9505476,0.0064131855,0.000023329958],"about_ca_topic_score_codex":0.0015109846,"about_ca_topic_score_gemma":0.0013811594,"teacher_disagreement_score":0.0068979263,"about_ca_system_score_codex":0.0013359593,"about_ca_system_score_gemma":0.0018081572,"threshold_uncertainty_score":0.03648013},"labels":[],"label_agreement":null},{"id":"W1969995880","doi":"10.7202/019874ar","title":"Proactive Description for Useful Applications: Researching Language Options for Better Translation Practice?*","year":2009,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Interpretation (philosophy); Translation (biology); Duty; Process (computing); Natural language processing; Filter (signal processing); Descriptive research; Linguistics; Artificial intelligence; Sociology; Programming language; Political science","score_opus":0.08503600536652481,"score_gpt":0.36331768590290986,"score_spread":0.278281680536385,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1969995880","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.34715495,0.0061180335,0.32113016,0.090036325,0.00038821241,0.0011955241,0.00038547482,0.0011378814,0.23245351],"genre_scores_gemma":[0.85412616,0.0030945311,0.12864873,0.0018989838,0.000080333724,0.0008655357,0.00024326262,0.0003163627,0.010726103],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.973046,0.022252403,0.00093262724,0.0013032395,0.001767262,0.00069851015],"domain_scores_gemma":[0.9508348,0.03219042,0.0036287918,0.006932315,0.0055574896,0.00085617404],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.029456742,0.000812698,0.0006569802,0.0024221162,0.0027689675,0.016884325,0.0028818909,0.0031330287,0.009412723],"category_scores_gemma":[0.048486907,0.0008366825,0.0005774805,0.004414449,0.008150006,0.037047297,0.004231236,0.0028335508,0.0029416224],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002133585,0.00056758814,0.009431585,0.0020494745,0.000037525875,0.0006982528,0.2275069,0.0008956339,0.0071940613,0.46484208,0.007841984,0.27872154],"study_design_scores_gemma":[0.00014156522,0.00059486873,0.006762642,0.0020512112,0.00007123833,0.0011388235,0.4788869,0.009209545,0.01052403,0.3035229,0.1869823,0.00011393455],"about_ca_topic_score_codex":0.0017629738,"about_ca_topic_score_gemma":0.002673137,"teacher_disagreement_score":0.029456742,"about_ca_system_score_codex":0.004036764,"about_ca_system_score_gemma":0.008242722,"threshold_uncertainty_score":0.15578407},"labels":[],"label_agreement":null},{"id":"W1970578121","doi":"10.7202/018525ar","title":"Compréhension et traduction des locutions verbales1","year":2008,"lang":"fr","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Philosophy; Political science","score_opus":0.0824141890987791,"score_gpt":0.304703360895136,"score_spread":0.2222891717963569,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1970578121","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.27389246,0.0033529575,0.67587495,0.0031017875,0.00050555146,0.00021397372,0.0024912003,0.002944045,0.037623033],"genre_scores_gemma":[0.73864955,0.0025508692,0.23725273,0.00067922793,0.00030446853,0.00029744444,0.0030898473,0.0014083906,0.015767463],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9937098,0.0029073828,0.0005698359,0.0010651025,0.0015155203,0.00023239017],"domain_scores_gemma":[0.97600937,0.016160754,0.0014627577,0.0025189356,0.0037248796,0.00012331078],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036773817,0.0010747792,0.00063943025,0.0022047982,0.00095039664,0.0037086182,0.0009251217,0.0013230785,0.0067073954],"category_scores_gemma":[0.033293854,0.00050403463,0.00067562616,0.001786318,0.0019683086,0.006027257,0.0019005906,0.0015932863,0.002205964],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005436289,0.000092733986,0.01894536,0.002294966,0.00014042665,0.0019606145,0.07137444,0.0026687246,0.13424306,0.09017553,0.009249164,0.66831136],"study_design_scores_gemma":[0.00007383561,0.00061813195,0.051768944,0.0017079609,0.00033968722,0.011221983,0.053648237,0.039527483,0.21049361,0.14076068,0.4893678,0.00047164183],"about_ca_topic_score_codex":0.0046697413,"about_ca_topic_score_gemma":0.0033594973,"teacher_disagreement_score":0.0067073954,"about_ca_system_score_codex":0.00097050064,"about_ca_system_score_gemma":0.0012100685,"threshold_uncertainty_score":0.022438467},"labels":[],"label_agreement":null},{"id":"W1971466919","doi":"10.1017/s0890060401154041","title":"Extracting information from free-text aircraft repair notes","year":2001,"lang":"en","type":"article","venue":"Artificial intelligence for engineering design analysis and manufacturing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Computer science; Technician; Crew; Process (computing); Domain (mathematical analysis); Information extraction; Natural language; Expression (computer science); Action (physics); Information retrieval; Parsing; Natural language processing; Lexical analysis; Artificial intelligence; Programming language; Engineering","score_opus":0.02463548484768225,"score_gpt":0.2577376995914167,"score_spread":0.23310221474373444,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1971466919","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4351331,0.0035483786,0.41714898,0.0014073149,0.00042529503,0.0018191934,0.092375055,0.012675318,0.03546741],"genre_scores_gemma":[0.41759944,0.0024237207,0.45762122,0.0002478247,0.0002486251,0.0006882927,0.10946863,0.000606916,0.0110953385],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99915516,0.00010994426,0.00016582497,0.00018442303,0.00031643346,0.00006816765],"domain_scores_gemma":[0.99466324,0.003091461,0.0004363985,0.00049318466,0.0012292801,0.00008643892],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00057500234,0.0010622587,0.0005934267,0.01074421,0.00064193347,0.0017244344,0.0011103315,0.0012067346,0.005190427],"category_scores_gemma":[0.0056997426,0.0005105598,0.0006524893,0.0063677058,0.00045692944,0.0030876892,0.00086345355,0.0006614865,0.0035271435],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007330038,0.00041558233,0.009031642,0.002650922,0.00008019823,0.0056085885,0.0024249756,0.0044987565,0.13527203,0.00830549,0.019982252,0.8109966],"study_design_scores_gemma":[0.00036314418,0.0007730116,0.09606522,0.0012694399,0.00075534283,0.009573588,0.007371663,0.09856138,0.4096115,0.029385056,0.34574455,0.0005261356],"about_ca_topic_score_codex":0.0032193947,"about_ca_topic_score_gemma":0.004385721,"teacher_disagreement_score":0.01074421,"about_ca_system_score_codex":0.000834095,"about_ca_system_score_gemma":0.0013489242,"threshold_uncertainty_score":0.017363727},"labels":[],"label_agreement":null},{"id":"W1971541835","doi":"10.1017/s1351324901002650","title":"Real-time automatic insertion of accents in French text","year":2001,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; Computer Research Institute of Montréal","funders":"","keywords":"Computer science; Stress (linguistics); Character (mathematics); Word (group theory); Natural language processing; Speech recognition; Artificial intelligence; Linguistics","score_opus":0.004005314190147125,"score_gpt":0.23514128188684852,"score_spread":0.2311359676967014,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1971541835","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.37409866,0.00090155145,0.50442195,0.00040042595,0.00038570666,0.00023840863,0.0009805821,0.11532991,0.0032429013],"genre_scores_gemma":[0.60856485,0.00030134374,0.3812218,0.00016510935,0.00015370714,0.0000858154,0.0014669432,0.0016079062,0.0064324653],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9982893,0.00038735272,0.00013042314,0.00062742137,0.00042655758,0.00013899038],"domain_scores_gemma":[0.99379265,0.0030624892,0.0007950116,0.0008634774,0.0012707309,0.0002156601],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014012033,0.0011549917,0.0009666161,0.0011740486,0.0006261873,0.0014268197,0.00119014,0.0010311931,0.0030241064],"category_scores_gemma":[0.0068376646,0.000530306,0.0004558305,0.00075581315,0.00049712486,0.001312222,0.0007186154,0.000709944,0.003942902],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014181717,0.00016912546,0.004771527,0.0003489088,0.000051880153,0.00067572587,0.0012338068,0.0048425416,0.24648944,0.0006641109,0.0076667136,0.73166806],"study_design_scores_gemma":[0.0001390378,0.0011224849,0.017320763,0.00005542443,0.00015419861,0.0029356822,0.00071048085,0.4064515,0.53901285,0.0015735055,0.030242985,0.00028109318],"about_ca_topic_score_codex":0.0033478497,"about_ca_topic_score_gemma":0.0031293465,"teacher_disagreement_score":0.0033478497,"about_ca_system_score_codex":0.00041222555,"about_ca_system_score_gemma":0.00039457693,"threshold_uncertainty_score":0.010116637},"labels":[],"label_agreement":null},{"id":"W1971737743","doi":"10.1145/1183614.1183744","title":"Filtering or adapting","year":2006,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science","score_opus":0.015664940486166685,"score_gpt":0.2601498708183328,"score_spread":0.24448493033216612,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1971737743","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.070905186,0.002689372,0.90235794,0.00096561323,0.0014176985,0.0008293829,0.0022072417,0.0089221075,0.009705512],"genre_scores_gemma":[0.24517529,0.0029125854,0.7171847,0.0019250037,0.0011598939,0.0015745676,0.011884858,0.003137549,0.015045527],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9940621,0.0018394665,0.00057769817,0.0021187298,0.0010817508,0.0003201889],"domain_scores_gemma":[0.9892279,0.0038527546,0.00050880515,0.0044410913,0.0018082717,0.00016114215],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0070981993,0.0026087316,0.00293569,0.0026726627,0.0016880962,0.0027156891,0.0019444375,0.002160806,0.007247391],"category_scores_gemma":[0.02726625,0.0009865857,0.0017166693,0.0045847823,0.0016481016,0.005805878,0.0027465993,0.002018962,0.0073920093],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001335339,0.00053324114,0.0058675604,0.0019902224,0.0003742955,0.0009165852,0.0016483983,0.022345576,0.10027023,0.011722973,0.030695362,0.8223002],"study_design_scores_gemma":[0.00039666874,0.0014096004,0.015333655,0.0006547051,0.0014266857,0.0036411514,0.0026045772,0.37505525,0.22809105,0.07387059,0.29700488,0.00051125523],"about_ca_topic_score_codex":0.0032225517,"about_ca_topic_score_gemma":0.004425776,"teacher_disagreement_score":0.007247391,"about_ca_system_score_codex":0.00058268115,"about_ca_system_score_gemma":0.0018326341,"threshold_uncertainty_score":0.037539303},"labels":[],"label_agreement":null},{"id":"W1971790006","doi":"10.1162/ling.2007.38.3.413","title":"Diagnosing Cyclicity in Sluicing","year":2007,"lang":"en","type":"article","venue":"Linguistic Inquiry","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":85,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Interrogative word; Phrase; Linguistics; Position (finance); Movement (music); Computer science; Natural language processing; Mathematics; Philosophy","score_opus":0.024722076815036582,"score_gpt":0.3269014120462697,"score_spread":0.30217933523123314,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1971790006","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.56379914,0.00080636697,0.3946036,0.0013054352,0.00008239442,0.00015670391,0.00028397638,0.0017590179,0.037203375],"genre_scores_gemma":[0.95984286,0.00011889891,0.03777774,0.000117554955,0.00002274679,0.000039312603,0.00012869446,0.00019481771,0.0017572657],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9968719,0.0012492565,0.00016993751,0.0007684836,0.00065584166,0.00028463115],"domain_scores_gemma":[0.9871609,0.008087618,0.0013017945,0.002015192,0.0012223293,0.00021220854],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032265102,0.0005157514,0.00041163774,0.0018339509,0.001418543,0.0031302106,0.0017374204,0.0010850339,0.0038924613],"category_scores_gemma":[0.021077566,0.00066003593,0.00045114374,0.0018998013,0.0059430436,0.006625626,0.0045173066,0.0015941209,0.00047094625],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004838493,0.00006547295,0.038986836,0.00041577907,0.00007145801,0.0019410598,0.034666806,0.008418085,0.023218108,0.6476031,0.0023702784,0.24175929],"study_design_scores_gemma":[0.000051525203,0.0001466479,0.012907907,0.00020227318,0.00009895996,0.0016698379,0.019421365,0.09139574,0.04095296,0.7911756,0.041820493,0.00015657],"about_ca_topic_score_codex":0.00552203,"about_ca_topic_score_gemma":0.005510144,"teacher_disagreement_score":0.00552203,"about_ca_system_score_codex":0.0014760611,"about_ca_system_score_gemma":0.0014568578,"threshold_uncertainty_score":0.017063558},"labels":[],"label_agreement":null},{"id":"W1972372917","doi":"10.1145/1871437.1871712","title":"Supervised identification and linking of concept mentions to a domain-specific ontology","year":2010,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Ontology; Annotation; Lift (data mining); Identification (biology); Information retrieval; Task (project management); Domain (mathematical analysis); Artificial intelligence; Field (mathematics); Set (abstract data type); Feature (linguistics); Natural language processing; Training set; Machine learning","score_opus":0.013948574391033105,"score_gpt":0.27521932780325364,"score_spread":0.26127075341222056,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1972372917","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03120998,0.00024792305,0.96082693,0.00026433685,0.00006312045,0.00018237044,0.0005612029,0.005205802,0.001438371],"genre_scores_gemma":[0.19671822,0.00019425996,0.7942129,0.00015502592,0.00007576624,0.0002611918,0.004037249,0.00029581334,0.0040496276],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.997303,0.00058403343,0.00016046996,0.0011273068,0.0006631705,0.00016195513],"domain_scores_gemma":[0.99314463,0.0028576888,0.0008534689,0.0016901281,0.0012306818,0.00022337887],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032688382,0.00096694403,0.0009678787,0.0038093769,0.0010861306,0.0017929572,0.0032906157,0.0015224012,0.0018934536],"category_scores_gemma":[0.009845733,0.0006407694,0.0012335964,0.0028895386,0.0009908779,0.003498014,0.002710036,0.0020189304,0.001427175],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00030465767,0.0008344854,0.013541903,0.0005159763,0.0001947188,0.00033103314,0.0010889842,0.037220024,0.028856048,0.01169727,0.014617231,0.8907977],"study_design_scores_gemma":[0.00003254631,0.00017356485,0.0039863475,0.00007024132,0.00010266004,0.00040797156,0.0004822506,0.90686566,0.035433855,0.033671174,0.018712632,0.00006110394],"about_ca_topic_score_codex":0.0054358733,"about_ca_topic_score_gemma":0.012161707,"teacher_disagreement_score":0.0054358733,"about_ca_system_score_codex":0.0011581767,"about_ca_system_score_gemma":0.0030710965,"threshold_uncertainty_score":0.017287493},"labels":[],"label_agreement":null},{"id":"W1972788481","doi":"10.3115/1599081.1599196","title":"Tighter integration of rule-based and statistical MT in serial system combination","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Computer science; Machine translation; NIST; Task (project management); Natural language processing; Phrase; Language model; Artificial intelligence","score_opus":0.012365454727046666,"score_gpt":0.25235695441439077,"score_spread":0.2399914996873441,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1972788481","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06676373,0.00086981093,0.91466254,0.00030514546,0.000095872434,0.00039963305,0.00012344333,0.010245369,0.0065344474],"genre_scores_gemma":[0.36240277,0.00032675892,0.6293037,0.0005087316,0.00015218774,0.0003288641,0.00089631934,0.001599951,0.0044807144],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9803051,0.008528123,0.0019838002,0.003185648,0.0054928954,0.0005044279],"domain_scores_gemma":[0.96992636,0.010901129,0.0016761675,0.011136932,0.0059540877,0.00040533402],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012701679,0.0018206189,0.002647646,0.003655811,0.00091237685,0.003378309,0.0024270515,0.0016075876,0.0033562528],"category_scores_gemma":[0.032277696,0.002000798,0.001047202,0.0037123198,0.0013924013,0.0057280273,0.004868894,0.0026541904,0.0030205932],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00066133897,0.0008175365,0.0070039057,0.00044331697,0.0005530776,0.00055464293,0.00096172653,0.08503534,0.08821434,0.005618179,0.0037514975,0.80638504],"study_design_scores_gemma":[0.00020032392,0.0009997207,0.008935238,0.00010680468,0.000548804,0.001122958,0.0002832968,0.87345546,0.07660519,0.018390108,0.019147255,0.00020485691],"about_ca_topic_score_codex":0.0044297953,"about_ca_topic_score_gemma":0.007069145,"teacher_disagreement_score":0.012701679,"about_ca_system_score_codex":0.00089716667,"about_ca_system_score_gemma":0.0018408044,"threshold_uncertainty_score":0.06717366},"labels":[],"label_agreement":null},{"id":"W1974219706","doi":"10.1109/icmlc.2009.5212797","title":"A special parser for learning English composition - Error analysis &amp;#x00026; learners' model for ILTS","year":2009,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Northern British Columbia","funders":"","keywords":"Computer science; Parsing; Natural language processing; Sentence; Artificial intelligence; Grammar; Table (database); Word (group theory); Matching (statistics); Speech recognition; Embedding; Linguistics; Database","score_opus":0.027443446178123096,"score_gpt":0.30557928556890085,"score_spread":0.27813583939077774,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1974219706","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0038648206,0.000027926855,0.9711283,0.0001013323,0.00004723057,0.00012260776,0.0008569199,0.020758685,0.003092252],"genre_scores_gemma":[0.16355413,0.00017508441,0.8000004,0.00027045803,0.00008993718,0.0005283351,0.005767648,0.006696312,0.022917574],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988507,0.00022623483,0.00011592615,0.000368903,0.00035758445,0.00008064082],"domain_scores_gemma":[0.99819404,0.00064242573,0.000078589044,0.00045904962,0.00057863386,0.000047300684],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014769855,0.0010106541,0.0007187181,0.0007870481,0.0005472515,0.0016876664,0.0020632364,0.0011929917,0.016058512],"category_scores_gemma":[0.004254785,0.0008641494,0.0011390401,0.00076915004,0.0007589025,0.0032483046,0.0013711778,0.0016431662,0.0065290905],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000653556,0.00038514897,0.0070458227,0.0008046025,0.00014000454,0.0018921916,0.0031294152,0.10871202,0.061082255,0.19603068,0.07781688,0.54230744],"study_design_scores_gemma":[0.0000784887,0.00012334954,0.0011029036,0.00006782728,0.000103236016,0.0008483857,0.00018568756,0.77899164,0.057122454,0.046747785,0.114545,0.00008324728],"about_ca_topic_score_codex":0.0037395784,"about_ca_topic_score_gemma":0.0040708664,"teacher_disagreement_score":0.016058512,"about_ca_system_score_codex":0.00078957656,"about_ca_system_score_gemma":0.001797527,"threshold_uncertainty_score":0.05372107},"labels":[],"label_agreement":null},{"id":"W1974289591","doi":"10.5539/ass.v5n3p25","title":"A Cognitive Model for Recognition of Genre","year":2009,"lang":"en","type":"article","venue":"Asian Social Science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Schema (genetic algorithms); Computer science; Cognition; Cognitive model; Image schema; Thread (computing); Top-down and bottom-up design; Natural language processing; Cognitive science; Artificial intelligence; Linguistics; Psychology; Cognitive linguistics; Information retrieval","score_opus":0.030277856193356927,"score_gpt":0.32755835779717185,"score_spread":0.2972805016038149,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1974289591","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11225355,0.0007723827,0.6782061,0.0045326157,0.00021443222,0.00036033953,0.00033534862,0.00051936763,0.202806],"genre_scores_gemma":[0.8106465,0.0005005803,0.17705964,0.0008388393,0.00013860452,0.00033466806,0.00048775904,0.000068700116,0.009924771],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9981609,0.0005787722,0.00011350212,0.00042044217,0.00055561546,0.00017080596],"domain_scores_gemma":[0.99699056,0.0013189674,0.00035525791,0.0004955323,0.0006076616,0.0002319463],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029000319,0.00058386306,0.00033350603,0.0031851153,0.0009930374,0.0045326296,0.0019549653,0.0013713686,0.005447965],"category_scores_gemma":[0.0077311364,0.0003527398,0.0018777649,0.0016158189,0.00469794,0.009779604,0.0021163407,0.0016708578,0.0009371043],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006247409,0.00009443134,0.0062772767,0.00021584056,0.00006149713,0.0002642842,0.012122914,0.0037397062,0.0032845046,0.90707296,0.00231274,0.06449139],"study_design_scores_gemma":[0.00004244762,0.0001615619,0.009760253,0.0001205054,0.000065508895,0.00071251893,0.0040995367,0.046233073,0.0014290684,0.91551626,0.021789383,0.00006987206],"about_ca_topic_score_codex":0.003591237,"about_ca_topic_score_gemma":0.0016153044,"teacher_disagreement_score":0.005447965,"about_ca_system_score_codex":0.0018410421,"about_ca_system_score_gemma":0.0011405385,"threshold_uncertainty_score":0.018225193},"labels":[],"label_agreement":null},{"id":"W1975376422","doi":"10.7202/003531ar","title":"Peut-on vérifier automatiquement la cohérence terminologique?","year":2002,"lang":"fr","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Philosophy; Humanities","score_opus":0.09016857769049691,"score_gpt":0.31564085966472366,"score_spread":0.22547228197422675,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1975376422","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05249233,0.00090702256,0.898968,0.0059098396,0.00070580747,0.0006039648,0.0011969274,0.019787755,0.019428499],"genre_scores_gemma":[0.46178073,0.00075906934,0.50982684,0.002642521,0.00032134948,0.00068774645,0.0025585727,0.0060085393,0.0154146375],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9729101,0.007628399,0.002058806,0.0053909244,0.010380938,0.0016309756],"domain_scores_gemma":[0.90570796,0.034911785,0.0040085847,0.031793684,0.022468815,0.0011091579],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.018419972,0.0009804827,0.001534411,0.00203454,0.0028307794,0.010155004,0.0033586528,0.0038140155,0.011277946],"category_scores_gemma":[0.10639068,0.0013997329,0.0014633943,0.0018175332,0.007071244,0.02142901,0.006994364,0.0055295634,0.007636083],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018159742,0.0002761277,0.02016124,0.002037486,0.00024593784,0.0010935621,0.011874514,0.00860587,0.055357344,0.39385894,0.02844556,0.4762275],"study_design_scores_gemma":[0.00039604967,0.00059925776,0.0065432233,0.0011420059,0.00039351475,0.0015242232,0.0070584537,0.12835905,0.1710833,0.43467262,0.2477878,0.00044047108],"about_ca_topic_score_codex":0.013568798,"about_ca_topic_score_gemma":0.009831425,"teacher_disagreement_score":0.018419972,"about_ca_system_score_codex":0.0034407435,"about_ca_system_score_gemma":0.009825379,"threshold_uncertainty_score":0.09741527},"labels":[],"label_agreement":null},{"id":"W1975598382","doi":"10.7202/003237ar","title":"La traduction automatique au service de l’utilisateur monolingue","year":2002,"lang":"fr","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Philosophy","score_opus":0.07198759953430947,"score_gpt":0.29796178867371187,"score_spread":0.2259741891394024,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1975598382","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.041972835,0.0009481254,0.90474564,0.0012598049,0.00034016304,0.0002722653,0.00033829478,0.015490693,0.03463213],"genre_scores_gemma":[0.38159966,0.0012813417,0.5108191,0.00080132176,0.00025543288,0.00042417579,0.0013295134,0.006443872,0.09704571],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9924561,0.0024238445,0.0005516187,0.0019470863,0.0021884094,0.00043299914],"domain_scores_gemma":[0.98789525,0.0047900686,0.00050622283,0.0040525002,0.0024961133,0.00025979307],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00505542,0.0015480954,0.000993883,0.0016531432,0.0020011703,0.0070097144,0.0023326355,0.0020409294,0.017932897],"category_scores_gemma":[0.015076139,0.0010944495,0.0014363143,0.0021636793,0.003757132,0.008209616,0.0047922395,0.0029442431,0.010798102],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00087251473,0.00020876963,0.003070792,0.0012937312,0.00014268047,0.0010346158,0.016125102,0.008782772,0.082577564,0.18105237,0.014016774,0.6908223],"study_design_scores_gemma":[0.00016894446,0.0005043639,0.00326447,0.0005839592,0.00025004975,0.0017881809,0.0041457885,0.079457134,0.25743598,0.09112515,0.5610162,0.00025974886],"about_ca_topic_score_codex":0.0069784354,"about_ca_topic_score_gemma":0.0038812326,"teacher_disagreement_score":0.017932897,"about_ca_system_score_codex":0.0022913076,"about_ca_system_score_gemma":0.0028623191,"threshold_uncertainty_score":0.05999148},"labels":[],"label_agreement":null},{"id":"W1976323603","doi":"10.1109/icde.2012.126","title":"AutoDict: Automated Dictionary Discovery","year":2012,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Set (abstract data type); Information retrieval; Process (computing); Information extraction; Knowledge extraction; Quality (philosophy); Natural language processing; Product (mathematics); Artificial intelligence; Data mining; Programming language","score_opus":0.00998374185412475,"score_gpt":0.2696992889318416,"score_spread":0.2597155470777168,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1976323603","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018301917,0.0013467508,0.71475327,0.0008247367,0.0004422078,0.0006403328,0.031969268,0.21932991,0.012391632],"genre_scores_gemma":[0.04637539,0.0005988564,0.86987525,0.00045210653,0.0000969047,0.0003301936,0.06852125,0.004918736,0.0088312635],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99747664,0.00037457646,0.00026446846,0.0007448288,0.0010107664,0.0001288243],"domain_scores_gemma":[0.9934848,0.002280269,0.00044148954,0.0023512885,0.0012057993,0.00023641264],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020730107,0.0013543308,0.00122816,0.006381378,0.0011254449,0.0029845012,0.0027720751,0.001279581,0.016552571],"category_scores_gemma":[0.011393848,0.0011308364,0.0015751371,0.0047475714,0.0009956761,0.0056338464,0.004337892,0.0017829151,0.011652495],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00046789364,0.00020040937,0.00487621,0.0012360393,0.00025332498,0.00083467516,0.0006751034,0.0050369627,0.017622238,0.021416293,0.32612565,0.62125516],"study_design_scores_gemma":[0.00027868233,0.00024843222,0.0037574375,0.0002942974,0.00011776803,0.0043107034,0.0006953093,0.3780728,0.07377192,0.049192283,0.4889902,0.00027025855],"about_ca_topic_score_codex":0.0030482248,"about_ca_topic_score_gemma":0.005845694,"teacher_disagreement_score":0.016552571,"about_ca_system_score_codex":0.0011326515,"about_ca_system_score_gemma":0.0024446086,"threshold_uncertainty_score":0.055373847},"labels":[],"label_agreement":null},{"id":"W1976965564","doi":"10.7202/1008337ar","title":"A Cognitive Model of Chinese Word Segmentation for Machine Translation","year":2012,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Machine translation; Computer science; Natural language processing; Text segmentation; Sentence; Artificial intelligence; Word (group theory); Segmentation; Bottleneck; Language translation; Translation (biology); Perspective (graphical); Linguistics","score_opus":0.054497600281491344,"score_gpt":0.3269029220178653,"score_spread":0.27240532173637394,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1976965564","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13417901,0.0010607217,0.7502899,0.0047550285,0.00020868948,0.00028973597,0.00041630643,0.0006710293,0.108129516],"genre_scores_gemma":[0.8632947,0.00042362354,0.12693976,0.00037111196,0.00008023135,0.0003366093,0.00029030294,0.00006023756,0.008203241],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99940205,0.00021272458,0.00003211048,0.00016341047,0.0001115327,0.00007811458],"domain_scores_gemma":[0.9988656,0.0006052808,0.00011724173,0.00011424232,0.00019937882,0.00009821891],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011264998,0.0007839159,0.00035272466,0.0015458276,0.0013539703,0.0027804398,0.0014466682,0.001004657,0.0054886704],"category_scores_gemma":[0.0031421832,0.0003600346,0.0016426853,0.0015154002,0.0029920307,0.004730863,0.0013462533,0.0011404641,0.0007813199],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011453996,0.00013951767,0.0030751345,0.00028124813,0.00010602078,0.0006212793,0.0082524605,0.027045527,0.0038155108,0.8923429,0.0029713968,0.06123442],"study_design_scores_gemma":[0.00006611732,0.00009093786,0.0028388624,0.0000624977,0.00009863221,0.0003301975,0.001274229,0.24098197,0.0015929944,0.7444385,0.008160586,0.000064455395],"about_ca_topic_score_codex":0.012770225,"about_ca_topic_score_gemma":0.008522944,"teacher_disagreement_score":0.012770225,"about_ca_system_score_codex":0.0025981697,"about_ca_system_score_gemma":0.0027954904,"threshold_uncertainty_score":0.025391817},"labels":[],"label_agreement":null},{"id":"W1978086734","doi":"10.1145/1030397.1030409","title":"Personal glossaries on the WWW","year":2004,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Glossary; Suite; Computer science; Meaning (existential); Reading (process); World Wide Web; Medical information; Information retrieval; Linguistics; Psychology; History","score_opus":0.01494007116185374,"score_gpt":0.2536164161781165,"score_spread":0.23867634501626273,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1978086734","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02324255,0.0043073427,0.17113815,0.00874543,0.0022695612,0.0010242718,0.22799368,0.050465405,0.5108136],"genre_scores_gemma":[0.10419948,0.0072716963,0.244137,0.0036814811,0.0010096024,0.0012718963,0.4048284,0.03776787,0.1958326],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9981457,0.00049569213,0.00046279925,0.00023094029,0.0005855497,0.00007941602],"domain_scores_gemma":[0.98758,0.0040620663,0.0007510686,0.0050077997,0.0022093898,0.00038957738],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028671683,0.0011035622,0.00082496944,0.0069499863,0.0017075383,0.005100595,0.0015917843,0.0012246356,0.10248313],"category_scores_gemma":[0.016654572,0.0010285543,0.0005148393,0.011229816,0.0008112526,0.010227844,0.0037993381,0.002504204,0.05030736],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022030859,0.00010945306,0.0014467244,0.0012852259,0.000043542237,0.00055275497,0.0030604587,0.0014465322,0.0069128326,0.07347466,0.7136349,0.1978126],"study_design_scores_gemma":[0.000022001335,0.000018410277,0.0011225422,0.0004182921,0.000020705667,0.0002993861,0.00049062475,0.0020337503,0.002778434,0.015092137,0.97764754,0.00005601428],"about_ca_topic_score_codex":0.0059571057,"about_ca_topic_score_gemma":0.011476207,"teacher_disagreement_score":0.10248313,"about_ca_system_score_codex":0.00090744114,"about_ca_system_score_gemma":0.0013490026,"threshold_uncertainty_score":0.3428402},"labels":[],"label_agreement":null},{"id":"W1978620866","doi":"10.1162/coli.2008.34.2.145","title":"Semantic Role Labeling: An Introduction to the Special Issue","year":2008,"lang":"en","type":"article","venue":"Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":278,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canada Research Chairs; Artificial Intelligence in Medicine (Canada); University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Task (project management); Field (mathematics); Computational linguistics; Identification (biology); Computational model; Semantic role labeling; Artificial intelligence; Natural language processing; Data science; Key (lock); Semantics (computer science)","score_opus":0.01392567728635698,"score_gpt":0.2788210233112759,"score_spread":0.26489534602491893,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1978620866","genre_codex":"editorial","genre_gemma":"editorial","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"editorial","genre_consensus":"editorial","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0015131038,0.11140237,0.120702736,0.108761564,0.5509495,0.0004921966,0.0014128347,0.001771003,0.10299473],"genre_scores_gemma":[0.010033012,0.12064396,0.057750743,0.033648025,0.63264966,0.00077625725,0.0036780136,0.0025266209,0.13829371],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9981395,0.0005108117,0.00027643688,0.00042751266,0.0005210843,0.00012472138],"domain_scores_gemma":[0.99050444,0.0058010416,0.00046382844,0.00069468585,0.0017388973,0.00079713826],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0038175555,0.0016167276,0.0018372805,0.0046720714,0.0027734505,0.0077905147,0.0019890477,0.0041521583,0.032827985],"category_scores_gemma":[0.010496428,0.0010864943,0.001887413,0.0035957475,0.0031203895,0.013293463,0.0034175601,0.010047957,0.02249422],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000027138987,0.000067600915,0.0001974107,0.00031080144,0.00001582576,0.00009813481,0.0001373242,0.00025420735,0.0004874906,0.022256937,0.8864222,0.08972489],"study_design_scores_gemma":[0.0000039369434,0.000024682844,0.0002786157,0.00026180365,0.000009207161,0.00035328724,0.0000995176,0.00039428822,0.00012993996,0.014843602,0.98358303,0.000018096058],"about_ca_topic_score_codex":0.0010502697,"about_ca_topic_score_gemma":0.001758201,"teacher_disagreement_score":0.032827985,"about_ca_system_score_codex":0.0020239237,"about_ca_system_score_gemma":0.002021655,"threshold_uncertainty_score":0.109820485},"labels":[],"label_agreement":null},{"id":"W1978799108","doi":"10.1007/s10032-012-0184-x","title":"A new approach for recognizing handwritten mathematics using relational grammars and fuzzy sets","year":2012,"lang":"en","type":"article","venue":"International Journal on Document Analysis and Recognition (IJDAR)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":96,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Parsing; Computer science; Artificial intelligence; Set (abstract data type); Natural language processing; Rule-based machine translation; Ambiguity; S-attributed grammar; Similarity (geometry); Fuzzy logic; Interpretation (philosophy); Programming language; Image (mathematics)","score_opus":0.043528409929608096,"score_gpt":0.3182998786613794,"score_spread":0.2747714687317713,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1978799108","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011230424,0.0002687554,0.98498154,0.00008887056,0.00006266083,0.0000777025,0.00015012184,0.0012715932,0.001868326],"genre_scores_gemma":[0.08782488,0.00033652308,0.908295,0.00007261696,0.000036378337,0.00008001833,0.0003037978,0.000110178626,0.0029406939],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99945265,0.000043844357,0.000047930753,0.00020223642,0.00022043056,0.000032991844],"domain_scores_gemma":[0.9996226,0.00012163773,0.000028618884,0.0000775695,0.00012481085,0.00002487896],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00040011227,0.0005687526,0.00077875843,0.0016561358,0.00062824937,0.0015613404,0.0013275777,0.00072604907,0.0029798665],"category_scores_gemma":[0.0010277682,0.0003929576,0.0012468932,0.0010989221,0.00080459984,0.0025073444,0.001115883,0.0010681481,0.0011368283],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012179937,0.00015498194,0.0019151809,0.00035841187,0.00017157425,0.0005866801,0.00093711785,0.024241382,0.11678146,0.093489125,0.0033227867,0.7579195],"study_design_scores_gemma":[0.000047225585,0.0001780705,0.0026842472,0.00014252706,0.00027284346,0.0012459039,0.0006653479,0.78648585,0.05954703,0.11160574,0.036987726,0.00013741312],"about_ca_topic_score_codex":0.005790562,"about_ca_topic_score_gemma":0.007975962,"teacher_disagreement_score":0.005790562,"about_ca_system_score_codex":0.00061070785,"about_ca_system_score_gemma":0.0010528148,"threshold_uncertainty_score":0.01151371},"labels":[],"label_agreement":null},{"id":"W1979123099","doi":"10.3115/1067807.1067830","title":"A general feature space for automatic verb classification","year":2003,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":32,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Feature (linguistics); Verb; Space (punctuation); Artificial intelligence; Scope (computer science); Natural language processing; Feature selection; Task (project management); Feature vector; Pattern recognition (psychology); Linguistics","score_opus":0.01809449139492498,"score_gpt":0.28583006036715114,"score_spread":0.26773556897222617,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1979123099","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005064489,0.00013158986,0.9922672,0.000057935413,0.000019924835,0.00008284339,0.00038036882,0.0015335276,0.00046206792],"genre_scores_gemma":[0.1323121,0.00024073293,0.8614499,0.000115523646,0.00007575489,0.0006460855,0.0030148106,0.0002759977,0.0018690566],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985304,0.00036343097,0.00013851866,0.0003672965,0.00048315362,0.00011720809],"domain_scores_gemma":[0.9984445,0.0005409159,0.00009919279,0.00034376112,0.00050891796,0.000062781604],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016748036,0.0011721026,0.001308433,0.0025278847,0.0006522919,0.0015908517,0.0017492998,0.0011634928,0.004553772],"category_scores_gemma":[0.0041101207,0.00038342446,0.0017612962,0.0024515816,0.00084985804,0.0027310343,0.0016566009,0.0014477769,0.0019367565],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002635027,0.00025473806,0.002076395,0.00034399386,0.00012894573,0.00021199365,0.00017082957,0.05946397,0.029697722,0.028996632,0.009311248,0.86908],"study_design_scores_gemma":[0.00004285798,0.00030163862,0.0023367237,0.000065428816,0.000051145398,0.00043332874,0.000078516525,0.8942071,0.015618086,0.070390806,0.016390845,0.000083440944],"about_ca_topic_score_codex":0.002141558,"about_ca_topic_score_gemma":0.0017003865,"teacher_disagreement_score":0.004553772,"about_ca_system_score_codex":0.000571737,"about_ca_system_score_gemma":0.00093298394,"threshold_uncertainty_score":0.015233874},"labels":[],"label_agreement":null},{"id":"W1979398480","doi":"10.1177/0265532214560799","title":"A prototype of a receptive lexical test for a polysynthetic heritage language: The case of Inuttitut in Labrador","year":2014,"lang":"en","type":"article","venue":"Language Testing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"The Scarborough Hospital; University of Toronto; Carleton University","funders":"","keywords":"Heritage language; Linguistics; Vocabulary; Psychology; Test (biology); Comprehension; Noun; Language proficiency; First language; Natural language processing; Computer science; Mathematics education","score_opus":0.016575572520713226,"score_gpt":0.2924684211711534,"score_spread":0.27589284865044017,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1979398480","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98232377,0.00013882674,0.0070403437,0.00051316933,0.000043658405,0.00033247203,0.00018174927,0.00023070442,0.0091953],"genre_scores_gemma":[0.94700766,0.00022151716,0.04564352,0.00035056076,0.000018414092,0.00038046678,0.00026329883,0.000089331246,0.006025229],"study_design_codex":"observational","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99907374,0.0004099875,0.000086578126,0.00013665878,0.00010735896,0.00018569983],"domain_scores_gemma":[0.99891627,0.00037801373,0.0001078801,0.00015487739,0.00023612377,0.0002068478],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017867974,0.000557159,0.00033201723,0.0009145301,0.0012376657,0.0009160139,0.0013687074,0.0008671948,0.0030452062],"category_scores_gemma":[0.0028208995,0.0002486055,0.0002425395,0.00048984477,0.0013491941,0.0008349089,0.0011144966,0.0008388538,0.00118376],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00095408683,0.004352829,0.4051856,0.0005323555,0.00003182844,0.05747019,0.105017856,0.0011488503,0.0899139,0.004290867,0.005081096,0.32602054],"study_design_scores_gemma":[0.00027050602,0.008038996,0.5610843,0.0006933736,0.00011558376,0.0978569,0.14854996,0.0071944254,0.076599844,0.004925144,0.09430142,0.00036961696],"about_ca_topic_score_codex":0.017569259,"about_ca_topic_score_gemma":0.05738471,"teacher_disagreement_score":0.98243076,"about_ca_system_score_codex":0.0007900199,"about_ca_system_score_gemma":0.001944705,"threshold_uncertainty_score":0.034933984},"labels":[],"label_agreement":null},{"id":"W1981981315","doi":"10.7202/014330ar","title":"La traduction multilingue des noms propres dans PROLEX","year":2006,"lang":"fr","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Philosophy","score_opus":0.03490857126676234,"score_gpt":0.2837965736166859,"score_spread":0.2488880023499236,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1981981315","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06686541,0.0008893223,0.8894722,0.0012624634,0.00043532727,0.00016603172,0.0013213776,0.0040076547,0.03558024],"genre_scores_gemma":[0.43938997,0.0013526176,0.5142183,0.0005120911,0.00020584621,0.00020120708,0.0016449309,0.0018455624,0.04062947],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9981008,0.00075004966,0.00022037608,0.00038220806,0.00040318817,0.00014332394],"domain_scores_gemma":[0.9973054,0.0013521648,0.00024165338,0.00052299025,0.00050752884,0.00007015245],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017359316,0.0007688737,0.00051901763,0.0007884886,0.001492087,0.00389382,0.00083567144,0.000666675,0.00688879],"category_scores_gemma":[0.0046789926,0.0006547044,0.0007624748,0.00088084716,0.0022133547,0.004089015,0.0014202691,0.00188379,0.003106683],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00073472894,0.00007659958,0.0040717325,0.0014595174,0.00008590339,0.0026142625,0.016643366,0.006581494,0.05745486,0.6112798,0.011671218,0.28732663],"study_design_scores_gemma":[0.00008948191,0.00032913362,0.0045478195,0.000750002,0.00016232977,0.0038912212,0.006936058,0.028856855,0.1266101,0.12086484,0.70675474,0.00020745372],"about_ca_topic_score_codex":0.0036496818,"about_ca_topic_score_gemma":0.003812713,"teacher_disagreement_score":0.00688879,"about_ca_system_score_codex":0.0010007585,"about_ca_system_score_gemma":0.0013018114,"threshold_uncertainty_score":0.023045242},"labels":[],"label_agreement":null},{"id":"W1982365650","doi":"10.3115/1706543.1706551","title":"Computing word similarity and identifying cognates with pair hidden Markov models","year":2005,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":45,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Hidden Markov model; Computer science; Word (group theory); Similarity (geometry); Focus (optics); Artificial intelligence; Markov chain; Markov model; Identification (biology); Maximum-entropy Markov model; Natural language processing; Variation (astronomy); Speech recognition; Variable-order Markov model; Pattern recognition (psychology); Machine learning; Mathematics","score_opus":0.020024557003524613,"score_gpt":0.26569999853543025,"score_spread":0.24567544153190563,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1982365650","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10114816,0.00038578385,0.8839238,0.00019940015,0.00008880029,0.00020707026,0.00070669834,0.010968416,0.0023718206],"genre_scores_gemma":[0.4206362,0.00015264597,0.5752787,0.00013693083,0.00006359742,0.00022400283,0.0018880251,0.00043147913,0.0011883394],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99748886,0.0009032932,0.0001977657,0.0007752315,0.00048707245,0.00014783783],"domain_scores_gemma":[0.99547726,0.0026888559,0.00043996962,0.0006799394,0.0005131595,0.000200814],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022416124,0.00091529236,0.0012518436,0.003221982,0.001010027,0.0017386938,0.0013507631,0.0016594091,0.0036876462],"category_scores_gemma":[0.012346015,0.0005664296,0.0011171551,0.0019220667,0.0007237055,0.004952183,0.0024401546,0.0011649883,0.0020067922],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0025849962,0.00080549,0.03273079,0.0007673424,0.00058342185,0.0008233525,0.0014007647,0.038173947,0.05792073,0.024997959,0.007697479,0.8315138],"study_design_scores_gemma":[0.00008848837,0.00034947635,0.004600266,0.000053037227,0.00017350545,0.0006162976,0.00030485325,0.9016777,0.038218684,0.04958495,0.004219357,0.00011339918],"about_ca_topic_score_codex":0.0023608934,"about_ca_topic_score_gemma":0.0029597203,"teacher_disagreement_score":0.0036876462,"about_ca_system_score_codex":0.0005565364,"about_ca_system_score_gemma":0.0010171628,"threshold_uncertainty_score":0.012336433},"labels":[],"label_agreement":null},{"id":"W1982660309","doi":"10.1016/j.ijar.2003.10.007","title":"Hyperrelations in version space","year":2003,"lang":"en","type":"article","venue":"International Journal of Approximate Reasoning","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":29,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Brock University","funders":"","keywords":"Space (punctuation); Computer science","score_opus":0.008912890490830418,"score_gpt":0.27260089415045335,"score_spread":0.2636880036596229,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1982660309","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09924824,0.0036215899,0.85861266,0.002927482,0.00029453452,0.0000785725,0.0008231144,0.0010297362,0.03336406],"genre_scores_gemma":[0.80288774,0.0017369924,0.17642191,0.0004944459,0.00041185715,0.000118635326,0.0010516308,0.0005757564,0.016301008],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9917217,0.003337848,0.0008014094,0.0013581124,0.002263118,0.00051779754],"domain_scores_gemma":[0.9742011,0.015790086,0.0015631963,0.006009226,0.0016869056,0.00074951886],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005865561,0.00071804516,0.0016817044,0.0039131506,0.0021056742,0.010294164,0.0021039885,0.0018081572,0.015179935],"category_scores_gemma":[0.032304972,0.0013027168,0.0016299954,0.0071063032,0.0041315425,0.030053472,0.0048369397,0.0042611836,0.0015924551],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010791825,0.0000290107,0.00062438066,0.0000490512,0.00002867914,0.00007477902,0.00026421458,0.004884897,0.00018938148,0.96859443,0.0015645705,0.02358862],"study_design_scores_gemma":[0.0000112143825,0.000010874682,0.00007081728,0.000015032443,0.00001781609,0.000072927985,0.00004349404,0.014586022,0.0002032157,0.9827293,0.0022308875,0.000008270229],"about_ca_topic_score_codex":0.0013583519,"about_ca_topic_score_gemma":0.0010037558,"teacher_disagreement_score":0.015179935,"about_ca_system_score_codex":0.0017264327,"about_ca_system_score_gemma":0.00093044323,"threshold_uncertainty_score":0.050781906},"labels":[],"label_agreement":null},{"id":"W1982816048","doi":"10.1023/b:coat.0000010117.98933.a0","title":"Trans Type: Development-Evaluation Cycles to Boost Translator's Productivity","year":2002,"lang":"en","type":"article","venue":"Machine Translation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":65,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Machine translation; Productivity; Context (archaeology); Translation (biology); Process (computing); Set (abstract data type); Interface (matter); Artificial intelligence; Evaluation of machine translation; Natural language processing; Rule-based machine translation; Software engineering; Machine learning; Programming language; Example-based machine translation; Machine translation software usability","score_opus":0.04654180184988662,"score_gpt":0.2987353309047347,"score_spread":0.2521935290548481,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1982816048","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10103918,0.0010610017,0.7478031,0.0027240305,0.0011406374,0.001031753,0.0012428623,0.124855414,0.019102046],"genre_scores_gemma":[0.32696995,0.0004554015,0.61438537,0.0013211038,0.0004101532,0.00079571875,0.002489549,0.030235574,0.022937177],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9893287,0.005963135,0.0009776169,0.0011840571,0.0020671522,0.00047939157],"domain_scores_gemma":[0.91537046,0.040610597,0.0032516995,0.017760726,0.020699162,0.0023072849],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010186775,0.0020191167,0.0012999034,0.0024226569,0.0013170682,0.0034305349,0.002484897,0.0023208505,0.02442676],"category_scores_gemma":[0.05710834,0.0018933635,0.0011092472,0.002194168,0.0008757556,0.007805006,0.0035315074,0.003389133,0.0141649535],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0026804123,0.0011951519,0.007819721,0.0012526829,0.00013363683,0.0005800332,0.0014847644,0.0066471333,0.0824338,0.009130913,0.05903308,0.82760864],"study_design_scores_gemma":[0.0012349515,0.0027504019,0.009457477,0.00044949012,0.0007932395,0.003325038,0.0012901999,0.2936416,0.5118759,0.0344213,0.14028007,0.00048034376],"about_ca_topic_score_codex":0.0013075679,"about_ca_topic_score_gemma":0.0027817106,"teacher_disagreement_score":0.02442676,"about_ca_system_score_codex":0.0010441467,"about_ca_system_score_gemma":0.0046602585,"threshold_uncertainty_score":0.08171564},"labels":[],"label_agreement":null},{"id":"W1982840093","doi":"10.3115/1072228.1072374","title":"Crosslinguistic transfer in automatic verb classification","year":2002,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Verb; Classifier (UML); Machine translation; Modal verb","score_opus":0.03082254637244029,"score_gpt":0.2767052643928321,"score_spread":0.24588271802039183,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1982840093","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.88153744,0.0005830894,0.10553343,0.00035003937,0.00012156063,0.00017591049,0.00038721762,0.0013045465,0.010006699],"genre_scores_gemma":[0.9736484,0.0001255573,0.023313524,0.00014154075,0.00005021236,0.00014605155,0.0009006329,0.00021391977,0.0014599899],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99160695,0.005488954,0.00040779787,0.0013201484,0.0008217719,0.00035433538],"domain_scores_gemma":[0.97291946,0.019337106,0.00116477,0.0037157475,0.0025456306,0.0003171548],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009790563,0.00065162306,0.0006428687,0.0017910029,0.000991003,0.0017820996,0.0009016229,0.00075869536,0.0022261431],"category_scores_gemma":[0.038273405,0.00033079134,0.00058038067,0.001772892,0.001207838,0.0037791529,0.0027422246,0.0014262506,0.0011946122],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014151024,0.0015527392,0.07105091,0.00044388595,0.00033704136,0.0009565925,0.0047060084,0.031746343,0.06807986,0.0066999854,0.0036982314,0.80931336],"study_design_scores_gemma":[0.00025461693,0.0016745821,0.1864668,0.00017682389,0.0004191371,0.0021916607,0.0048549837,0.582801,0.15065987,0.051565684,0.018612564,0.00032232975],"about_ca_topic_score_codex":0.002454772,"about_ca_topic_score_gemma":0.0029846448,"teacher_disagreement_score":0.009790563,"about_ca_system_score_codex":0.0008233375,"about_ca_system_score_gemma":0.0007416769,"threshold_uncertainty_score":0.051778078},"labels":[],"label_agreement":null},{"id":"W1982855231","doi":"10.3166/ria.18.679-707","title":"Traitement des erreurs d'accord. Une analyse syntagmatique pour la détection et une analyse multicritère pour la correction","year":2004,"lang":"fr","type":"article","venue":"Revue d intelligence artificielle","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Philosophy; Political science; Physics","score_opus":0.03985155260779159,"score_gpt":0.33156295255865287,"score_spread":0.29171139995086126,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1982855231","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08605294,0.0013932962,0.89726734,0.00085096166,0.0006827712,0.0001637142,0.00038455467,0.005269656,0.007934768],"genre_scores_gemma":[0.5572204,0.0006611903,0.4242639,0.000291928,0.00020058504,0.00015374771,0.00062072405,0.0013197797,0.015267692],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99427235,0.0011907087,0.0003977432,0.0011648244,0.0027827816,0.00019165626],"domain_scores_gemma":[0.9857964,0.0058142347,0.001811742,0.0023730993,0.003997605,0.00020680335],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003760294,0.0013099605,0.0009781278,0.0022440432,0.0009184504,0.0018232013,0.0008171943,0.0012580209,0.0054234182],"category_scores_gemma":[0.02403977,0.00047101156,0.0009030711,0.001261771,0.00094165554,0.0024138005,0.0010607111,0.0016369029,0.002578682],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006672328,0.00012511849,0.011734185,0.00088425254,0.00024081758,0.0013085204,0.0015263179,0.014270845,0.16740733,0.020367105,0.0063724383,0.7750958],"study_design_scores_gemma":[0.00011420291,0.0007993848,0.0272414,0.00034156445,0.00036251993,0.0064319447,0.002075583,0.37891287,0.46209797,0.04842181,0.072863944,0.0003367635],"about_ca_topic_score_codex":0.0017248022,"about_ca_topic_score_gemma":0.0015264126,"teacher_disagreement_score":0.0054234182,"about_ca_system_score_codex":0.0006109967,"about_ca_system_score_gemma":0.00084724964,"threshold_uncertainty_score":0.019886553},"labels":[],"label_agreement":null},{"id":"W1984610102","doi":"10.1075/term.15.1.04bow","title":"Better integration for better preparation","year":2009,"lang":"en","type":"article","venue":"Terminology International Journal of Theoretical and Applied Issues in Specialized Communication","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"University of Ottawa","keywords":"Terminology; Computer science; Translation (biology); Data science; Linguistics; Philosophy; Chemistry","score_opus":0.01205162221153747,"score_gpt":0.3347169038539985,"score_spread":0.322665281642461,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1984610102","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.041405804,0.0026185713,0.51663935,0.052880216,0.004813856,0.0013303694,0.00086484867,0.01250745,0.36693946],"genre_scores_gemma":[0.1892815,0.0018540252,0.5123744,0.010870022,0.0013260586,0.0011400193,0.0027840578,0.007306697,0.27306324],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9908785,0.0029718073,0.0008369759,0.0017605042,0.002337259,0.0012149235],"domain_scores_gemma":[0.974878,0.002817551,0.0013196255,0.012219128,0.0060289516,0.002736699],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01162595,0.0010876708,0.0009875064,0.0023423142,0.0036219677,0.010573955,0.00262003,0.003732834,0.11323023],"category_scores_gemma":[0.030106226,0.0008864112,0.0013257217,0.0029972915,0.0020531272,0.020675803,0.014233206,0.0049945056,0.053779315],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00031623823,0.0008522561,0.0045543583,0.0005619038,0.000045972803,0.0007958064,0.012535431,0.0021127625,0.011925425,0.2346543,0.09978089,0.63186467],"study_design_scores_gemma":[0.00005226532,0.00017585127,0.0034100066,0.0003558807,0.000036084075,0.00050006964,0.0030663158,0.0022067549,0.004962576,0.05934368,0.9258224,0.0000681799],"about_ca_topic_score_codex":0.0028829721,"about_ca_topic_score_gemma":0.0032370877,"teacher_disagreement_score":0.11323023,"about_ca_system_score_codex":0.0032104722,"about_ca_system_score_gemma":0.010111241,"threshold_uncertainty_score":0.37879282},"labels":[],"label_agreement":null},{"id":"W1986353013","doi":"10.1145/1328964.1328989","title":"Toponym resolution in text","year":2007,"lang":"en","type":"article","venue":"ACM SIGIR Forum","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":158,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Referent; Ambiguity; Task (project management); Coreference; Information retrieval; Focus (optics); Natural language processing; Toponymy; Geographic coordinate system; Artificial intelligence; Proper noun; Ambiguity resolution; Resolution (logic); Linguistics; Global Positioning System; Cartography; Geography","score_opus":0.0114038817565472,"score_gpt":0.281587358509245,"score_spread":0.27018347675269777,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1986353013","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0130672725,0.010078619,0.9037261,0.0052766227,0.0023104548,0.0009920496,0.027391233,0.01526307,0.021894602],"genre_scores_gemma":[0.09740822,0.004640647,0.84655464,0.0018865044,0.0011705967,0.0005172047,0.035015132,0.001396771,0.0114103705],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9881995,0.0026559534,0.001378983,0.0038768642,0.0033879315,0.0005008569],"domain_scores_gemma":[0.988694,0.005312815,0.0015263718,0.002438113,0.0016997098,0.00032891685],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037286421,0.0023127152,0.002096635,0.015168776,0.0023658562,0.0067215087,0.0026539657,0.002842357,0.015965061],"category_scores_gemma":[0.023303254,0.00089894695,0.0023805175,0.016293986,0.0020893875,0.01607594,0.009458659,0.0032242585,0.016529828],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029096924,0.00015658159,0.002281785,0.0033467351,0.00022896074,0.0023533527,0.0029723546,0.0101688495,0.0121739255,0.09958512,0.15263794,0.7138034],"study_design_scores_gemma":[0.000066548404,0.00012587583,0.002807285,0.0007714731,0.00019301695,0.0042250445,0.0037837406,0.09958175,0.018011292,0.41809055,0.45207092,0.0002724645],"about_ca_topic_score_codex":0.0030975242,"about_ca_topic_score_gemma":0.003258569,"teacher_disagreement_score":0.015965061,"about_ca_system_score_codex":0.0018533169,"about_ca_system_score_gemma":0.0020673168,"threshold_uncertainty_score":0.053408504},"labels":[],"label_agreement":null},{"id":"W1986634806","doi":"10.1111/j.1467-9612.2004.00005.x","title":"A Computational Algebraic Approach to English Grammar","year":2004,"lang":"en","type":"article","venue":"Syntax","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":31,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Mathematical proof; Simple (philosophy); Iterated function; Context (archaeology); Type (biology); Algebraic number; Computer science; Feature (linguistics); Grammar; String (physics); Parsing; Mathematics; Element (criminal law); Rule-based machine translation; Algebra over a field; Pure mathematics; Linguistics; Artificial intelligence","score_opus":0.008785415566468675,"score_gpt":0.23879661650283318,"score_spread":0.2300112009363645,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1986634806","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.052856743,0.0016997526,0.7820258,0.009383512,0.00026858086,0.0000853862,0.0006043346,0.00044531483,0.1526306],"genre_scores_gemma":[0.7176134,0.002101039,0.25267404,0.0008696559,0.00047453083,0.00023123829,0.0006207989,0.00017166874,0.025243618],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99932826,0.0003082524,0.00004590055,0.00009353208,0.00017651582,0.000047584632],"domain_scores_gemma":[0.9988017,0.000768541,0.00007709298,0.00013133806,0.00017036832,0.0000510352],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009324356,0.00031145682,0.00038549522,0.00126457,0.0012178766,0.003254944,0.0012291776,0.0004314571,0.008777852],"category_scores_gemma":[0.0034011137,0.00029915964,0.0009976679,0.0010493351,0.0042092954,0.005428552,0.0019974473,0.0014345298,0.00064976513],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000035458672,0.000003511659,0.00005127361,0.000018436842,0.00000290513,0.000026739624,0.00011845608,0.001712625,0.000059569364,0.9953414,0.00040253118,0.002259065],"study_design_scores_gemma":[0.000007702233,0.000004309516,0.00005810451,0.00001502271,0.0000052489063,0.000038920003,0.00008104225,0.011961722,0.00011584017,0.97845924,0.009247482,0.000005360989],"about_ca_topic_score_codex":0.0030641742,"about_ca_topic_score_gemma":0.004539369,"teacher_disagreement_score":0.008777852,"about_ca_system_score_codex":0.002462358,"about_ca_system_score_gemma":0.0010399459,"threshold_uncertainty_score":0.029364884},"labels":[],"label_agreement":null},{"id":"W1986648269","doi":"10.3115/1690359.1690360","title":"Exploration of the LTAG-spinal formalism and Treebank for semantic role labeling","year":2009,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Treebank; Computer science; Artificial intelligence; Natural language processing; Formalism (music); Annotation; Phrase; Tree-adjoining grammar; Grammar; Linguistics; Rule-based machine translation","score_opus":0.018299527297429646,"score_gpt":0.2785890764424094,"score_spread":0.2602895491449797,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1986648269","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0040866137,0.00024737057,0.98351353,0.0007801324,0.00013975357,0.00020014078,0.0015661147,0.0052456097,0.0042206775],"genre_scores_gemma":[0.08036103,0.00047002351,0.91003776,0.0004450184,0.000111516085,0.00045906013,0.004110529,0.0016576052,0.0023475194],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99643815,0.0015548541,0.00037636622,0.0007093712,0.0007715765,0.00014975634],"domain_scores_gemma":[0.9933106,0.002564052,0.0004048864,0.0023164013,0.0011407257,0.00026336036],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007446822,0.0011652719,0.0011390819,0.0036452194,0.0018324406,0.0039605787,0.002973596,0.0014486879,0.008270499],"category_scores_gemma":[0.011024292,0.0012268669,0.0015258278,0.0048833475,0.002279375,0.011902347,0.0038263607,0.003918866,0.0034293972],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022558999,0.00019875409,0.0010154783,0.0004469684,0.00006138031,0.0003562151,0.001458009,0.0121129425,0.008008247,0.7692608,0.02756113,0.17929444],"study_design_scores_gemma":[0.000085961634,0.00008088698,0.0005484739,0.00020871103,0.0000838547,0.0003481918,0.0003397364,0.2524737,0.0079898015,0.63275254,0.10499359,0.00009451155],"about_ca_topic_score_codex":0.010701226,"about_ca_topic_score_gemma":0.017631946,"teacher_disagreement_score":0.010701226,"about_ca_system_score_codex":0.0029548542,"about_ca_system_score_gemma":0.0056728283,"threshold_uncertainty_score":0.039383054},"labels":[],"label_agreement":null},{"id":"W1986993264","doi":"10.1016/s0898-1221(03)90034-4","title":"A graph unification machine for NL parsing","year":2003,"lang":"en","type":"article","venue":"Computers & Mathematics with Applications","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Unification; Parsing; Graph; Mathematics; Bottom-up parsing; Top-down parsing; Theoretical computer science; Top-down parsing language; Programming language; Computer science; Natural language processing; Discrete mathematics","score_opus":0.0149863356026497,"score_gpt":0.26678771673575263,"score_spread":0.2518013811331029,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1986993264","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008146612,0.0007826958,0.93325955,0.0009913365,0.00028454402,0.00018370213,0.0022272235,0.048509967,0.005614432],"genre_scores_gemma":[0.08626753,0.0006138795,0.8927724,0.00046212832,0.00019000922,0.0002581182,0.007477211,0.003847522,0.008111275],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992331,0.00019785737,0.00007586091,0.0002918019,0.00013529437,0.00006606642],"domain_scores_gemma":[0.99849164,0.00078428834,0.00006140262,0.00042989964,0.00017988845,0.000052924945],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010876645,0.0007916654,0.0012464065,0.0025789654,0.0017202204,0.0023282918,0.0018700814,0.0017925731,0.016332282],"category_scores_gemma":[0.0042000096,0.0010310747,0.0018050788,0.0031607405,0.0011824026,0.006555958,0.0031989596,0.0028266152,0.0071923337],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00033499088,0.00019832113,0.0011590829,0.00053597946,0.00015330069,0.0003776276,0.00052196864,0.01743959,0.013284199,0.1268701,0.061584033,0.777541],"study_design_scores_gemma":[0.00015697273,0.00009805693,0.0010325826,0.00019886406,0.0002467188,0.00034529786,0.00021906885,0.4574812,0.021603841,0.43548894,0.08302767,0.00010076921],"about_ca_topic_score_codex":0.004733953,"about_ca_topic_score_gemma":0.007876155,"teacher_disagreement_score":0.016332282,"about_ca_system_score_codex":0.0009161269,"about_ca_system_score_gemma":0.0016155211,"threshold_uncertainty_score":0.054636896},"labels":[],"label_agreement":null},{"id":"W1987756646","doi":"10.1007/s10994-009-5151-5","title":"A co-classification approach to learning from multilingual corpora","year":2009,"lang":"en","type":"article","venue":"Machine Learning","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":40,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Computer science; Categorization; Artificial intelligence; Boosting (machine learning); Regularization (linguistics); Natural language processing; Benchmarking; Consistency (knowledge bases); Text categorization; Machine learning","score_opus":0.02645328759020278,"score_gpt":0.2968939080229041,"score_spread":0.27044062043270134,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1987756646","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0080079,0.0017579424,0.979044,0.0007989653,0.00052453764,0.00030842036,0.0010738048,0.0032097616,0.005274647],"genre_scores_gemma":[0.13062905,0.0014225852,0.83797413,0.0006938911,0.00087115937,0.0011307513,0.008809052,0.0009248661,0.017544473],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98922604,0.003962787,0.0010183395,0.0026877576,0.0025744932,0.00053063664],"domain_scores_gemma":[0.97193897,0.013269096,0.0008657626,0.0062672673,0.0071126027,0.00054638175],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008427502,0.0016713088,0.0023265514,0.011195211,0.0034681763,0.0052818297,0.0046775984,0.0032678468,0.0062615057],"category_scores_gemma":[0.02449719,0.00089439884,0.002709803,0.013345106,0.0015800376,0.008500831,0.0056246673,0.004708035,0.00577611],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000357585,0.000728891,0.004864538,0.00045603243,0.0005209795,0.00041917549,0.0005801214,0.009186373,0.0065309834,0.021546522,0.030353462,0.9244554],"study_design_scores_gemma":[0.00011871718,0.00028605916,0.0045628957,0.00029870583,0.0007444952,0.0013311305,0.0009359181,0.78175753,0.022866204,0.1128578,0.07400073,0.00023975014],"about_ca_topic_score_codex":0.0070183952,"about_ca_topic_score_gemma":0.01559406,"teacher_disagreement_score":0.011195211,"about_ca_system_score_codex":0.0013975591,"about_ca_system_score_gemma":0.0033589723,"threshold_uncertainty_score":0.044569433},"labels":[],"label_agreement":null},{"id":"W1988016002","doi":"10.1093/jos/ffn007","title":"Syntax and Semantics of It-Clefts: A Tree Adjoining Grammar Analysis","year":2008,"lang":"en","type":"article","venue":"Journal of Semantics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":26,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Computer science; Syntax; Programming language; Grammar; Semantics (computer science); Linguistics; Tree (set theory); Natural language processing; Abstract syntax tree; Philosophy; Mathematics; Combinatorics","score_opus":0.017241096335214583,"score_gpt":0.25963297037319355,"score_spread":0.24239187403797896,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1988016002","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.044218305,0.00029947222,0.93438745,0.0009844557,0.00007121579,0.000113567905,0.00026121153,0.00054615573,0.01911822],"genre_scores_gemma":[0.69501,0.00043688592,0.29798362,0.0004225577,0.000099801204,0.00016736716,0.00050364767,0.0005448457,0.004831215],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99884015,0.00045668666,0.000119331125,0.00020129494,0.0002819512,0.000100606165],"domain_scores_gemma":[0.9983437,0.0007379955,0.00015295814,0.0003163381,0.00037220583,0.00007683058],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016013827,0.00045993156,0.0005317658,0.0012394759,0.0011641479,0.0023465278,0.0010283364,0.00095599564,0.0034767452],"category_scores_gemma":[0.0027478153,0.0003787675,0.0014271553,0.0013778651,0.004216595,0.005344014,0.0018409289,0.001639588,0.0005818096],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000007034123,0.0000062695794,0.00030002045,0.000030707855,0.000005897684,0.00018370392,0.001260874,0.0013106916,0.0010935833,0.98917574,0.00041368202,0.0062118145],"study_design_scores_gemma":[0.0000072830044,0.000017826073,0.00034235627,0.000032720447,0.000022336015,0.00027489665,0.0005374652,0.02109999,0.0017122999,0.96563184,0.010306768,0.000014275676],"about_ca_topic_score_codex":0.002763104,"about_ca_topic_score_gemma":0.0015240497,"teacher_disagreement_score":0.0034767452,"about_ca_system_score_codex":0.0012227689,"about_ca_system_score_gemma":0.0013707294,"threshold_uncertainty_score":0.011630833},"labels":[],"label_agreement":null},{"id":"W1988085378","doi":"10.1162/ling.2006.37.2.275","title":"In Defense of a Quantificational Account of Definite DPs","year":2006,"lang":"en","type":"article","venue":"Linguistic Inquiry","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Positive-definite matrix; Mathematics; Witness; Set (abstract data type); Pure mathematics; Linguistics; Computer science; Eigenvalues and eigenvectors; Physics; Philosophy","score_opus":0.027151873642400993,"score_gpt":0.3011279714391424,"score_spread":0.27397609779674137,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1988085378","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.042708732,0.0030266545,0.48986077,0.061734304,0.0005533951,0.000068245616,0.00041654697,0.0007007124,0.40093064],"genre_scores_gemma":[0.89680827,0.0016882615,0.058522776,0.0064080437,0.001204065,0.0001661535,0.00024289725,0.00030956924,0.03465003],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9980077,0.0006343549,0.00009645422,0.000474167,0.00061785587,0.00016951845],"domain_scores_gemma":[0.99636024,0.001857445,0.00027467363,0.0007668537,0.0005292605,0.00021146759],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004411491,0.0005839162,0.0005259339,0.0019245974,0.002077559,0.005462408,0.002879754,0.0030318003,0.009949257],"category_scores_gemma":[0.0065353694,0.0005256835,0.0009319764,0.0012588006,0.013860565,0.018757397,0.005865091,0.0045368033,0.001006578],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000022378288,0.000001788072,0.000040080755,0.000009645424,0.0000012307441,0.000017440456,0.00024307484,0.00009366724,0.000069197886,0.99800175,0.0004039783,0.0011159157],"study_design_scores_gemma":[0.000006128522,0.00000420672,0.00016914784,0.000019189778,0.0000034121747,0.00007729609,0.00022294006,0.0012781954,0.00017928758,0.98309934,0.014934764,0.0000058771393],"about_ca_topic_score_codex":0.0020396449,"about_ca_topic_score_gemma":0.0018568955,"teacher_disagreement_score":0.009949257,"about_ca_system_score_codex":0.0020674889,"about_ca_system_score_gemma":0.0013045892,"threshold_uncertainty_score":0.03328359},"labels":[],"label_agreement":null},{"id":"W1988220974","doi":"10.3115/1654679.1654682","title":"Challenges in evaluating summaries of short stories","year":2006,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Sentence; Quality (philosophy); Natural language processing; Information retrieval; Artificial intelligence; Data science; Epistemology","score_opus":0.07351163251175052,"score_gpt":0.34697714454452666,"score_spread":0.27346551203277614,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1988220974","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.57536215,0.035102073,0.35186884,0.0054065315,0.0011342848,0.0022437389,0.005158075,0.010228117,0.013496136],"genre_scores_gemma":[0.6643196,0.004749167,0.31799766,0.0005119473,0.00075009064,0.0008570103,0.007822552,0.0008996951,0.002092182],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.95899487,0.024019627,0.006057788,0.0030677293,0.0075151036,0.00034490062],"domain_scores_gemma":[0.81343144,0.14801142,0.008529831,0.007470276,0.020599091,0.0019579087],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02631622,0.0011520792,0.0017569236,0.0029473954,0.0010977194,0.0044810446,0.0018600665,0.0021720757,0.0020935107],"category_scores_gemma":[0.1351774,0.00047083446,0.0007124932,0.002460411,0.00077304745,0.006402799,0.0017549294,0.0011694144,0.0014019335],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021629331,0.000681394,0.009402839,0.006886185,0.00077731785,0.0006921521,0.007969795,0.017110104,0.038380817,0.0036486152,0.015722938,0.8965649],"study_design_scores_gemma":[0.0022364028,0.018800395,0.0994631,0.0034568417,0.0023271968,0.0069503123,0.030176533,0.31727517,0.20227326,0.054501526,0.2611459,0.0013933292],"about_ca_topic_score_codex":0.0013217428,"about_ca_topic_score_gemma":0.0017698245,"teacher_disagreement_score":0.02631622,"about_ca_system_score_codex":0.000879011,"about_ca_system_score_gemma":0.0007728913,"threshold_uncertainty_score":0.13917518},"labels":[],"label_agreement":null},{"id":"W1989001126","doi":"10.3115/1220355.1220502","title":"Fast computation of lexical affinity models","year":2004,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Coordenação de Aperfeiçoamento de Pessoal de Nível Superior","keywords":"Computer science; Computation; Scalability; Terabyte; Independence (probability theory); Artificial intelligence; Parametric statistics; Similarity (geometry); Focus (optics); Natural language processing; Theoretical computer science; Algorithm; Mathematics; Database","score_opus":0.02419785949263349,"score_gpt":0.28466108550980973,"score_spread":0.26046322601717625,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1989001126","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009211029,0.00013413474,0.9873854,0.00008927397,0.00002386211,0.000038100938,0.00018266802,0.0019853706,0.0009501084],"genre_scores_gemma":[0.24781089,0.0002767197,0.745422,0.00013738756,0.00008513967,0.0003353063,0.0015904676,0.00081670616,0.0035253929],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99800545,0.000482282,0.00013025233,0.00040968522,0.0007788868,0.00019337061],"domain_scores_gemma":[0.99562055,0.0024018993,0.00024058894,0.0008820199,0.00069363107,0.00016129858],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019070231,0.0010013059,0.0015753118,0.0024409646,0.0011387445,0.0027698325,0.0024593123,0.0018046983,0.0067196693],"category_scores_gemma":[0.01627198,0.0012175136,0.0012729391,0.0025461693,0.0007292145,0.0061759274,0.0040879142,0.0024814361,0.0035547023],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00045625033,0.00020185295,0.00439289,0.00040703814,0.00021530438,0.00043476524,0.00059782824,0.26501852,0.016650088,0.123216815,0.01265514,0.57575357],"study_design_scores_gemma":[0.00002126378,0.000025134528,0.0003545301,0.000010027387,0.000015831873,0.00008941439,0.000052365074,0.8916491,0.0027969142,0.10181878,0.0031431874,0.000023507719],"about_ca_topic_score_codex":0.005358063,"about_ca_topic_score_gemma":0.008709565,"teacher_disagreement_score":0.0067196693,"about_ca_system_score_codex":0.0011772537,"about_ca_system_score_gemma":0.001893989,"threshold_uncertainty_score":0.022479594},"labels":[],"label_agreement":null},{"id":"W1989867740","doi":"10.3115/1118693.1118705","title":"Extensions to HMM-based statistical word alignment models","year":2002,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":88,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Bigram; Hidden Markov model; Computer science; Word (group theory); Viterbi algorithm; Speech recognition; Artificial intelligence; Machine translation; Natural language processing; Translation (biology); Pattern recognition (psychology); Mathematics","score_opus":0.03464454925532364,"score_gpt":0.27975693241025507,"score_spread":0.24511238315493142,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1989867740","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0057336534,0.0004287214,0.9854033,0.00015432545,0.00014059305,0.00008589938,0.00065931823,0.0052791904,0.0021149998],"genre_scores_gemma":[0.15380639,0.0013119729,0.82774746,0.00027686046,0.00023440775,0.00046436267,0.00418101,0.0015962423,0.0103814015],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99855286,0.0005142239,0.00013798897,0.00032014106,0.00039938337,0.000075394404],"domain_scores_gemma":[0.9957703,0.0024230976,0.000185314,0.00082741206,0.0007136031,0.00008028372],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026230412,0.0008029427,0.00091682805,0.000775965,0.00054579397,0.0011474557,0.0018071618,0.00079111446,0.007946076],"category_scores_gemma":[0.009596001,0.00089113606,0.001059738,0.0013120533,0.00043577992,0.0028469707,0.0011491649,0.0020624588,0.0070154653],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00061955675,0.0002488788,0.003258343,0.00069615664,0.0004734881,0.00028539918,0.0007682518,0.3900378,0.017373877,0.04488555,0.014466403,0.5268863],"study_design_scores_gemma":[0.00004501801,0.00008151234,0.0007492266,0.00006186584,0.000104569735,0.00013726766,0.000048220507,0.91793674,0.006201328,0.043183412,0.031388514,0.00006232763],"about_ca_topic_score_codex":0.007005798,"about_ca_topic_score_gemma":0.013818277,"teacher_disagreement_score":0.007946076,"about_ca_system_score_codex":0.00068302476,"about_ca_system_score_gemma":0.0013892449,"threshold_uncertainty_score":0.026582241},"labels":[],"label_agreement":null},{"id":"W1990591837","doi":"10.1353/lan.2014.0076","title":"How to investigate linguistic diversity: Lessons from the Pacific Northwest","year":2014,"lang":"en","type":"article","venue":"Language","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":57,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Falsifiability; Syntax; Linguistics; Diversity (politics); Lexical diversity; Typology; Rule-based machine translation; Semantics (computer science); Historical linguistics; Computer science; Scale (ratio); Test (biology); Relative clause; Linguistic universal; Natural language processing; Artificial intelligence; Theoretical linguistics; Sociology; Geography; Epistemology; Philosophy; Archaeology; Ecology; Vocabulary","score_opus":0.015365291500231373,"score_gpt":0.2480672163176107,"score_spread":0.23270192481737934,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1990591837","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8449806,0.005749624,0.047061723,0.036591835,0.00007795156,0.00025962674,0.0002588539,0.000083005434,0.064936675],"genre_scores_gemma":[0.9531337,0.0029323027,0.039392088,0.0012662052,0.000036518337,0.00023241917,0.00020770518,0.00007335306,0.0027256687],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9923459,0.005846518,0.00019369725,0.00048771608,0.00075715536,0.0003690012],"domain_scores_gemma":[0.9692573,0.024377763,0.0010911757,0.0022165566,0.002303136,0.0007540535],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.014164527,0.00045663072,0.0007244739,0.0031667463,0.008373997,0.004416541,0.0023183862,0.0017893227,0.002281676],"category_scores_gemma":[0.02590174,0.00037926127,0.0003477214,0.0055227634,0.016461603,0.013984244,0.007539113,0.0038546456,0.00020213504],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020143774,0.0002584039,0.07271968,0.0007086844,0.00009057517,0.004413503,0.54268336,0.0029906346,0.002208378,0.13356884,0.0051164245,0.2350401],"study_design_scores_gemma":[0.000053501448,0.00009923893,0.043629467,0.0014381728,0.00004990608,0.0018663033,0.6102156,0.004630827,0.0027881889,0.24842958,0.08669904,0.00010011929],"about_ca_topic_score_codex":0.08786052,"about_ca_topic_score_gemma":0.13974547,"teacher_disagreement_score":0.08786052,"about_ca_system_score_codex":0.0048890947,"about_ca_system_score_gemma":0.0058902507,"threshold_uncertainty_score":0.17469823},"labels":[],"label_agreement":null},{"id":"W1990825561","doi":"10.1007/s10579-008-9072-x","title":"Disambiguation of partial cognates","year":2008,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Cognate; Computer science; Natural language processing; Meaning (existential); Context (archaeology); Artificial intelligence; Word (group theory); Word-sense disambiguation; Machine translation; Linguistics; Psychology; Biology","score_opus":0.024028377565879798,"score_gpt":0.3017247175024333,"score_spread":0.2776963399365535,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1990825561","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6149464,0.00576524,0.31771243,0.0014550672,0.0009577194,0.0006895741,0.0048763915,0.012086456,0.041510735],"genre_scores_gemma":[0.8598629,0.0007685002,0.12552702,0.0002803565,0.00018952382,0.0001334095,0.005487444,0.0012066874,0.0065439995],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9912361,0.0036727358,0.0010356903,0.0017416734,0.0016690759,0.0006447629],"domain_scores_gemma":[0.98567617,0.0072215837,0.0004057398,0.002683434,0.003360565,0.0006524947],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0061160387,0.0014597352,0.002303403,0.005921434,0.003163668,0.005768831,0.0020420318,0.0018114211,0.008848233],"category_scores_gemma":[0.021465054,0.00069514266,0.0015359846,0.0024295684,0.0017877342,0.010530666,0.005356506,0.0017131853,0.0030974403],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004896315,0.00060613337,0.019059723,0.0016155575,0.00066002546,0.0016622196,0.0024975943,0.017934924,0.050799996,0.06224063,0.021059701,0.8169671],"study_design_scores_gemma":[0.0005376345,0.0009701094,0.015281488,0.0005511554,0.0018818927,0.0050650216,0.0053892117,0.4316488,0.25222245,0.21089761,0.07501578,0.0005388839],"about_ca_topic_score_codex":0.0045514232,"about_ca_topic_score_gemma":0.0056298412,"teacher_disagreement_score":0.008848233,"about_ca_system_score_codex":0.0010176038,"about_ca_system_score_gemma":0.0028566916,"threshold_uncertainty_score":0.032345057},"labels":[],"label_agreement":null},{"id":"W1991311321","doi":"10.3758/bf03210725","title":"Meaning resolution processes for words: A parallel independent model","year":2000,"lang":"en","type":"article","venue":"Psychonomic Bulletin & Review","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":62,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Killam Trusts","keywords":"Ambiguity; Context (archaeology); Meaning (existential); Modularity (biology); Ambiguity resolution; Psychology; Resolution (logic); Linguistics; Focus (optics); Computer science; Cognitive science; Artificial intelligence","score_opus":0.02491332796593182,"score_gpt":0.2962225518819005,"score_spread":0.2713092239159687,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1991311321","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0819267,0.002866466,0.851767,0.0053264652,0.00044308315,0.00040140547,0.0006921366,0.0010612932,0.055515498],"genre_scores_gemma":[0.8044033,0.0029617494,0.16310598,0.0009488527,0.0005930372,0.0005481607,0.0011143152,0.0004426045,0.02588201],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99824333,0.00038527182,0.00009336275,0.0006484622,0.0004673523,0.00016220308],"domain_scores_gemma":[0.99288166,0.0039274334,0.00055791636,0.0014479851,0.0008902768,0.00029463368],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00269963,0.0012312756,0.0019309913,0.0019499924,0.0015550449,0.0066456413,0.0053785797,0.003456523,0.015475309],"category_scores_gemma":[0.014902359,0.0014618733,0.0042134146,0.001714041,0.0032177656,0.017350769,0.003168374,0.0046095415,0.004335693],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00044244697,0.0003741515,0.0022641534,0.00054082135,0.00036828427,0.000505385,0.0018424701,0.02319491,0.007140048,0.8711535,0.0047655506,0.087408245],"study_design_scores_gemma":[0.000097132084,0.000074368145,0.0011043533,0.000030410944,0.0001230866,0.00026199667,0.00012111409,0.111417174,0.0011493838,0.88313955,0.0024408759,0.00004051142],"about_ca_topic_score_codex":0.0034579984,"about_ca_topic_score_gemma":0.0015854412,"teacher_disagreement_score":0.015475309,"about_ca_system_score_codex":0.0014589792,"about_ca_system_score_gemma":0.001694738,"threshold_uncertainty_score":0.05177003},"labels":[],"label_agreement":null},{"id":"W1991383860","doi":"10.3115/1072228.1072355","title":"Location normalization for information extraction","year":2002,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":93,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"Air Force Research Laboratory","keywords":"Computer science; Normalization (sociology); Ambiguity; Grammar; Visualization; Natural language processing; Information retrieval; Context (archaeology); Artificial intelligence; Geography; Linguistics","score_opus":0.014278964483148528,"score_gpt":0.26418631843228985,"score_spread":0.24990735394914132,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1991383860","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007965174,0.00063469604,0.9643453,0.00048080162,0.00015215224,0.00030410674,0.0034399035,0.017398419,0.005279521],"genre_scores_gemma":[0.07919998,0.00068457064,0.90393645,0.0003038158,0.000105016654,0.00046444687,0.009677348,0.0020817406,0.003546613],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9969716,0.0009657299,0.00042463533,0.0007506107,0.00078530435,0.000101989186],"domain_scores_gemma":[0.99629503,0.001299551,0.0003020512,0.001106533,0.0009411627,0.000055768134],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002250168,0.0012597062,0.0013218619,0.004441252,0.001423416,0.0025916314,0.0013761013,0.000889498,0.009554722],"category_scores_gemma":[0.00783268,0.00072640064,0.0012482634,0.006110003,0.0008515327,0.004175785,0.0023486386,0.0012570262,0.008108558],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000119227676,0.000081245125,0.0026855958,0.0011877151,0.00016526482,0.0006372319,0.0008069314,0.0062552835,0.04775597,0.03815553,0.042422228,0.8597277],"study_design_scores_gemma":[0.00008190747,0.00014107456,0.010102135,0.0005765894,0.00032420843,0.0029964296,0.00217307,0.21708547,0.12360062,0.22404331,0.41859517,0.0002799557],"about_ca_topic_score_codex":0.0025288009,"about_ca_topic_score_gemma":0.004041905,"teacher_disagreement_score":0.009554722,"about_ca_system_score_codex":0.0012033931,"about_ca_system_score_gemma":0.0019972096,"threshold_uncertainty_score":0.031963706},"labels":[],"label_agreement":null},{"id":"W1991522508","doi":"10.1007/s10590-008-9036-3","title":"Semi-supervised model adaptation for statistical machine translation","year":2007,"lang":"en","type":"article","venue":"Machine Translation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":49,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University; National Research Council Canada","funders":"","keywords":"Machine translation; Computer science; Translation (biology); Artificial intelligence; Adaptation (eye); Computational linguistics; Natural language processing; Machine learning; Statistical analysis; Statistical model; Statistics; Mathematics; Psychology","score_opus":0.03837286427808482,"score_gpt":0.3087072662634866,"score_spread":0.27033440198540176,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1991522508","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006437753,0.00047188334,0.98792225,0.00014826703,0.000119990415,0.000057531357,0.00020725407,0.003953703,0.0006812831],"genre_scores_gemma":[0.37064767,0.0007788675,0.6145535,0.0004329,0.00028865342,0.00071622775,0.0040956163,0.0016313577,0.0068551493],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99645823,0.002217133,0.00019514315,0.0005478479,0.00044535298,0.00013626448],"domain_scores_gemma":[0.9922271,0.004414101,0.00039194638,0.0016664059,0.0011878703,0.00011258102],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029819102,0.0011883305,0.0017570348,0.0010666877,0.0008568317,0.0013103118,0.0020853062,0.0016352035,0.0031166538],"category_scores_gemma":[0.010683542,0.0010628257,0.0015789294,0.0015252045,0.0007327557,0.0021928144,0.0018198211,0.0029090731,0.0033698091],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000558919,0.00033244822,0.0009888941,0.00037573226,0.0005720557,0.00030146656,0.00020685022,0.2845377,0.018035725,0.009712054,0.01840716,0.665971],"study_design_scores_gemma":[0.000014114233,0.00003371117,0.0001775496,0.000007801083,0.00002636368,0.000061385486,0.00001254275,0.9856309,0.0040161265,0.008662061,0.0013421801,0.000015290072],"about_ca_topic_score_codex":0.0028815798,"about_ca_topic_score_gemma":0.0046369247,"teacher_disagreement_score":0.0031166538,"about_ca_system_score_codex":0.00060539646,"about_ca_system_score_gemma":0.0013803153,"threshold_uncertainty_score":0.015770078},"labels":[],"label_agreement":null},{"id":"W1992249591","doi":"10.3166/isi.7.1-2.95-123","title":"Quand la réponse se trouve dans un grand corpus","year":2002,"lang":"fr","type":"article","venue":"Ingénierie des systèmes d information","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Sentence; Question answering; Computer science; Natural language processing; Artificial intelligence; Domain (mathematical analysis); Linguistics; Information retrieval; Humanities; Philosophy; Mathematics","score_opus":0.026586198288331135,"score_gpt":0.2448324174171462,"score_spread":0.21824621912881506,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1992249591","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3403815,0.016605187,0.23805526,0.061032664,0.015145532,0.00082759367,0.06500065,0.012563285,0.25038838],"genre_scores_gemma":[0.49309748,0.0074243094,0.12185357,0.0050720223,0.004823942,0.00094107626,0.10400348,0.010776828,0.25200728],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9954131,0.0016881023,0.00033072924,0.0009705685,0.0013714818,0.00022598276],"domain_scores_gemma":[0.97548246,0.013948302,0.00056136923,0.0028919748,0.006558294,0.0005575464],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005106803,0.00094924646,0.0009378466,0.004847208,0.0036012444,0.0050218767,0.0011090492,0.002636086,0.040989637],"category_scores_gemma":[0.039301176,0.0006731687,0.0008776309,0.004160375,0.0020386484,0.008119531,0.0025926495,0.002877386,0.013077313],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001903248,0.00041134213,0.011773214,0.0029883,0.00040912308,0.003433852,0.010419261,0.004129499,0.07438411,0.06773642,0.42873695,0.3936747],"study_design_scores_gemma":[0.00015380619,0.0002809404,0.021817988,0.00054866995,0.00034003583,0.003033014,0.005905332,0.01794318,0.04357532,0.02584342,0.8803486,0.00020970569],"about_ca_topic_score_codex":0.01739687,"about_ca_topic_score_gemma":0.026958464,"teacher_disagreement_score":0.040989637,"about_ca_system_score_codex":0.0020066088,"about_ca_system_score_gemma":0.0030815466,"threshold_uncertainty_score":0.137124},"labels":[],"label_agreement":null},{"id":"W1992520036","doi":"10.5539/ijel.v3n1p31","title":"Linguistic Divergences in English to Bengali Translation","year":2013,"lang":"en","type":"article","venue":"International Journal of English Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"Division of Human Resource Development; Ministry of Education, India; University of Calcutta","keywords":"Bengali; Computer science; Machine translation; Natural language processing; Example-based machine translation; Artificial intelligence; Divergence (linguistics); Linguistics; Evaluation of machine translation; Machine translation software usability; Translation (biology); Context (archaeology); Rule-based machine translation; Dynamic and formal equivalence; Process (computing); Lexicon; Task (project management); History; Programming language","score_opus":0.01620308796378544,"score_gpt":0.29078770657003167,"score_spread":0.2745846186062462,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1992520036","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.90839076,0.006945514,0.024292938,0.0051915273,0.0001913433,0.000073813884,0.00022994263,0.00017835513,0.054505765],"genre_scores_gemma":[0.99342936,0.00059617916,0.0038018478,0.00016376066,0.000026462341,0.000020133906,0.00008701759,0.000051878535,0.0018234273],"study_design_codex":"qualitative","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9949315,0.0020356583,0.0006095571,0.0004914314,0.0015483076,0.0003836091],"domain_scores_gemma":[0.99371815,0.0031028718,0.0008388731,0.00050049234,0.0016691919,0.00017042074],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037239105,0.00025515695,0.000606438,0.0019189735,0.0028384863,0.0031001419,0.0007145627,0.00055374764,0.0014708142],"category_scores_gemma":[0.019238744,0.00024901258,0.00030005313,0.0034533825,0.0034877537,0.002565939,0.0032932376,0.0013438265,0.00038946778],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00073295034,0.000103535065,0.10051696,0.0012820325,0.00013772989,0.009458452,0.319001,0.0049921684,0.025788892,0.23162143,0.004754306,0.30161056],"study_design_scores_gemma":[0.00006398814,0.00023736451,0.28909796,0.0009201879,0.0002558347,0.011640051,0.1968287,0.01779775,0.0201973,0.29735866,0.16523822,0.00036401307],"about_ca_topic_score_codex":0.013319996,"about_ca_topic_score_gemma":0.012973451,"teacher_disagreement_score":0.013319996,"about_ca_system_score_codex":0.0037674622,"about_ca_system_score_gemma":0.0015492013,"threshold_uncertainty_score":0.027334988},"labels":[],"label_agreement":null},{"id":"W1993301141","doi":"10.1145/1645953.1646205","title":"Real-word spelling correction using Google web 1Tn-gram data set","year":2009,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":40,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"n-gram; Computer science; Spelling; Word (group theory); Set (abstract data type); Precision and recall; String (physics); Edit distance; Natural language processing; Error detection and correction; Matching (statistics); Fraction (chemistry); Data set; Artificial intelligence; String searching algorithm; String metric; Information retrieval; Pattern matching; Algorithm; Mathematics; Language model; Statistics; Programming language","score_opus":0.05859422628538493,"score_gpt":0.3450595133657176,"score_spread":0.2864652870803327,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1993301141","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6296915,0.0035801178,0.19220664,0.0008384446,0.00176338,0.0012422871,0.07311647,0.0892106,0.008350637],"genre_scores_gemma":[0.45990327,0.000800662,0.41966388,0.00025514566,0.00025578908,0.00072859676,0.11060555,0.0023622673,0.005424781],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9925242,0.0012903048,0.0012857494,0.0013493912,0.0032530562,0.00029729994],"domain_scores_gemma":[0.9775123,0.0050495802,0.0028259596,0.0047991807,0.00939655,0.00041644683],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002167397,0.0026136395,0.0016124124,0.009845657,0.0013009884,0.0011691332,0.0017410989,0.0015759831,0.0021955653],"category_scores_gemma":[0.020414013,0.00037129605,0.0011263647,0.0077713714,0.0006019082,0.0019725647,0.0014046129,0.0012643954,0.005063293],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014238337,0.00057652534,0.039672684,0.0024025303,0.00052500085,0.0018227686,0.000772185,0.012308031,0.08787284,0.0007915749,0.0520136,0.7998185],"study_design_scores_gemma":[0.0002742348,0.0011277075,0.103930555,0.00032321474,0.00049969956,0.004227017,0.001303689,0.31497523,0.48994413,0.0034938203,0.07906217,0.0008384533],"about_ca_topic_score_codex":0.01606338,"about_ca_topic_score_gemma":0.025817566,"teacher_disagreement_score":0.01606338,"about_ca_system_score_codex":0.0008810582,"about_ca_system_score_gemma":0.0021705963,"threshold_uncertainty_score":0.031939805},"labels":[],"label_agreement":null},{"id":"W1993925072","doi":"10.1177/0261927x03260811","title":"Bilingualism","year":2004,"lang":"en","type":"article","venue":"Journal of Language and Social Psychology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"St. Francis Xavier University","funders":"","keywords":"Neuroscience of multilingualism; Psychology; Linguistics; Sociology; Philosophy","score_opus":0.018012377317899803,"score_gpt":0.3628646012471192,"score_spread":0.34485222392921944,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1993925072","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.31224763,0.0023362197,0.0023035144,0.007644231,0.00059809757,0.00004458696,0.0008386307,0.00019549439,0.67379165],"genre_scores_gemma":[0.9556977,0.00060103316,0.0005876032,0.0007850782,0.00011274537,0.000018739569,0.00026976294,0.00007034249,0.041856926],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.9992848,0.00022348881,0.000035513887,0.00017221589,0.00013108506,0.00015294735],"domain_scores_gemma":[0.99854726,0.00034923185,0.00015251835,0.00024354193,0.00034281777,0.00036468636],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00060404406,0.00019681104,0.0003271801,0.0009412075,0.0015909283,0.0014245897,0.00022636227,0.00047169003,0.044647492],"category_scores_gemma":[0.0023719866,0.0001334251,0.00013199037,0.0005776792,0.0010041527,0.0014162848,0.001332522,0.0007653291,0.004138344],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011966294,0.00034162035,0.0407915,0.00046242838,0.00007528381,0.0034926825,0.012949227,0.0001194664,0.01881031,0.62402904,0.05765338,0.24007842],"study_design_scores_gemma":[0.00025907642,0.00039311472,0.12301381,0.00035044618,0.00014632913,0.017409932,0.014325934,0.0006735911,0.011865438,0.22270899,0.6087819,0.000071485214],"about_ca_topic_score_codex":0.0039877263,"about_ca_topic_score_gemma":0.006710157,"teacher_disagreement_score":0.044647492,"about_ca_system_score_codex":0.0009325836,"about_ca_system_score_gemma":0.0010880076,"threshold_uncertainty_score":0.14936072},"labels":[],"label_agreement":null},{"id":"W1994173598","doi":"10.1142/s0218213001000660","title":"AN ASSUMPTIVE LOGIC PROGRAMMING METHODOLOGY FOR PARSING","year":2001,"lang":"en","type":"article","venue":"International Journal of Artificial Intelligence Tools","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Datalog; Parsing; Parsing expression grammar; Programming language; L-attributed grammar; Rule-based machine translation; Phrase structure grammar; Logic programming; Artificial intelligence; Natural language processing; S-attributed grammar; Parser combinator; Continuation; Context-free grammar","score_opus":0.2408640637981287,"score_gpt":0.45734304363168704,"score_spread":0.21647897983355835,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1994173598","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00036433144,0.000020033356,0.9973653,0.0001561665,0.000015891945,0.00003457488,0.000043498996,0.000659854,0.001340259],"genre_scores_gemma":[0.033690784,0.00016411697,0.9623782,0.00027888844,0.000059263362,0.00019110974,0.00017075655,0.00032439534,0.0027425247],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99728596,0.00090950454,0.00019774432,0.00060833606,0.0008493404,0.00014918952],"domain_scores_gemma":[0.99561006,0.0024814545,0.00025146554,0.0011132392,0.0004401324,0.000103681174],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0041547203,0.0010288542,0.0005294802,0.001238451,0.0013920082,0.0035521465,0.00358017,0.0015440799,0.008812308],"category_scores_gemma":[0.00911253,0.0008365449,0.001924438,0.0012310304,0.0038189269,0.005951735,0.003022399,0.0044027083,0.0023688595],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000025899712,0.000036313522,0.00018241354,0.00015846339,0.000018868803,0.00018560006,0.0005085963,0.008957563,0.0025661304,0.9286578,0.0029940251,0.05570824],"study_design_scores_gemma":[0.000026229302,0.000050538154,0.00009584732,0.00008704589,0.0000617073,0.00039286155,0.00013173593,0.12133605,0.014510935,0.8117332,0.0515244,0.000049443715],"about_ca_topic_score_codex":0.001214088,"about_ca_topic_score_gemma":0.0016778584,"teacher_disagreement_score":0.008812308,"about_ca_system_score_codex":0.0014877539,"about_ca_system_score_gemma":0.0023730034,"threshold_uncertainty_score":0.0294801},"labels":[],"label_agreement":null},{"id":"W1994535609","doi":"10.1145/1562764.1562799","title":"How effective is Google's translation service in search?","year":2009,"lang":"en","type":"article","venue":"Communications of the ACM","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":38,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"Schweizerischer Nationalfonds zur Förderung der Wissenschaftlichen Forschung","keywords":"Computer science; World Wide Web; Portuguese; German; The Internet; Service (business); Machine translation; Government (linguistics); Artificial intelligence; Linguistics; Business","score_opus":0.03983667977309491,"score_gpt":0.32500336739025554,"score_spread":0.28516668761716063,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1994535609","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12839122,0.09000633,0.030797772,0.16405867,0.008568696,0.0010129804,0.025847325,0.093169436,0.45814764],"genre_scores_gemma":[0.7527525,0.05520624,0.048732184,0.021179948,0.0059723863,0.0005092386,0.036302723,0.008232066,0.07111281],"study_design_codex":"not_applicable","study_design_gemma":"observational","domain_scores_codex":[0.99072087,0.0032359564,0.00070216076,0.00072197366,0.003679512,0.00093957124],"domain_scores_gemma":[0.98757,0.0057242266,0.0006579653,0.0011671949,0.0040388387,0.00084176514],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004936116,0.0017033807,0.0019806686,0.009384922,0.001348806,0.008926832,0.0018266019,0.0035881193,0.018490378],"category_scores_gemma":[0.027896993,0.00058479677,0.0013577738,0.012666846,0.0014820852,0.019700674,0.0021187428,0.0012602858,0.028862592],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00070964795,0.00012933463,0.007552049,0.002163487,0.00016511229,0.00043175445,0.0009079288,0.0011247462,0.0013206961,0.009745296,0.53343683,0.44231308],"study_design_scores_gemma":[0.0003278797,0.0005172804,0.021012656,0.0016510106,0.00049544865,0.002478724,0.006818065,0.026386712,0.0060386565,0.02593578,0.90774345,0.00059435586],"about_ca_topic_score_codex":0.034868687,"about_ca_topic_score_gemma":0.022906596,"teacher_disagreement_score":0.034868687,"about_ca_system_score_codex":0.0022750318,"about_ca_system_score_gemma":0.0030996026,"threshold_uncertainty_score":0.06933147},"labels":[],"label_agreement":null},{"id":"W1995915182","doi":"10.7202/002712ar","title":"The Realization of Individual Instances in a Multilingual Generation System","year":2002,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Realization (probability); Computer science; Sentence; Machine translation; Natural language processing; Artificial intelligence; Translation (biology); Linguistics; Mathematics; Philosophy","score_opus":0.055591237793201166,"score_gpt":0.28180667910595586,"score_spread":0.2262154413127547,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1995915182","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04590347,0.0001578442,0.929635,0.00051481376,0.00015734657,0.00011028427,0.00038242716,0.006268528,0.01687029],"genre_scores_gemma":[0.45519993,0.00014611763,0.5365146,0.0001833674,0.000077353994,0.00011663364,0.0011875513,0.0009818836,0.0055925394],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989774,0.00048219858,0.00007784012,0.00019370313,0.00019490038,0.00007386657],"domain_scores_gemma":[0.9992847,0.00036859425,0.000039510578,0.0001682772,0.00010963841,0.000029275956],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013918802,0.0003626262,0.0004285584,0.0005234924,0.00086577726,0.0019684527,0.0010411514,0.00093999004,0.0044513308],"category_scores_gemma":[0.0023287092,0.0004039588,0.0006620372,0.00042702,0.0010803918,0.0020543418,0.0015549021,0.0010429939,0.0015518559],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00040640778,0.000120681696,0.0017223156,0.00037681955,0.000078816585,0.002124947,0.0043401844,0.032677837,0.038677204,0.678403,0.013902896,0.22716892],"study_design_scores_gemma":[0.00016228984,0.00019589852,0.0008569236,0.00016110016,0.00019035573,0.0010720823,0.0008344729,0.4473175,0.10067732,0.30905268,0.13933386,0.00014555693],"about_ca_topic_score_codex":0.0011123504,"about_ca_topic_score_gemma":0.0015193766,"teacher_disagreement_score":0.0044513308,"about_ca_system_score_codex":0.00061993144,"about_ca_system_score_gemma":0.00055142597,"threshold_uncertainty_score":0.014891207},"labels":[],"label_agreement":null},{"id":"W1996401972","doi":"10.1109/vast.2012.6400530","title":"LensingWikipedia: Parsing text for the interactive visualization of human history","year":2012,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Coordenação de Aperfeiçoamento de Pessoal de Nível Superior; Boeing","keywords":"Computer science; Parsing; Novelty; Natural language processing; Visualization; Artificial intelligence; Word (group theory); Interactive visualization; Information extraction; Information retrieval; World Wide Web; Linguistics","score_opus":0.03369276750148003,"score_gpt":0.3298876716702605,"score_spread":0.29619490416878047,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1996401972","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019672394,0.0013204728,0.7081845,0.0012130579,0.00036112973,0.0004681547,0.03345467,0.21921715,0.016108532],"genre_scores_gemma":[0.10340189,0.0017816562,0.8270162,0.00032775954,0.00017930298,0.0009828905,0.027443687,0.02706173,0.011804909],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99973744,0.00006366278,0.00002764192,0.000071825176,0.00007217358,0.000027192253],"domain_scores_gemma":[0.9985947,0.00075416017,0.00012446177,0.00020653348,0.00017808935,0.0001419668],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006024859,0.0015529572,0.0005549116,0.004786773,0.00081860804,0.0024532017,0.0013869242,0.00092676,0.025917783],"category_scores_gemma":[0.0039377613,0.0006393608,0.0006526523,0.0024951785,0.0005315991,0.0040559517,0.0026423626,0.0014376903,0.0063032326],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006470076,0.00019377907,0.0046940264,0.00366032,0.00023983832,0.0021173158,0.008067454,0.0038862592,0.08565955,0.029423371,0.3601567,0.5012545],"study_design_scores_gemma":[0.0001465163,0.00010614116,0.007312626,0.00062816334,0.00009535243,0.0017939579,0.0019962993,0.076622546,0.06979008,0.04594412,0.79530954,0.00025461553],"about_ca_topic_score_codex":0.0027040206,"about_ca_topic_score_gemma":0.008646376,"teacher_disagreement_score":0.025917783,"about_ca_system_score_codex":0.00039273137,"about_ca_system_score_gemma":0.00093632424,"threshold_uncertainty_score":0.08670366},"labels":[],"label_agreement":null},{"id":"W1996772524","doi":"10.7202/018528ar","title":"Restrictions paradigmatiques et traduction de schémas d’arguments","year":2008,"lang":"fr","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Philosophy","score_opus":0.06662396260270612,"score_gpt":0.31414383227831316,"score_spread":0.24751986967560705,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1996772524","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.027318513,0.0014176243,0.9285123,0.0028728643,0.0003370464,0.00027541275,0.00076344225,0.001104738,0.037398178],"genre_scores_gemma":[0.2884556,0.0029757293,0.6739245,0.001323887,0.0005226461,0.0011685112,0.0028391303,0.0015169542,0.02727307],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.979474,0.0074553783,0.0026498141,0.0037494025,0.0058757807,0.0007955631],"domain_scores_gemma":[0.97396654,0.012516697,0.001445983,0.0080195265,0.0036749868,0.00037629314],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01372294,0.0012335435,0.0012450384,0.0038838934,0.0038775636,0.013561136,0.0036181684,0.0032909454,0.007905235],"category_scores_gemma":[0.025088646,0.0024728335,0.003121774,0.0046817893,0.014653185,0.026657347,0.0059695262,0.0084973555,0.003278678],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000024733876,0.000011363826,0.00033962846,0.000100601,0.000018739607,0.0001315658,0.0023475857,0.00042500003,0.0015037571,0.9836211,0.00085677404,0.010619191],"study_design_scores_gemma":[0.00005753211,0.00003946405,0.0006850751,0.00022063802,0.00007240664,0.0010873945,0.0023136816,0.005698456,0.009698615,0.8022524,0.17779392,0.00008041849],"about_ca_topic_score_codex":0.006766831,"about_ca_topic_score_gemma":0.003462971,"teacher_disagreement_score":0.01372294,"about_ca_system_score_codex":0.0048033344,"about_ca_system_score_gemma":0.005436508,"threshold_uncertainty_score":0.072574735},"labels":[],"label_agreement":null},{"id":"W1996845866","doi":"10.1142/s0218213006003089","title":"COORDINATION AND APPLICATIVE CATEGORIAL TYPE LOGIC","year":2006,"lang":"en","type":"article","venue":"International Journal of Artificial Intelligence Tools","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Trois-Rivières","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Categorial grammar; Combinatory categorial grammar; Conjunction (astronomy); Programming language; Type (biology); Process (computing); Artificial intelligence; Generative grammar; Theoretical computer science; Natural language processing; Link grammar; Mildly context-sensitive grammar formalism; Head-driven phrase structure grammar; Emergent grammar","score_opus":0.041097519509761586,"score_gpt":0.3361296487219334,"score_spread":0.2950321292121718,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1996845866","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12168495,0.007292578,0.65974945,0.007278872,0.0008927489,0.0001142488,0.0004859623,0.0019905204,0.20051073],"genre_scores_gemma":[0.9136348,0.0018072344,0.0641093,0.00080760644,0.0004698467,0.00008427519,0.00024115779,0.00019319044,0.018652571],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9981179,0.0007124846,0.00013414602,0.0003640376,0.00039097367,0.00028047548],"domain_scores_gemma":[0.9982161,0.0008988482,0.0002087717,0.0002942578,0.00027048594,0.000111530775],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024123152,0.00048130457,0.00058245653,0.0019078571,0.0015821446,0.0041971807,0.0010584994,0.0013237263,0.0045006997],"category_scores_gemma":[0.002718453,0.00040741154,0.001228439,0.0017953524,0.0057314364,0.004644973,0.0022229638,0.0015507592,0.00071163545],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000068510535,0.0000026436348,0.00016055466,0.000016385025,0.0000044005383,0.00008246742,0.00019082491,0.0007392413,0.00021340541,0.99564123,0.00038570253,0.0025562998],"study_design_scores_gemma":[0.00000873805,0.000007482809,0.00017494512,0.000014000034,0.0000089637815,0.00013219322,0.00009711742,0.0034588317,0.0002850343,0.98650396,0.00929955,0.000009198502],"about_ca_topic_score_codex":0.0050267037,"about_ca_topic_score_gemma":0.002376051,"teacher_disagreement_score":0.0050267037,"about_ca_system_score_codex":0.0030277586,"about_ca_system_score_gemma":0.0015686511,"threshold_uncertainty_score":0.021968007},"labels":[],"label_agreement":null},{"id":"W1997585218","doi":"10.1109/somet.2013.6645660","title":"A reason to optimize information processing with a core property of natural language","year":2013,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Question answering; Focus (optics); Argument (complex analysis); Computer science; Property (philosophy); Natural language; Natural (archaeology); Natural language processing; Information structure; Core (optical fiber); Artificial intelligence; Information retrieval; Linguistics; Epistemology; Philosophy","score_opus":0.007094482436050728,"score_gpt":0.23539939421846667,"score_spread":0.22830491178241594,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1997585218","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.031734075,0.0011914488,0.9428234,0.010360523,0.00021747779,0.00014062058,0.00023033672,0.0011968729,0.01210523],"genre_scores_gemma":[0.47285256,0.0010830599,0.51501703,0.0028609298,0.0008224326,0.00039702898,0.0005345287,0.00081885676,0.00561357],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99550605,0.0011509099,0.00046219863,0.0011958797,0.0013197325,0.00036519516],"domain_scores_gemma":[0.9769527,0.0109768165,0.001741489,0.0075150724,0.0023142886,0.000499654],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0057117892,0.00091158604,0.0010699965,0.001126362,0.0012898663,0.0048441156,0.001853729,0.0030263972,0.0053423108],"category_scores_gemma":[0.028754806,0.00088728743,0.0014768942,0.0014732821,0.004981551,0.018559728,0.0030782437,0.0036132913,0.0019196493],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00042978267,0.00028427795,0.0041543352,0.00095281546,0.0002061109,0.00037408172,0.0008280668,0.017561475,0.048754502,0.78588706,0.010858637,0.12970889],"study_design_scores_gemma":[0.00010374797,0.00013625128,0.0008923821,0.000051948715,0.00009941195,0.00041821157,0.0001482483,0.05216475,0.027603442,0.9036142,0.014706062,0.00006140067],"about_ca_topic_score_codex":0.0006840583,"about_ca_topic_score_gemma":0.0005467343,"teacher_disagreement_score":0.0057117892,"about_ca_system_score_codex":0.0010043647,"about_ca_system_score_gemma":0.0020561041,"threshold_uncertainty_score":0.030207217},"labels":[],"label_agreement":null},{"id":"W19976681","doi":"","title":"Computer-assisted vocabulary learning: multimedia annotations, word concreteness, and individualized instruction","year":2010,"lang":"en","type":"dissertation","venue":"Summit (Simon Fraser University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Simon Fraser University","keywords":"Concreteness; Computer science; Multimedia; Word (group theory); Vocabulary; Vocabulary learning; Vocabulary development; Word learning; Natural language processing; Linguistics; Psychology; Cognitive psychology","score_opus":0.00882778504942462,"score_gpt":0.23032901279306128,"score_spread":0.22150122774363667,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W19976681","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9870394,0.0006371482,0.0043157414,0.00018602505,0.000013152629,0.00009560054,0.000015238995,0.000066732224,0.0076308823],"genre_scores_gemma":[0.9920534,0.00036865575,0.006604659,0.000051107738,0.000011134351,0.000080777056,0.000028850356,0.000013470869,0.0007879847],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9981536,0.00093769986,0.000103113154,0.00028063005,0.00044434334,0.000080507576],"domain_scores_gemma":[0.9868521,0.010568421,0.001340735,0.00036236783,0.00044896468,0.000427454],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027483262,0.0002778868,0.00033858398,0.00056140544,0.00032881898,0.001573157,0.00049387774,0.00047922047,0.0016370711],"category_scores_gemma":[0.017074695,0.00015693805,0.00023617137,0.00047471235,0.0007063995,0.002134685,0.0010949564,0.0005188991,0.00013075695],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013724412,0.0045438567,0.06181171,0.0012158349,0.00021174559,0.00019050243,0.011427209,0.0035636497,0.02632436,0.0058265566,0.00074667705,0.8827654],"study_design_scores_gemma":[0.0006824835,0.016186628,0.7825594,0.0019157715,0.00177862,0.0012075823,0.017530367,0.043110948,0.08124277,0.03556233,0.01792785,0.00029520353],"about_ca_topic_score_codex":0.001154467,"about_ca_topic_score_gemma":0.0013594953,"teacher_disagreement_score":0.0027483262,"about_ca_system_score_codex":0.0005068719,"about_ca_system_score_gemma":0.000988039,"threshold_uncertainty_score":0.014534712},"labels":[],"label_agreement":null},{"id":"W1998064521","doi":"10.7202/001856ar","title":"La lexicographie assistée par ordinateur. L'expérience d'UZEI","year":2002,"lang":"fr","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"Université de Montréal; Université Laval; Université du Québec à Montréal","funders":"","keywords":"Humanities; Unix; Philosophy; Computer science; Art; Programming language; Software","score_opus":0.04721879845460589,"score_gpt":0.2787349160415302,"score_spread":0.23151611758692434,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1998064521","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17872572,0.0068801707,0.63403666,0.0027999626,0.0004438621,0.0007319054,0.00817095,0.07344392,0.0947668],"genre_scores_gemma":[0.32018104,0.00329774,0.61067575,0.00053929625,0.00012683867,0.0002504976,0.010143184,0.0049097016,0.049876],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9950169,0.0018155854,0.00047040635,0.00084469124,0.0015468611,0.00030561956],"domain_scores_gemma":[0.9875485,0.007527028,0.0002780037,0.0023247157,0.002045189,0.00027663083],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005617471,0.0009796872,0.0011691038,0.003042596,0.0015215707,0.004916794,0.0011579666,0.0010320123,0.026943943],"category_scores_gemma":[0.01762572,0.00066398864,0.00058976724,0.0053511714,0.000977975,0.005519687,0.002637558,0.0016693518,0.012226219],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001381697,0.00022712043,0.0059683225,0.0011398676,0.00006650643,0.00063906267,0.011089888,0.0024042716,0.04079633,0.013057353,0.016865648,0.9063639],"study_design_scores_gemma":[0.00025105535,0.0005964495,0.01399057,0.0004348355,0.00020115863,0.0020890047,0.0068864515,0.02442261,0.135107,0.009483603,0.8063251,0.00021215285],"about_ca_topic_score_codex":0.01929758,"about_ca_topic_score_gemma":0.018594828,"teacher_disagreement_score":0.026943943,"about_ca_system_score_codex":0.0012588006,"about_ca_system_score_gemma":0.0024258639,"threshold_uncertainty_score":0.09013653},"labels":[],"label_agreement":null},{"id":"W1998161940","doi":"10.1109/icassp.2013.6639315","title":"Unsupervised topic model for broadcast program segmentation","year":2013,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure; Computer Research Institute of Montréal","funders":"","keywords":"Computer science; Segmentation; Hidden Markov model; Simple (philosophy); Artificial intelligence; Space (punctuation); Machine learning; Dynamic programming; Algorithm","score_opus":0.022390245173163643,"score_gpt":0.3009526615130787,"score_spread":0.2785624163399151,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1998161940","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019216338,0.00043724483,0.97587186,0.00024535655,0.00005388052,0.00007676158,0.0012407379,0.001338341,0.0015195388],"genre_scores_gemma":[0.57569146,0.0012303143,0.39380562,0.00027464592,0.0005093093,0.0010284495,0.012418804,0.00097390264,0.014067505],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99890983,0.00034467247,0.000056685396,0.00038351674,0.00017075783,0.00013453237],"domain_scores_gemma":[0.99789494,0.0014623073,0.00012849492,0.00019722189,0.00025834097,0.00005866869],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012922684,0.0007744879,0.0013319376,0.0020227528,0.00062830775,0.0012560018,0.0020486854,0.0014409772,0.004045302],"category_scores_gemma":[0.0046307687,0.00065832905,0.0013386173,0.0023095324,0.00071874895,0.0024858143,0.001004524,0.0018615543,0.0016931833],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009254375,0.0003061257,0.005839751,0.0005438597,0.00034789566,0.00044750934,0.0015661936,0.56671333,0.017466143,0.09835727,0.018985482,0.288501],"study_design_scores_gemma":[0.000019408639,0.00002058512,0.0008834556,0.0000110708215,0.000026094795,0.00006722838,0.000043684428,0.97576356,0.0013527651,0.018712386,0.0030832307,0.00001648766],"about_ca_topic_score_codex":0.008991179,"about_ca_topic_score_gemma":0.012117105,"teacher_disagreement_score":0.008991179,"about_ca_system_score_codex":0.001051905,"about_ca_system_score_gemma":0.0011422684,"threshold_uncertainty_score":0.017877698},"labels":[],"label_agreement":null},{"id":"W1998749067","doi":"10.7202/1006186ar","title":"The Shifting of the Demonstrative Determiner in French and Dutch in Parallel Corpora: From Translation Mechanisms to Structural Differences","year":2011,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Demonstrative; Determiner; Determiner phrase; Linguistics; Adverb; Noun phrase; Syntagmatic analysis; Computer science; Noun; Natural language processing; Personal pronoun; Artificial intelligence; Philosophy","score_opus":0.05149424141017771,"score_gpt":0.25897408349963563,"score_spread":0.2074798420894579,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1998749067","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9268624,0.0039805234,0.02440182,0.00062332104,0.00015436033,0.00032966054,0.0075480747,0.00025839056,0.035841536],"genre_scores_gemma":[0.96252495,0.0012667198,0.01955023,0.00009414598,0.00006688019,0.00048786547,0.011801954,0.00032931232,0.0038779783],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99541664,0.0017448999,0.00048947916,0.0011349332,0.0010172845,0.00019669364],"domain_scores_gemma":[0.992008,0.004401103,0.00086484564,0.0010380297,0.001575759,0.00011231759],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0038906643,0.00046555576,0.00050206744,0.003285393,0.0016574732,0.0018434646,0.0005265659,0.0005040642,0.0043172864],"category_scores_gemma":[0.012734242,0.00036373307,0.00046574374,0.00471101,0.0014205154,0.0015616666,0.0012286889,0.0005669394,0.0007234407],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002217724,0.00043945937,0.108121894,0.0045006988,0.0007211829,0.007127255,0.07957571,0.0043878034,0.19059771,0.05062859,0.016140772,0.53554124],"study_design_scores_gemma":[0.00028590983,0.0004528276,0.5590009,0.0005226249,0.00050919084,0.009366945,0.024564672,0.011349382,0.059137013,0.007707328,0.32683527,0.00026799584],"about_ca_topic_score_codex":0.02410359,"about_ca_topic_score_gemma":0.036970798,"teacher_disagreement_score":0.02410359,"about_ca_system_score_codex":0.0020541912,"about_ca_system_score_gemma":0.0011083452,"threshold_uncertainty_score":0.047926545},"labels":[],"label_agreement":null},{"id":"W1998787753","doi":"10.1142/s0218213003001216","title":"Logic Grammars for Diagnosis and Repair","year":2003,"lang":"en","type":"article","venue":"International Journal of Artificial Intelligence Tools","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Parsing; Rule-based machine translation; String (physics); Transformation (genetics); Natural language processing; Constraint (computer-aided design); Artificial intelligence; Grammar; Programming language; Parsing expression grammar; Feature (linguistics); Context-free grammar; L-attributed grammar; Definite clause grammar; Grammar induction; Theoretical computer science; Mathematics; Linguistics","score_opus":0.07888665829628552,"score_gpt":0.3605100441032646,"score_spread":0.2816233858069791,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1998787753","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0010409808,0.00014719673,0.9948297,0.0004989479,0.00003896124,0.00009082471,0.00010549987,0.0010893686,0.0021584716],"genre_scores_gemma":[0.046515565,0.00034950968,0.9489635,0.00040369996,0.00007035289,0.00028117362,0.00037017747,0.00025024466,0.0027957186],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99643904,0.001034095,0.00038193288,0.00072118145,0.0012120868,0.0002116736],"domain_scores_gemma":[0.9952147,0.002638554,0.00034497597,0.0011297354,0.00055019883,0.00012181059],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030038643,0.0010932012,0.0008303077,0.0019779878,0.00094577944,0.0027791045,0.0036825933,0.002489036,0.006138124],"category_scores_gemma":[0.010325321,0.00077980733,0.0021438869,0.0012293865,0.004037044,0.004254829,0.0031174761,0.003352956,0.002124764],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006343215,0.00007578625,0.0005936661,0.0003244192,0.000058384685,0.0005701138,0.0005465422,0.08553819,0.004455068,0.79792297,0.004524351,0.10532705],"study_design_scores_gemma":[0.00004284826,0.000042249223,0.00011363436,0.00009574924,0.000056636894,0.000358241,0.0001136684,0.20741485,0.0061487774,0.7579405,0.02763334,0.00003942348],"about_ca_topic_score_codex":0.003143695,"about_ca_topic_score_gemma":0.0027533208,"teacher_disagreement_score":0.006138124,"about_ca_system_score_codex":0.0014731284,"about_ca_system_score_gemma":0.0028759323,"threshold_uncertainty_score":0.020534039},"labels":[],"label_agreement":null},{"id":"W1999148771","doi":"10.7202/002580ar","title":"Préalables linguistiques pour une traduction assistée par ordinateur du système verbal arabe-français","year":2002,"lang":"fr","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Philosophy; Art","score_opus":0.039994886537101135,"score_gpt":0.26563770632472744,"score_spread":0.2256428197876263,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1999148771","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.095163606,0.0019195377,0.8596048,0.0032297364,0.00031260928,0.00034745494,0.0011206478,0.014424727,0.023876905],"genre_scores_gemma":[0.3185168,0.0013846012,0.64605147,0.0004570174,0.00012819344,0.00020603671,0.002357792,0.0009455831,0.029952489],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99777335,0.00086319936,0.00025368444,0.00038562843,0.0005854853,0.00013864154],"domain_scores_gemma":[0.9963387,0.0015634289,0.00019074492,0.0006007899,0.0012024969,0.000103882405],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036870488,0.00075614994,0.0004983513,0.0015509061,0.0018328547,0.004469464,0.0008386484,0.0009491077,0.009776894],"category_scores_gemma":[0.007743946,0.0005152546,0.0010043345,0.0012842873,0.001354799,0.0042459536,0.0011157001,0.0017805208,0.0032108051],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008633541,0.0002176403,0.008827985,0.0014620607,0.00021080927,0.0006653543,0.009644716,0.009615309,0.09474417,0.18237592,0.016904168,0.6744685],"study_design_scores_gemma":[0.00016895786,0.00067374256,0.008734773,0.0007377721,0.0006139134,0.0016373186,0.010055887,0.13779525,0.19793405,0.066255815,0.5750548,0.00033765173],"about_ca_topic_score_codex":0.02783061,"about_ca_topic_score_gemma":0.041417487,"teacher_disagreement_score":0.02783061,"about_ca_system_score_codex":0.0021258024,"about_ca_system_score_gemma":0.0037859522,"threshold_uncertainty_score":0.05533719},"labels":[],"label_agreement":null},{"id":"W1999506550","doi":"10.1075/term.10.1.12dro","title":"Review of Jackson &amp; Moulinier (2002): Natural language processing for online applications: Text retrieval, extraction and categorization","year":2004,"lang":"en","type":"article","venue":"Terminology International Journal of Theoretical and Applied Issues in Specialized Communication","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Categorization; Natural language processing; Computer science; Artificial intelligence; Keyword extraction; Information retrieval; Linguistics; Philosophy","score_opus":0.01488603802225284,"score_gpt":0.36230131512476893,"score_spread":0.3474152771025161,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1999506550","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.001490427,0.9083163,0.05602964,0.014846702,0.004320367,0.00011621718,0.00064339855,0.00038194665,0.013855043],"genre_scores_gemma":[0.010747149,0.90144324,0.059686646,0.007760638,0.0033035788,0.00014445987,0.0012080411,0.00022825929,0.01547799],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9984236,0.0003906854,0.00017903792,0.0002673092,0.00067009975,0.00006928809],"domain_scores_gemma":[0.9936241,0.0025610812,0.00024708937,0.00038527406,0.0029791608,0.00020327537],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029504057,0.0010833199,0.0015790947,0.009550248,0.0009366933,0.0028943839,0.001867502,0.0016501741,0.0042683785],"category_scores_gemma":[0.009587417,0.00041020845,0.0005437899,0.0137745235,0.0014585749,0.0063030357,0.0019037982,0.001640779,0.0058590164],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00004461814,0.000065475615,0.00039147402,0.004397595,0.000047162386,0.00006912656,0.00025687428,0.00042385477,0.0016465224,0.020211166,0.12921558,0.8432306],"study_design_scores_gemma":[0.000005093017,0.00002136545,0.0009131548,0.0013918444,0.00007066252,0.00023870198,0.00013496207,0.0004847287,0.0015856095,0.005785815,0.98934376,0.000024261328],"about_ca_topic_score_codex":0.0080663515,"about_ca_topic_score_gemma":0.011603031,"teacher_disagreement_score":0.009550248,"about_ca_system_score_codex":0.0023765946,"about_ca_system_score_gemma":0.004608308,"threshold_uncertainty_score":0.017243505},"labels":[],"label_agreement":null},{"id":"W1999646438","doi":"10.1075/term.10.1.07lan","title":"General-purpose statistical translation engine and domain specific texts","year":2004,"lang":"en","type":"article","venue":"Terminology International Journal of Theoretical and Applied Issues in Specialized Communication","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Terminology; Computer science; Machine translation; Natural language processing; Word error rate; Word (group theory); Artificial intelligence; Field (mathematics); Domain (mathematical analysis); Information retrieval; Linguistics; Mathematics","score_opus":0.011482441705812961,"score_gpt":0.3026671055725336,"score_spread":0.29118466386672065,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1999646438","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.65286094,0.0026040373,0.31293315,0.0007467393,0.0002338795,0.00034569585,0.00090351625,0.015507881,0.01386422],"genre_scores_gemma":[0.7749357,0.00051533914,0.21514277,0.0002550068,0.00006398449,0.00015171748,0.0024517414,0.00076573243,0.0057179197],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9974796,0.0010101931,0.0002599604,0.000511584,0.0006246506,0.000114002774],"domain_scores_gemma":[0.9878996,0.006864264,0.00071341934,0.0024759676,0.0018568066,0.00018987857],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0044354913,0.00085392315,0.0012990221,0.0010975813,0.0007315919,0.001788707,0.0013271682,0.0015192355,0.0036723414],"category_scores_gemma":[0.0245526,0.0005122406,0.000535623,0.0021667639,0.0008099886,0.0038801837,0.0011840921,0.0012033094,0.0033570186],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0022288566,0.0007532499,0.020005371,0.0019948012,0.0005969697,0.0022462895,0.001301103,0.14641039,0.20049247,0.0342225,0.010159446,0.57958853],"study_design_scores_gemma":[0.00009236258,0.0008104489,0.0061921906,0.000066642475,0.00023256028,0.0016731364,0.00038680766,0.82191706,0.13071917,0.019967599,0.0178512,0.0000908579],"about_ca_topic_score_codex":0.0015601774,"about_ca_topic_score_gemma":0.0024867118,"teacher_disagreement_score":0.0044354913,"about_ca_system_score_codex":0.0005710715,"about_ca_system_score_gemma":0.0010793406,"threshold_uncertainty_score":0.023457408},"labels":[],"label_agreement":null},{"id":"W1999912290","doi":"10.1145/1088622.1088626","title":"Extracting knowledge from evaluative text","year":2005,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":199,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Redundancy (engineering); Feature extraction; Artificial intelligence; Task (project management); Natural language processing; Information retrieval; Knowledge extraction; Feature (linguistics); Semantic feature; Pattern recognition (psychology); Data mining; Engineering","score_opus":0.02521412532662159,"score_gpt":0.34087819817873305,"score_spread":0.31566407285211145,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1999912290","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.26810306,0.0018668675,0.68102974,0.001229481,0.00012160202,0.0008366831,0.009395691,0.0043166745,0.033100244],"genre_scores_gemma":[0.7148268,0.0011855323,0.26069188,0.00016443813,0.00020159468,0.00048537477,0.014742619,0.00025760906,0.0074441205],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99834335,0.00047399997,0.00014470183,0.00036314415,0.0005493842,0.00012547658],"domain_scores_gemma":[0.9928307,0.004021548,0.0006861488,0.00092734577,0.001389822,0.00014447981],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013822293,0.0008209645,0.0005812357,0.005622126,0.00061859813,0.0018998671,0.00068896747,0.0007366753,0.003054311],"category_scores_gemma":[0.012605609,0.0002875057,0.00058950426,0.0032516194,0.00075514545,0.0039673564,0.0013297602,0.00081710593,0.0017315397],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00031557694,0.0002673102,0.008332964,0.0012455032,0.0001268014,0.0012623848,0.0026512642,0.007939621,0.035888676,0.021806208,0.010187508,0.9099762],"study_design_scores_gemma":[0.000121107376,0.0007626138,0.07787216,0.0011831389,0.0005205178,0.0026996667,0.008856044,0.31028798,0.17698519,0.2244711,0.19590463,0.00033595203],"about_ca_topic_score_codex":0.0012452749,"about_ca_topic_score_gemma":0.0023244417,"teacher_disagreement_score":0.005622126,"about_ca_system_score_codex":0.0008705085,"about_ca_system_score_gemma":0.0007886227,"threshold_uncertainty_score":0.010217726},"labels":[],"label_agreement":null},{"id":"W200075660","doi":"","title":"Parallel web text mining for cross-language IR","year":2000,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":72,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Machine translation; Parallel corpora; Web mining; Cross-language information retrieval; Natural language processing; Information retrieval; Artificial intelligence; Data mining; Web service; World Wide Web","score_opus":0.013050783588800812,"score_gpt":0.3073651442071318,"score_spread":0.294314360618331,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W200075660","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010343946,0.00089926634,0.9682152,0.0005909474,0.00016843011,0.00047577152,0.0009109467,0.014616537,0.0037789915],"genre_scores_gemma":[0.07798153,0.00055334833,0.9124344,0.0002178545,0.00018323191,0.0006337071,0.0032932416,0.00071569846,0.0039869854],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9966127,0.0014192337,0.0003398066,0.00069546304,0.00082143187,0.000111356625],"domain_scores_gemma":[0.9911376,0.0037336068,0.0006845661,0.0025843852,0.0017129101,0.00014692664],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0043849377,0.00097031204,0.001201212,0.0049497774,0.0015088308,0.0020343647,0.0017086358,0.0010898074,0.009304539],"category_scores_gemma":[0.012287208,0.00066913234,0.0012606946,0.0052965516,0.0008033365,0.0045708604,0.0020514384,0.0016794892,0.007704584],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003341811,0.00035387967,0.0016588181,0.00045250286,0.00012896686,0.00042453004,0.00031044206,0.0129497945,0.016474184,0.014293239,0.016832544,0.9357869],"study_design_scores_gemma":[0.000226114,0.00031157403,0.0035500508,0.00016820457,0.0001546787,0.001672269,0.00045093792,0.66739374,0.062385883,0.1406763,0.122871675,0.00013859902],"about_ca_topic_score_codex":0.0015008327,"about_ca_topic_score_gemma":0.0019913523,"teacher_disagreement_score":0.009304539,"about_ca_system_score_codex":0.0007361212,"about_ca_system_score_gemma":0.0013973308,"threshold_uncertainty_score":0.031126857},"labels":[],"label_agreement":null},{"id":"W2001023236","doi":"10.1145/1236181.1236184","title":"Statistical query translation models for cross-language information retrieval","year":2006,"lang":"en","type":"article","venue":"ACM Transactions on Asian Language Information Processing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":29,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Cross-language information retrieval; Natural language processing; Query expansion; Artificial intelligence; Machine translation; Query language; RDF query language; Translation (biology); Dependency (UML); Query optimization; Context (archaeology); Information retrieval; Web query classification; Web search query; Search engine","score_opus":0.011854377535112912,"score_gpt":0.2911378651152146,"score_spread":0.2792834875801017,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2001023236","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0038930932,0.0014682588,0.99129665,0.00044836756,0.00010034733,0.00016307377,0.000256553,0.0013606228,0.0010130305],"genre_scores_gemma":[0.31193283,0.0039788326,0.67111474,0.0009181272,0.0008314403,0.0021889664,0.0030232635,0.0008937448,0.0051180655],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9893519,0.006758378,0.0006463814,0.001008171,0.0019691407,0.00026594347],"domain_scores_gemma":[0.9831176,0.011672894,0.0011419146,0.0020508685,0.0018993072,0.00011741098],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01093543,0.0019273307,0.0020915251,0.0033353376,0.0010002394,0.0022549091,0.0026221666,0.0019548547,0.0040180315],"category_scores_gemma":[0.02368716,0.0010173658,0.002663944,0.004964612,0.0015108975,0.0057662628,0.0017057176,0.0023086711,0.004110179],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00063704577,0.00038823852,0.0026442688,0.0009888009,0.0006410162,0.00035344835,0.0005833211,0.40438482,0.0066293823,0.14353588,0.015639707,0.423574],"study_design_scores_gemma":[0.00004511391,0.0001167858,0.00035402263,0.000021394584,0.000060774597,0.00014545806,0.00004308364,0.936632,0.0012116797,0.056485657,0.004835965,0.000048157886],"about_ca_topic_score_codex":0.0049704784,"about_ca_topic_score_gemma":0.00472648,"teacher_disagreement_score":0.01093543,"about_ca_system_score_codex":0.0021179656,"about_ca_system_score_gemma":0.0019304518,"threshold_uncertainty_score":0.057832778},"labels":[],"label_agreement":null},{"id":"W2001343910","doi":"10.3115/974147.974165","title":"An automatic reviser","year":2000,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Toolbox; Exploit; Computer science; Machine translation; Computational linguistics; Artificial intelligence; Translation (biology); Volume (thermodynamics); Natural language processing; Work (physics); Software engineering; Programming language; Engineering; Computer security","score_opus":0.007714444247325436,"score_gpt":0.2759246238959363,"score_spread":0.2682101796486109,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2001343910","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.058585558,0.0041766716,0.6058077,0.0031655515,0.00526678,0.0018134079,0.00867149,0.18863094,0.123881906],"genre_scores_gemma":[0.20682538,0.0019342887,0.61356664,0.0016739026,0.0010057837,0.0005104715,0.01479371,0.013638825,0.14605097],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99662834,0.00058634894,0.0003212734,0.0012838154,0.0009871025,0.00019315223],"domain_scores_gemma":[0.9911375,0.0024226033,0.0005152425,0.0033864,0.0023133007,0.00022490548],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026810942,0.0016608443,0.0014437126,0.0031568783,0.0017237795,0.0032885938,0.0024050488,0.0014368348,0.036631484],"category_scores_gemma":[0.014531639,0.0007695107,0.0009852472,0.0024541086,0.00087724585,0.0040446375,0.0025067802,0.0019294848,0.022149272],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00051684014,0.00016582149,0.0022841885,0.0006619798,0.000060095066,0.00093810837,0.0008169528,0.0019507855,0.03671946,0.014306791,0.117882885,0.8236961],"study_design_scores_gemma":[0.00018883198,0.00038157494,0.004193612,0.00034642365,0.0002996179,0.0030667356,0.0009603512,0.07703987,0.10887611,0.02323188,0.7811735,0.00024142097],"about_ca_topic_score_codex":0.0033027644,"about_ca_topic_score_gemma":0.006077312,"teacher_disagreement_score":0.036631484,"about_ca_system_score_codex":0.0007637168,"about_ca_system_score_gemma":0.0023821965,"threshold_uncertainty_score":0.12254459},"labels":[],"label_agreement":null},{"id":"W2002373856","doi":"10.3115/1119176.1119186","title":"Semi-supervised verb class discovery using noisy features","year":2003,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Artificial intelligence; Cluster analysis; Class (philosophy); Natural language processing; Set (abstract data type); Feature (linguistics); Verb; Feature selection; Task (project management); Selection (genetic algorithm); Unsupervised learning; Supervised learning; Face (sociological concept); Pattern recognition (psychology); Artificial neural network; Linguistics","score_opus":0.015156126113758977,"score_gpt":0.2672699766217805,"score_spread":0.2521138505080215,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2002373856","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.32397604,0.0002829713,0.66929704,0.0002744146,0.00006231689,0.0003369221,0.001258415,0.0022914654,0.0022203398],"genre_scores_gemma":[0.7701841,0.000063869185,0.22358307,0.000087943845,0.00005180257,0.00035351323,0.00423686,0.00015928992,0.001279479],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.996777,0.0011452023,0.00026221553,0.00080069475,0.00074879883,0.00026599556],"domain_scores_gemma":[0.9864927,0.008528125,0.0013694753,0.001778862,0.0015728575,0.00025801023],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0038726141,0.00113055,0.001987777,0.0029211477,0.0008534632,0.001668555,0.0020539705,0.001440678,0.0010209067],"category_scores_gemma":[0.015936574,0.00043975073,0.0012351542,0.0018132094,0.0010868364,0.0025005024,0.0013159428,0.0013165339,0.00075099035],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0026340245,0.0016621555,0.053962696,0.00082325353,0.0006150819,0.00088359276,0.0011479415,0.12260163,0.06790795,0.008372345,0.010603008,0.7287863],"study_design_scores_gemma":[0.00009499291,0.00022749399,0.012860339,0.00003889893,0.00009601701,0.00037206517,0.00024247669,0.93274546,0.036191873,0.014136495,0.0029179247,0.00007599816],"about_ca_topic_score_codex":0.001614418,"about_ca_topic_score_gemma":0.0030330804,"teacher_disagreement_score":0.0038726141,"about_ca_system_score_codex":0.00065398356,"about_ca_system_score_gemma":0.0010215066,"threshold_uncertainty_score":0.020480633},"labels":[],"label_agreement":null},{"id":"W2002451058","doi":"10.1590/s0103-18132005000100003","title":"[NO TITLE AVAILABLE]","year":2005,"lang":"en","type":"article","venue":"Trabalhos em Linguística Aplicada","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Intertek (Canada)","funders":"","keywords":"Modal verb; Variety (cybernetics); Linguistics; Modal; Sentence; Computer science; Point (geometry); Czech; Psychology; Natural language processing; Artificial intelligence; Verb; Mathematics","score_opus":0.012517982137925838,"score_gpt":0.2562548571364905,"score_spread":0.24373687499856464,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2002451058","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0006127201,0.0019138668,0.0010258333,0.0035423292,0.010274238,0.0003716686,0.031678148,0.0019927928,0.9485884],"genre_scores_gemma":[0.0025632135,0.0017215661,0.00097390637,0.0010354907,0.00083779055,0.00015209909,0.015095611,0.0006937044,0.9769265],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99943846,0.000070106464,0.00006477162,0.00011707148,0.00024297691,0.00006657148],"domain_scores_gemma":[0.9980732,0.00028425406,0.00012646218,0.00030407257,0.00094815623,0.00026382707],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.00079251797,0.0008244099,0.0011265283,0.004418987,0.0020206824,0.005103499,0.0025881024,0.0017528523,0.8213821],"category_scores_gemma":[0.0043704645,0.00055247656,0.0008048099,0.0051218118,0.0005190812,0.0025896786,0.0018232376,0.0012836863,0.72876006],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000035246077,0.000024211107,0.0001737439,0.0003905549,0.0000058961054,0.00006878823,0.000059174585,0.000031966894,0.00027009766,0.0032416822,0.93439823,0.061300397],"study_design_scores_gemma":[0.000006035668,0.000009427253,0.00041964723,0.00009811186,0.0000018409767,0.00003778363,0.000034501812,0.00001856096,0.000063979984,0.0003550861,0.99895084,0.00000420612],"about_ca_topic_score_codex":0.0072211935,"about_ca_topic_score_gemma":0.012454238,"teacher_disagreement_score":0.1786179,"about_ca_system_score_codex":0.0012071512,"about_ca_system_score_gemma":0.001705039,"threshold_uncertainty_score":0.25477678},"labels":[],"label_agreement":null},{"id":"W2002586403","doi":"10.3115/1067807.1067851","title":"Bootstrapping statistical parsers from small datasets","year":2003,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":156,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Defense Advanced Research Projects Agency; Johns Hopkins University; National Science Foundation","keywords":"Bootstrapping (finance); Parsing; Computer science; Strapping; Domain (mathematical analysis); Artificial intelligence; Natural language processing; Statistical analysis; Statistics","score_opus":0.0253997854399811,"score_gpt":0.2756000191421251,"score_spread":0.250200233702144,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2002586403","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04777188,0.00032897881,0.9247446,0.00047986413,0.00018521513,0.0006520098,0.0020159853,0.022290083,0.0015313622],"genre_scores_gemma":[0.18952361,0.00023102442,0.78611267,0.00059479085,0.00024557597,0.0021403702,0.016541293,0.0032735122,0.0013371938],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9868347,0.007621726,0.0006979769,0.002562178,0.0019610897,0.00032233735],"domain_scores_gemma":[0.86930454,0.098263286,0.0026328983,0.021080453,0.00785659,0.000862266],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01332503,0.0020372034,0.001894092,0.004020429,0.0016208428,0.0016918201,0.0034158733,0.00234103,0.003974121],"category_scores_gemma":[0.101873875,0.001752707,0.0018236232,0.00371102,0.001438325,0.0047289706,0.0035396027,0.0046056444,0.0041050566],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00083951454,0.0010020575,0.0070716,0.0009190144,0.0005725211,0.0012521478,0.001027194,0.12030747,0.047803357,0.006629388,0.038373705,0.774202],"study_design_scores_gemma":[0.00020445671,0.00031856523,0.0041955416,0.00011753811,0.000185963,0.0006089423,0.0003027159,0.8909331,0.044352926,0.039852522,0.018775344,0.00015237872],"about_ca_topic_score_codex":0.0015756254,"about_ca_topic_score_gemma":0.0040461803,"teacher_disagreement_score":0.01332503,"about_ca_system_score_codex":0.0010891769,"about_ca_system_score_gemma":0.0026466162,"threshold_uncertainty_score":0.07047039},"labels":[],"label_agreement":null},{"id":"W2002977992","doi":"10.1075/term.19.1.07ber","title":"Review of Andersen (2012): Exploring Newspaper Language: Using the web to create and investigate a large corpus of modern Norwegian","year":2013,"lang":"en","type":"article","venue":"Terminology International Journal of Theoretical and Applied Issues in Specialized Communication","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Norwegian; Newspaper; Computer science; Web page; Linguistics; Corpus linguistics; World Wide Web; Natural language processing; Media studies; Sociology; Philosophy","score_opus":0.02705234104219418,"score_gpt":0.32150602135134926,"score_spread":0.2944536803091551,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2002977992","genre_codex":"review","genre_gemma":"commentary","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"commentary","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0005162345,0.9824835,0.0019498452,0.0071391314,0.0015458924,0.000047159443,0.00087576295,0.00004891443,0.005393518],"genre_scores_gemma":[0.00395251,0.9833929,0.0035877738,0.003188723,0.0010210508,0.00010243918,0.0012953317,0.00008143768,0.003377846],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9977763,0.00081777346,0.00030705924,0.00030709585,0.0007123712,0.000079264304],"domain_scores_gemma":[0.98197824,0.011972752,0.0006984989,0.00051595917,0.004303943,0.0005305734],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0062549612,0.00097426755,0.00136482,0.01741824,0.0010603559,0.0033058848,0.0012864107,0.0011733918,0.0059909574],"category_scores_gemma":[0.020138534,0.00054518855,0.0005173664,0.02122658,0.0026820404,0.0051079346,0.0025893028,0.0018069312,0.0052888216],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006004125,0.00003890117,0.0017116831,0.027398832,0.0001575036,0.00022990924,0.002626757,0.00028278824,0.0010962966,0.008215229,0.27742815,0.6807539],"study_design_scores_gemma":[0.000004170861,0.000015139889,0.0045911474,0.009979937,0.00007805117,0.00024117922,0.00077251287,0.000045452965,0.0002727662,0.0012619667,0.9827116,0.000026131802],"about_ca_topic_score_codex":0.02373279,"about_ca_topic_score_gemma":0.06132232,"teacher_disagreement_score":0.02373279,"about_ca_system_score_codex":0.0025012076,"about_ca_system_score_gemma":0.0063321763,"threshold_uncertainty_score":0.047189295},"labels":[],"label_agreement":null},{"id":"W2003165214","doi":"10.1353/lan.2005.0105","title":"<b>Computers and translation</b> : A translator’s guide. Ed. by Harold Somers. Amsterdam: John Benjamins, 2003. Pp. xvi, 351. ISBN 1588113779. $115 (Hb).","year":2005,"lang":"en","type":"article","venue":"Language","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Terminology; Computer science; Computer-assisted translation; Machine translation; Translation (biology); Translation studies; Linguistics; Artificial intelligence; Natural language processing; Library science; Philosophy","score_opus":0.006120431128672434,"score_gpt":0.24585417169522059,"score_spread":0.23973374056654814,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2003165214","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00058914913,0.6007471,0.016592972,0.032746036,0.018951936,0.00036917548,0.0023922562,0.0025553838,0.32505602],"genre_scores_gemma":[0.0034729552,0.3150011,0.01757343,0.008634059,0.0037391463,0.00051630003,0.0028542352,0.0023802065,0.6458286],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99808335,0.00043452813,0.00021048391,0.000328307,0.0008408082,0.00010256101],"domain_scores_gemma":[0.9972011,0.0012253749,0.00022770886,0.00019128727,0.0009155941,0.00023893424],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020228866,0.0028814839,0.002079003,0.0032487726,0.0018617096,0.0072578527,0.0018858353,0.0036773838,0.19089016],"category_scores_gemma":[0.006038749,0.0018946003,0.0010490542,0.0057968516,0.0020566883,0.010456211,0.0027896105,0.007709722,0.22008248],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000017392944,0.00001786463,0.000050775343,0.00093769754,0.000007816144,0.00006786164,0.0006469104,0.00010833881,0.0004410982,0.0044761337,0.862832,0.1303961],"study_design_scores_gemma":[0.0000027012723,0.000009749062,0.0001428159,0.00069001183,0.0000027466385,0.00017300117,0.00021438736,0.0000316026,0.00008846253,0.0014799663,0.99715376,0.000010740795],"about_ca_topic_score_codex":0.006807429,"about_ca_topic_score_gemma":0.009974473,"teacher_disagreement_score":0.19089016,"about_ca_system_score_codex":0.0017741278,"about_ca_system_score_gemma":0.004295921,"threshold_uncertainty_score":0.6385912},"labels":[],"label_agreement":null},{"id":"W2004156982","doi":"10.3115/1119176.1119189","title":"Confidence estimation for translation prediction","year":2003,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":53,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Correctness; Computer science; Perceptron; Artificial intelligence; Machine translation; Translation (biology); Machine learning; Context (archaeology); Artificial neural network; Estimation; Task (project management); Algorithm; Engineering","score_opus":0.022550888077476256,"score_gpt":0.28888168094738037,"score_spread":0.26633079286990413,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2004156982","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05639526,0.0007637513,0.93809766,0.0006375368,0.00005998638,0.000049701204,0.00022786694,0.002301673,0.0014664925],"genre_scores_gemma":[0.87954974,0.00026840522,0.11816895,0.00019877915,0.00012276362,0.00007843044,0.0007494049,0.00021788561,0.0006455658],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9922725,0.0042908536,0.00051116373,0.0010134033,0.0015842507,0.00032793096],"domain_scores_gemma":[0.8870212,0.097584724,0.004763342,0.004236457,0.0057722065,0.0006220819],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012147553,0.0011791307,0.001501198,0.0024742198,0.0007460688,0.0025390922,0.0017122844,0.0019691554,0.0026123433],"category_scores_gemma":[0.11508086,0.00061177515,0.0008755052,0.0017979309,0.0012201065,0.0051732888,0.0020051731,0.0038596804,0.0010092859],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014922926,0.00019777151,0.0211356,0.0004982731,0.00040422296,0.00020629945,0.00049832824,0.525589,0.0070313565,0.025823066,0.0037720292,0.41335192],"study_design_scores_gemma":[0.000020172482,0.00004226606,0.0011608894,0.000025487157,0.000027246113,0.00006326037,0.000019328387,0.982264,0.0031504647,0.012892334,0.00031103296,0.000023492277],"about_ca_topic_score_codex":0.003124884,"about_ca_topic_score_gemma":0.0022168297,"teacher_disagreement_score":0.012147553,"about_ca_system_score_codex":0.0011888733,"about_ca_system_score_gemma":0.0010954593,"threshold_uncertainty_score":0.0642432},"labels":[],"label_agreement":null},{"id":"W2004595560","doi":"10.3166/ria.18.367-381","title":"Techniques de coopération pour la reconnaissance d'écriture en contexte","year":2004,"lang":"fr","type":"article","venue":"Revue d intelligence artificielle","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Art","score_opus":0.032759122714290156,"score_gpt":0.30536635653821487,"score_spread":0.2726072338239247,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2004595560","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007280468,0.0005409587,0.9907199,0.00013475392,0.000019496229,0.000047059068,0.000011162904,0.0003475494,0.00089847733],"genre_scores_gemma":[0.14683469,0.00082604727,0.8489208,0.00010088232,0.00007093146,0.00023206131,0.00009512049,0.00012961516,0.0027898578],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9905578,0.004922761,0.0004621528,0.001382177,0.0023338855,0.00034114302],"domain_scores_gemma":[0.9883139,0.0080370065,0.0005391807,0.0016971941,0.0012279323,0.00018491056],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006880375,0.0023149068,0.0013525351,0.0017531519,0.0013644587,0.0029022165,0.0020281235,0.002631102,0.0028100945],"category_scores_gemma":[0.01876824,0.000993777,0.0016552228,0.0014940263,0.0024815898,0.0041017556,0.0031921964,0.0027592606,0.0016584158],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006959338,0.00028647817,0.0024765101,0.00065397605,0.00036816494,0.0007495123,0.006116059,0.081832,0.064197905,0.0871435,0.0041628554,0.751317],"study_design_scores_gemma":[0.00020975439,0.00056305603,0.0033294086,0.00018430752,0.00022887882,0.0023411345,0.0015813505,0.78285676,0.06243252,0.08966991,0.05643482,0.00016807031],"about_ca_topic_score_codex":0.0035819856,"about_ca_topic_score_gemma":0.0032845016,"teacher_disagreement_score":0.006880375,"about_ca_system_score_codex":0.0008498596,"about_ca_system_score_gemma":0.0011284617,"threshold_uncertainty_score":0.036387324},"labels":[],"label_agreement":null},{"id":"W2005055804","doi":"10.7202/004050ar","title":"User Driven Development: METAL as an Integrated Multilingual System","year":2002,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Terminology; Desk; Rule-based machine translation; Lexicon; User interface; Natural language processing; Software engineering; Programming language; Artificial intelligence; Linguistics","score_opus":0.034725775436874357,"score_gpt":0.2769748488540261,"score_spread":0.24224907341715177,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2005055804","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17496821,0.00032586852,0.660689,0.00059172197,0.00009077164,0.00086148176,0.0006255739,0.11957965,0.04226766],"genre_scores_gemma":[0.42433268,0.00023762298,0.52144945,0.00040460005,0.00006653861,0.0006774942,0.0028067913,0.00996863,0.04005617],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99703383,0.0010146938,0.00019286113,0.000615584,0.0009460025,0.00019701713],"domain_scores_gemma":[0.99698144,0.0007994541,0.00010556494,0.00095079513,0.0007935437,0.00036923154],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002947819,0.0005985367,0.00043509292,0.00090450794,0.0004186416,0.0022669728,0.0019669216,0.0008759041,0.0073560495],"category_scores_gemma":[0.005477773,0.00066586316,0.00036899257,0.00051353243,0.0004920895,0.0022780804,0.0036228884,0.000808178,0.0053278296],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002627734,0.0010917836,0.010460696,0.0005813549,0.00016318687,0.0032846301,0.009651575,0.008406909,0.1833929,0.015015849,0.032932173,0.7323912],"study_design_scores_gemma":[0.00084792316,0.002496762,0.010553798,0.00025231723,0.00039735335,0.0055284365,0.0019279998,0.20272629,0.21802521,0.010654452,0.54612005,0.0004693326],"about_ca_topic_score_codex":0.0006536461,"about_ca_topic_score_gemma":0.0007197384,"teacher_disagreement_score":0.0073560495,"about_ca_system_score_codex":0.00047009368,"about_ca_system_score_gemma":0.00086542254,"threshold_uncertainty_score":0.024608433},"labels":[],"label_agreement":null},{"id":"W2005440678","doi":"10.1007/s10590-005-2403-4","title":"The Long-Term Forecast for Weather Bulletin Translation","year":2005,"lang":"en","type":"article","venue":"Machine Translation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; Computer Research Institute of Montréal","funders":"Natural Sciences and Engineering Research Council of Canada; Fonds Québécois de la Recherche sur la Nature et les Technologies","keywords":"Machine translation; Computer science; Phrase; Task (project management); Sentence; Focus (optics); Natural language processing; Term (time); Artificial intelligence; Translation (biology); Rule-based machine translation; Computational linguistics; Artificial neural network; Machine learning; Engineering","score_opus":0.019574542661990844,"score_gpt":0.28103619519751294,"score_spread":0.2614616525355221,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2005440678","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.103102475,0.0034738313,0.65236425,0.008625738,0.0073855706,0.000509668,0.06862,0.049234036,0.106684424],"genre_scores_gemma":[0.52500725,0.0023032706,0.33582813,0.0006536253,0.0009922974,0.0003168089,0.07504172,0.006300353,0.05355654],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9990426,0.00031577802,0.00009389056,0.0002565211,0.0002104711,0.00008081446],"domain_scores_gemma":[0.99775606,0.0007075338,0.00012316693,0.00046550055,0.0008795362,0.00006808102],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010056117,0.0007730615,0.0005257946,0.0016654023,0.0013255974,0.002480092,0.00059219834,0.0009403474,0.03354944],"category_scores_gemma":[0.0058700503,0.0005928354,0.0005515387,0.0020437778,0.00045335657,0.0027124123,0.0012636905,0.0013527431,0.021476598],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007920467,0.00015944545,0.0030117068,0.0009198116,0.00011457796,0.001263115,0.001097011,0.011656643,0.029019497,0.05375557,0.32086608,0.57734436],"study_design_scores_gemma":[0.00025262518,0.0003302205,0.00821831,0.0003425138,0.00028508736,0.0012566275,0.0013709193,0.33724743,0.068648525,0.061043248,0.5208083,0.00019618259],"about_ca_topic_score_codex":0.007959601,"about_ca_topic_score_gemma":0.007262327,"teacher_disagreement_score":0.03354944,"about_ca_system_score_codex":0.0010082733,"about_ca_system_score_gemma":0.0023400423,"threshold_uncertainty_score":0.112234},"labels":[],"label_agreement":null},{"id":"W2005994535","doi":"10.3406/medi.2002.1536","title":"La lemmatisation et l'encodage grammatical permettent-ils de reconnaître l'auteur d'un texte ?","year":2002,"lang":"en","type":"article","venue":"Médiévales","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Linguistic Association","funders":"","keywords":"Lemmatisation; Linguistics; Authorship attribution; Computer science; Artificial intelligence; Point (geometry); Corpus linguistics; Natural language processing; Humanities; Philosophy; Mathematics","score_opus":0.023424281505212072,"score_gpt":0.27066786598253234,"score_spread":0.24724358447732026,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2005994535","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.34484506,0.008320478,0.54742587,0.016964762,0.0013282715,0.0001594429,0.0010898543,0.0017751919,0.07809113],"genre_scores_gemma":[0.8144999,0.0024315438,0.17133966,0.0005036527,0.00037476886,0.000105607345,0.00046423872,0.00073226966,0.009548275],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9971073,0.00165325,0.00021828103,0.00054586236,0.00032460937,0.00015077075],"domain_scores_gemma":[0.9914703,0.004699314,0.0010793522,0.0016952044,0.00093853,0.00011730412],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033256393,0.00046259694,0.00051233434,0.001782171,0.0010641149,0.0058517107,0.0006020175,0.001038592,0.006126456],"category_scores_gemma":[0.016599521,0.0003770061,0.00049772894,0.0027666893,0.005775795,0.009599002,0.0010244655,0.0014053482,0.002819234],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004313104,0.00008752947,0.023039645,0.0006450908,0.00009772853,0.00064326223,0.013903298,0.0014208702,0.033088714,0.50422585,0.00483073,0.41758588],"study_design_scores_gemma":[0.00009083248,0.00025958195,0.049196538,0.00079108815,0.00016698448,0.0037279746,0.021245534,0.024848288,0.056209266,0.48512793,0.35811725,0.00021877365],"about_ca_topic_score_codex":0.0015772247,"about_ca_topic_score_gemma":0.0022282233,"teacher_disagreement_score":0.006126456,"about_ca_system_score_codex":0.0011565165,"about_ca_system_score_gemma":0.0011388579,"threshold_uncertainty_score":0.020495057},"labels":[],"label_agreement":null},{"id":"W2006302088","doi":"10.7202/016072ar","title":"SVO Word Order Errors in English-Arabic Translation","year":2007,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":54,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Word order; Linguistics; Computer science; Competence (human resources); Natural language processing; Arabic; Artificial intelligence; Syntactic structure; Error analysis; Psychology; Syntax; Mathematics; Social psychology","score_opus":0.03198124227858313,"score_gpt":0.28908925919981987,"score_spread":0.25710801692123675,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2006302088","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99516183,0.00021449354,0.001797449,0.0000922593,0.000033240787,0.000040605377,0.0004971544,0.00007229433,0.0020908227],"genre_scores_gemma":[0.9911843,0.0002411148,0.0057744063,0.000032443735,0.000015682032,0.000054597094,0.0011987257,0.000089904206,0.0014087614],"study_design_codex":"design_other","study_design_gemma":"qualitative","domain_scores_codex":[0.9952578,0.0019090202,0.00076806283,0.00057067897,0.001338089,0.00015635918],"domain_scores_gemma":[0.9779515,0.014263619,0.0023389675,0.0015208446,0.003650391,0.00027464278],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022630577,0.0005078803,0.00046619104,0.0022082268,0.00096189725,0.0014494723,0.00031423327,0.00047088432,0.0022648522],"category_scores_gemma":[0.02291818,0.00030238935,0.00017405787,0.002219508,0.00077374093,0.00072117266,0.0011932304,0.0006551123,0.00093402195],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0032315876,0.0009129375,0.28144872,0.002171789,0.00017572401,0.0077328198,0.09960404,0.0072703757,0.09506697,0.008922423,0.008681242,0.48478147],"study_design_scores_gemma":[0.00017977557,0.0011894873,0.6610202,0.0008003308,0.00024222638,0.021216363,0.033428054,0.037921704,0.18123332,0.008227682,0.054283477,0.00025728735],"about_ca_topic_score_codex":0.0022458513,"about_ca_topic_score_gemma":0.0033189543,"teacher_disagreement_score":0.0022648522,"about_ca_system_score_codex":0.00058572803,"about_ca_system_score_gemma":0.0008948922,"threshold_uncertainty_score":0.011968315},"labels":[],"label_agreement":null},{"id":"W2006617528","doi":"10.3115/1626431.1626467","title":"Improving Arabic-Chinese statistical machine translation using English as pivot language","year":2009,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":51,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"National Research Council Canada","keywords":"Machine translation; Computer science; Natural language processing; Phrase; Arabic; Evaluation of machine translation; Artificial intelligence; Machine translation software usability; Sentence; Example-based machine translation; Translation (biology); Transfer-based machine translation; Speech recognition; Linguistics","score_opus":0.009009416888915548,"score_gpt":0.2929979197001801,"score_spread":0.28398850281126453,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2006617528","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2926747,0.0028922262,0.669053,0.0008090435,0.0005873356,0.0003708889,0.0013248468,0.0138434535,0.018444581],"genre_scores_gemma":[0.5204266,0.0009513295,0.46864983,0.0002652543,0.00017235827,0.00015883506,0.002435141,0.0006383569,0.006302382],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.998939,0.00043011134,0.00012739717,0.00014507082,0.00027211625,0.000086275184],"domain_scores_gemma":[0.9976356,0.0007857831,0.00011849586,0.0003350143,0.0010496405,0.00007552382],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016842136,0.001091789,0.00086792273,0.0009580416,0.00079958397,0.0009673703,0.00050579105,0.0003740803,0.0036412317],"category_scores_gemma":[0.0049192603,0.0003024069,0.00039353597,0.0014598492,0.0003484918,0.00086089206,0.0009168871,0.00054989464,0.0030898228],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010956025,0.00030917805,0.008097642,0.0009070542,0.0001990694,0.0006408652,0.0008295362,0.03857573,0.118182905,0.0067528053,0.020258948,0.8041507],"study_design_scores_gemma":[0.0004355875,0.0011495756,0.013253376,0.000107900225,0.0005467301,0.0013651756,0.0005997445,0.4840753,0.43166056,0.006352293,0.060261205,0.00019250085],"about_ca_topic_score_codex":0.0066405404,"about_ca_topic_score_gemma":0.011048246,"teacher_disagreement_score":0.0066405404,"about_ca_system_score_codex":0.00048675018,"about_ca_system_score_gemma":0.0016546919,"threshold_uncertainty_score":0.0132038},"labels":[],"label_agreement":null},{"id":"W2006742608","doi":"10.1111/j.1749-818x.2009.00171.x","title":"Teaching &amp; Learning Guide for: Noun Incorporation: Essentials and Extensions","year":2009,"lang":"en","type":"article","venue":"Language and Linguistics Compass","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Noun; Linguistics; Computer science; Psychology; Natural language processing; Artificial intelligence; Philosophy","score_opus":0.015146895335192529,"score_gpt":0.3118682759763159,"score_spread":0.29672138064112336,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2006742608","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0012319282,0.0062939166,0.061761327,0.021354675,0.008626659,0.0009999025,0.016743058,0.03037923,0.8526092],"genre_scores_gemma":[0.0033030238,0.005235611,0.040559422,0.0055373353,0.0014307911,0.0006755808,0.0050698384,0.007889046,0.9302993],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99946004,0.00010030291,0.000051418843,0.000076853226,0.00026206044,0.00004918343],"domain_scores_gemma":[0.9967784,0.0011020079,0.00013134694,0.0002985398,0.0011573376,0.0005323535],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0011756608,0.0012637374,0.0009949517,0.001726639,0.0009459912,0.0032361462,0.001961414,0.0023136388,0.55139744],"category_scores_gemma":[0.005148386,0.00073712086,0.00087444804,0.0015739489,0.0008215197,0.008119258,0.0029601522,0.0027799597,0.55752176],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000013497245,0.000026665763,0.00006531253,0.000126411,8.7591985e-7,0.000054184216,0.00012783453,0.00004981089,0.00018286286,0.0020030094,0.87415934,0.123190075],"study_design_scores_gemma":[0.0000064468227,0.000008351159,0.00012342512,0.000088978915,6.271659e-7,0.00008711714,0.000078888035,0.000084479434,0.000107734755,0.0031943547,0.99621373,0.0000058827654],"about_ca_topic_score_codex":0.001938491,"about_ca_topic_score_gemma":0.005463035,"teacher_disagreement_score":0.55139744,"about_ca_system_score_codex":0.0009723971,"about_ca_system_score_gemma":0.0021618004,"threshold_uncertainty_score":0.6398771},"labels":[],"label_agreement":null},{"id":"W2006877875","doi":"10.1080/01690960902840279","title":"A computational model of learning semantic roles from child-directed language","year":2009,"lang":"en","type":"article","venue":"Language and Cognitive Processes","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":43,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Natural language processing; Verb; Predicate (mathematical logic); Ambiguity; Artificial intelligence; Semantic role labeling; Comprehension; Language acquisition; Semantics (computer science); Meaning (existential); Probabilistic logic; Event (particle physics); Linguistics; Psychology; Sentence","score_opus":0.006790039251520495,"score_gpt":0.2534377029089888,"score_spread":0.24664766365746832,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2006877875","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04786696,0.00027840282,0.9431205,0.0011552933,0.000025202682,0.00006742189,0.00027776478,0.00051665003,0.0066917953],"genre_scores_gemma":[0.65844536,0.00061980274,0.3320635,0.00030293723,0.000044003315,0.00046844478,0.00067012117,0.00014459743,0.007241185],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994178,0.00024137968,0.000028521941,0.0001417898,0.00011271904,0.00005793233],"domain_scores_gemma":[0.997837,0.0016096593,0.00014625277,0.0001802763,0.0001300499,0.00009680649],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014988675,0.0005629092,0.00057831354,0.0007934482,0.0005207026,0.0015681947,0.0024067138,0.0013323685,0.0033020552],"category_scores_gemma":[0.0053191385,0.0008677139,0.0014223665,0.0007610193,0.0022512292,0.0046422095,0.0013123632,0.0019500646,0.0005193972],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012820658,0.0001336177,0.0029506474,0.00022536087,0.00010245704,0.0006347887,0.001280238,0.32415956,0.0030542824,0.6176138,0.0020684074,0.04764863],"study_design_scores_gemma":[0.000027664466,0.000039259186,0.00035414603,0.00001699891,0.000024941499,0.00021249177,0.000040012812,0.74134856,0.0007278065,0.25514787,0.00203946,0.000020743848],"about_ca_topic_score_codex":0.0058500743,"about_ca_topic_score_gemma":0.0079991985,"teacher_disagreement_score":0.0058500743,"about_ca_system_score_codex":0.0018019952,"about_ca_system_score_gemma":0.0013660874,"threshold_uncertainty_score":0.013074458},"labels":[],"label_agreement":null},{"id":"W2009287484","doi":"10.11139/cj.28.3.662-676","title":"Computer Assisted Reading in German as a Foreign Language, Developing and Testing an NLP-based Application","year":2011,"lang":"en","type":"article","venue":"CALICO Journal","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Saskatchewan","funders":"","keywords":"German; Natural language processing; Computer science; Artificial intelligence; Reading (process); Linguistics; Foreign language; Philosophy","score_opus":0.043046795933450024,"score_gpt":0.3159701930859908,"score_spread":0.27292339715254077,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2009287484","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9435579,0.00013643011,0.03912522,0.00020397977,0.000039029663,0.0008825841,0.0010352301,0.005863384,0.0091562625],"genre_scores_gemma":[0.8088714,0.00026280258,0.17080626,0.00023534142,0.00001995958,0.0008880998,0.0028487847,0.00068584405,0.015381479],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9990404,0.0004493238,0.000068941954,0.00023516134,0.00012908484,0.00007697313],"domain_scores_gemma":[0.9958813,0.0031754354,0.00008168055,0.00036680748,0.00028827164,0.00020654667],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018245174,0.0008305166,0.0005338943,0.0007642479,0.0005356659,0.0011569214,0.0010785891,0.0011736071,0.005988431],"category_scores_gemma":[0.005651638,0.00031857094,0.0003706266,0.0006082614,0.0005355324,0.002052952,0.0012651737,0.0005417017,0.0017960097],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0031763883,0.0067435997,0.014541997,0.0017183785,0.00012227,0.0043569403,0.03310992,0.014149884,0.14564498,0.006809512,0.023862101,0.7457641],"study_design_scores_gemma":[0.0020879058,0.022296984,0.069304,0.00044660806,0.00041282322,0.00751184,0.02648998,0.21301737,0.3962905,0.0121080205,0.24940518,0.00062881],"about_ca_topic_score_codex":0.0016872503,"about_ca_topic_score_gemma":0.0025473626,"teacher_disagreement_score":0.005988431,"about_ca_system_score_codex":0.0004387372,"about_ca_system_score_gemma":0.00055961974,"threshold_uncertainty_score":0.02003324},"labels":[],"label_agreement":null},{"id":"W2009305298","doi":"10.1007/s11168-005-1286-0","title":"A Computational Algebraic Approach to Latin Grammar","year":2005,"lang":"en","type":"article","venue":"Research on Language and Computation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Syntax; Grammar; Computer science; Computational linguistics; Linguistics; Type (biology); Natural language processing; Coherence (philosophical gambling strategy); Artificial intelligence; Programming language; Mathematics; Philosophy","score_opus":0.049979300898794446,"score_gpt":0.38452348004026937,"score_spread":0.33454417914147494,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2009305298","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022762647,0.0017900016,0.88094884,0.0073110047,0.00040065736,0.000056280143,0.00031274412,0.00052342453,0.085894436],"genre_scores_gemma":[0.53389376,0.0030166202,0.431718,0.001598219,0.0015294246,0.00025410036,0.0008088302,0.0004910878,0.026689874],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9990651,0.0004369533,0.000058803394,0.00016635518,0.00020315041,0.00006959263],"domain_scores_gemma":[0.9983974,0.0009811049,0.00009761831,0.00021698917,0.00021882476,0.000087990855],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010207633,0.0005646292,0.00088540453,0.0022130737,0.0025579007,0.0045592785,0.0021062123,0.001009257,0.011718524],"category_scores_gemma":[0.004404776,0.0006496753,0.0018368852,0.0029843224,0.0080611035,0.011645033,0.0027720963,0.0033139628,0.00142186],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000029142798,0.0000041032845,0.000039497136,0.000012130808,0.0000022189752,0.000009803209,0.000089012756,0.00065948337,0.00005314495,0.99654984,0.00047340666,0.0021042854],"study_design_scores_gemma":[0.000003834656,0.0000024247554,0.000022862812,0.000005488158,0.000002737786,0.000014224573,0.000032809046,0.0029176536,0.000042201602,0.99283993,0.004111795,0.0000041531857],"about_ca_topic_score_codex":0.004514899,"about_ca_topic_score_gemma":0.004862281,"teacher_disagreement_score":0.011718524,"about_ca_system_score_codex":0.00253252,"about_ca_system_score_gemma":0.0014333957,"threshold_uncertainty_score":0.039202332},"labels":[],"label_agreement":null},{"id":"W2009433545","doi":"10.1145/1183614.1183746","title":"Improving query translation with confidence estimation for cross language information retrieval","year":2006,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Cross-language information retrieval; Translation (biology); Query language; Artificial intelligence; Query expansion; Natural language processing; Information retrieval; Machine translation","score_opus":0.006574963776798162,"score_gpt":0.2654325345671096,"score_spread":0.2588575707903114,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2009433545","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020710103,0.002199591,0.96671283,0.00043190565,0.00016195784,0.00010514824,0.00038194793,0.007672826,0.0016236805],"genre_scores_gemma":[0.3572962,0.0011927851,0.63495964,0.000250016,0.000519181,0.00021500592,0.0024392363,0.00082466024,0.0023034068],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99545884,0.0023357521,0.0003345308,0.00048114677,0.0011529806,0.00023673355],"domain_scores_gemma":[0.98403513,0.010324324,0.00077061757,0.0022356317,0.0024098651,0.00022456876],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0042285686,0.0012636894,0.0021411586,0.002966265,0.0007327093,0.0021171889,0.0016606719,0.001413262,0.006911754],"category_scores_gemma":[0.029881578,0.0005251184,0.0009500898,0.0030649973,0.00066722353,0.0041937004,0.0023249867,0.001728013,0.004515542],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008927669,0.00024798478,0.0018014669,0.00056145235,0.00022369716,0.00016361306,0.00017492725,0.034525212,0.027349805,0.0047783796,0.016121542,0.9131591],"study_design_scores_gemma":[0.00013734667,0.0002619179,0.0015294703,0.000047382837,0.00017981417,0.000318233,0.000084178006,0.95193905,0.028976552,0.012328336,0.0041301255,0.00006760571],"about_ca_topic_score_codex":0.0024938092,"about_ca_topic_score_gemma":0.0022112033,"teacher_disagreement_score":0.006911754,"about_ca_system_score_codex":0.0004824267,"about_ca_system_score_gemma":0.0009274247,"threshold_uncertainty_score":0.023122132},"labels":[],"label_agreement":null},{"id":"W2010044725","doi":"10.3917/docsi.421.0012","title":"Scénarios de production pour l'indexation d'images animées","year":2005,"lang":"fr","type":"article","venue":"Documentaliste-Sciences de l Information","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Indexation; Humanities; Philosophy; Economics","score_opus":0.025203196164920585,"score_gpt":0.3218724016325294,"score_spread":0.29666920546760883,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2010044725","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2537801,0.001417168,0.6865678,0.0013609522,0.00033859885,0.0012175717,0.0024555756,0.010964751,0.041897528],"genre_scores_gemma":[0.4212342,0.000994563,0.54652745,0.00019670885,0.00010990561,0.0005786967,0.0029263238,0.0012994489,0.026132634],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9988146,0.00038649907,0.00008876754,0.00022264649,0.00041811325,0.00006935398],"domain_scores_gemma":[0.9944195,0.0041663046,0.0001784327,0.0005494702,0.0005007759,0.00018560453],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015378571,0.00094720104,0.00049403287,0.0011199836,0.00095063343,0.00257551,0.0008838136,0.0015223426,0.017581116],"category_scores_gemma":[0.0073765125,0.00041944475,0.00066086557,0.00082932523,0.0007211077,0.0026403842,0.0017341231,0.00085692276,0.004115644],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021940405,0.00047202435,0.00467457,0.0022127652,0.00012135069,0.005133715,0.011033007,0.015194355,0.40521887,0.02519428,0.016154027,0.512397],"study_design_scores_gemma":[0.00030707818,0.00152308,0.011940775,0.00064262084,0.00020962063,0.005615507,0.006880358,0.14895564,0.4100408,0.018932743,0.39469993,0.00025179796],"about_ca_topic_score_codex":0.0016733954,"about_ca_topic_score_gemma":0.001775772,"teacher_disagreement_score":0.017581116,"about_ca_system_score_codex":0.00048616133,"about_ca_system_score_gemma":0.00045937137,"threshold_uncertainty_score":0.058814645},"labels":[],"label_agreement":null},{"id":"W2010177639","doi":"10.1353/ils.2011.0022","title":"Bilingual Document Clustering: Evaluating Cognates as Features / Le groupage de documents bilingues : l’évaluation des cognats comme caractéristiques","year":2011,"lang":"fr","type":"article","venue":"Canadian Journal of Information and Library Science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Valuation (finance); Linguistics; Cluster analysis; Computer science; Philosophy; Artificial intelligence; Business; Accounting","score_opus":0.05025219599413446,"score_gpt":0.3097175813893523,"score_spread":0.25946538539521785,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2010177639","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.89260393,0.004043379,0.08626672,0.0002866256,0.00017217074,0.0004019842,0.0015562387,0.0037295965,0.010939331],"genre_scores_gemma":[0.9084615,0.0005059153,0.08285001,0.00006907374,0.00007543924,0.0001290966,0.005239176,0.00025021945,0.0024195425],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99736017,0.00096399407,0.00021232883,0.0007017692,0.0005660682,0.00019566338],"domain_scores_gemma":[0.99625033,0.0015465404,0.00031267444,0.00046014017,0.0010013543,0.00042892483],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033238714,0.0014210505,0.0010873864,0.00510824,0.0013642163,0.0020337552,0.00081159367,0.0015112924,0.0022411721],"category_scores_gemma":[0.00752059,0.0002292268,0.00089443073,0.0029821869,0.00057967636,0.002272614,0.0013913739,0.00051376695,0.0012845778],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.009265416,0.001119271,0.050316717,0.0013049962,0.0012677416,0.00033446425,0.0011176455,0.03418056,0.042619362,0.0018279623,0.007969343,0.8486766],"study_design_scores_gemma":[0.00080810336,0.0040467177,0.12891881,0.00026907676,0.0017246567,0.0018821921,0.0040558707,0.7214351,0.111877166,0.005183317,0.019485533,0.00031346077],"about_ca_topic_score_codex":0.009651128,"about_ca_topic_score_gemma":0.011107024,"teacher_disagreement_score":0.009651128,"about_ca_system_score_codex":0.0010042116,"about_ca_system_score_gemma":0.0014806237,"threshold_uncertainty_score":0.019189894},"labels":[],"label_agreement":null},{"id":"W2010707932","doi":"10.7202/003228ar","title":"La bi-textualité : vers une nouvelle génération d’aides à la traduction et la terminologie","year":2002,"lang":"fr","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Philosophy; Humanities","score_opus":0.06360170631839047,"score_gpt":0.3037331938894501,"score_spread":0.24013148757105962,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2010707932","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024075216,0.001336859,0.9348678,0.0021077124,0.00045952515,0.00023808407,0.0009672402,0.0065637073,0.02938388],"genre_scores_gemma":[0.1964195,0.0011310948,0.7266152,0.000853009,0.00028008723,0.0003725241,0.002646781,0.005236855,0.066445015],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9943528,0.0020047445,0.00042554733,0.001296607,0.0017033032,0.0002170795],"domain_scores_gemma":[0.98417735,0.0065812534,0.0008923669,0.0043853354,0.003619051,0.00034467192],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0056337235,0.0011602703,0.0008043568,0.0020131366,0.0015930511,0.0072173416,0.0020495611,0.0019926832,0.015510418],"category_scores_gemma":[0.018238258,0.000856446,0.0012148821,0.001787028,0.004519992,0.008343926,0.003412111,0.0033557091,0.010941723],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009026687,0.00012740387,0.0038748728,0.0016723071,0.00011576759,0.0010317717,0.017663043,0.006861224,0.08066782,0.38982502,0.020846535,0.4764115],"study_design_scores_gemma":[0.00007996208,0.0002848061,0.0028966742,0.0007905161,0.00014496382,0.0016566183,0.003302377,0.03925011,0.08981328,0.16390538,0.69765204,0.00022331356],"about_ca_topic_score_codex":0.004156532,"about_ca_topic_score_gemma":0.0031730733,"teacher_disagreement_score":0.015510418,"about_ca_system_score_codex":0.0021070032,"about_ca_system_score_gemma":0.0025901108,"threshold_uncertainty_score":0.051887512},"labels":[],"label_agreement":null},{"id":"W2010856309","doi":"10.1155/2012/484580","title":"Learning to Translate: A Statistical and Computational Analysis","year":2012,"lang":"en","type":"article","venue":"Advances in Artificial Intelligence","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"Engineering and Physical Sciences Research Council","keywords":"Computer science; Phrase; Zipf's law; Machine translation; Algorithmic learning theory; Artificial intelligence; Inference; Natural language processing; Statistical inference; Machine learning; Point (geometry); Active learning (machine learning); Mathematics","score_opus":0.02152639447210036,"score_gpt":0.36337125332381126,"score_spread":0.3418448588517109,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2010856309","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20831327,0.0026215657,0.76447386,0.0045712544,0.00015541456,0.00025813296,0.0014378495,0.0017692622,0.016399441],"genre_scores_gemma":[0.7407311,0.0011330958,0.25114012,0.00044317753,0.0002590361,0.00062937936,0.002437994,0.0004523306,0.0027737743],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9960427,0.0017743564,0.0002236228,0.00047762442,0.0013066368,0.00017508418],"domain_scores_gemma":[0.91430503,0.07552568,0.0016780251,0.0057171197,0.0024928537,0.00028123386],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0064859735,0.000740149,0.0009898499,0.0023834682,0.0009678219,0.0021848183,0.0012410512,0.00096130173,0.0061095837],"category_scores_gemma":[0.0625427,0.00042754912,0.00086241093,0.003229137,0.001672622,0.0053737457,0.0017482926,0.0018950297,0.0011333427],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004744181,0.0005839247,0.016435096,0.0006865155,0.00018997285,0.00030739143,0.00046997456,0.43190026,0.008130557,0.17496061,0.011549743,0.35431153],"study_design_scores_gemma":[0.000014473836,0.00008904805,0.0027465576,0.00001996443,0.000018300658,0.00015193578,0.00007799226,0.93632877,0.0031969892,0.055687092,0.0016468441,0.000022019423],"about_ca_topic_score_codex":0.0028414193,"about_ca_topic_score_gemma":0.0029282072,"teacher_disagreement_score":0.0064859735,"about_ca_system_score_codex":0.0014868022,"about_ca_system_score_gemma":0.0014893586,"threshold_uncertainty_score":0.03430152},"labels":[],"label_agreement":null},{"id":"W2011189381","doi":"10.1145/2093346.2093351","title":"Report on the first summer school on NLP and IR in Beijing","year":2012,"lang":"en","type":"article","venue":"ACM SIGIR Forum","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Beijing; Computer science; Event (particle physics); Natural language processing; Artificial intelligence; China; History","score_opus":0.020369354272696084,"score_gpt":0.28229319489464927,"score_spread":0.2619238406219532,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2011189381","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.050748512,0.038589764,0.022871148,0.17527089,0.09906211,0.006020371,0.024650102,0.0057054656,0.5770816],"genre_scores_gemma":[0.0396268,0.006952745,0.0045486987,0.005226189,0.0065164273,0.0011342582,0.011735175,0.00123635,0.9230233],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99345124,0.001086547,0.00027596942,0.0008159337,0.0029412408,0.0014290736],"domain_scores_gemma":[0.9847652,0.0007373154,0.0003161778,0.001268997,0.0057036104,0.007208667],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0120089855,0.0018189118,0.0017273559,0.0036894458,0.0051922537,0.0052653956,0.0022668422,0.0033778418,0.1671713],"category_scores_gemma":[0.0063561304,0.0007892445,0.0013276063,0.0038391615,0.0013160958,0.0046030898,0.007632564,0.0038960585,0.087707974],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003170031,0.0005346345,0.00285174,0.0002433393,0.000042466876,0.00033030263,0.0005185016,0.00027037069,0.0023347968,0.0021652079,0.89743674,0.09295483],"study_design_scores_gemma":[0.00007318156,0.00027236805,0.017002052,0.00007843811,0.000019217283,0.00007301307,0.00063777313,0.00018752391,0.0014926142,0.0010448203,0.97909284,0.000026232692],"about_ca_topic_score_codex":0.024310406,"about_ca_topic_score_gemma":0.041707747,"teacher_disagreement_score":0.1671713,"about_ca_system_score_codex":0.0048597273,"about_ca_system_score_gemma":0.012064596,"threshold_uncertainty_score":0.5592437},"labels":[],"label_agreement":null},{"id":"W2011536487","doi":"10.3115/1072133.1072188","title":"Information extraction with term frequencies","year":2001,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Information retrieval; Question answering; Information extraction; The Internet; Component (thermodynamics); Parsing; Term (time); Premise; Process (computing); Web page; Subject (documents); Element (criminal law); World Wide Web; Natural language processing; Programming language","score_opus":0.009078382317059852,"score_gpt":0.2535023023446993,"score_spread":0.24442392002763944,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2011536487","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011472777,0.0038960113,0.95808256,0.00065141177,0.0005415502,0.0006626349,0.0087457225,0.007830383,0.008116943],"genre_scores_gemma":[0.09907386,0.0031012027,0.8711926,0.00022259176,0.0007766786,0.00094715285,0.017790155,0.00063353794,0.0062622125],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9958026,0.0010662305,0.00063292496,0.00090795645,0.0013267492,0.00026361953],"domain_scores_gemma":[0.9913846,0.0057010585,0.0004942331,0.0011287311,0.0011720012,0.00011935205],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002750366,0.0017016513,0.0017165558,0.028745761,0.0015812004,0.004232956,0.0016139445,0.0014978196,0.011318502],"category_scores_gemma":[0.020194182,0.000830447,0.0026341858,0.02207026,0.00068683055,0.0057792463,0.0021886607,0.0018641931,0.011187242],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00030767245,0.00017693773,0.0029717542,0.0010073252,0.0003540846,0.00041784096,0.0002889114,0.003924186,0.012251773,0.016748296,0.016881915,0.94466937],"study_design_scores_gemma":[0.00035246983,0.0006692658,0.018049605,0.0013035696,0.0020398195,0.005164044,0.0011663769,0.2273402,0.09205372,0.33726397,0.31391415,0.0006828539],"about_ca_topic_score_codex":0.002103385,"about_ca_topic_score_gemma":0.0016521999,"teacher_disagreement_score":0.028745761,"about_ca_system_score_codex":0.0006869988,"about_ca_system_score_gemma":0.0015843636,"threshold_uncertainty_score":0.03786415},"labels":[],"label_agreement":null},{"id":"W2011612519","doi":"10.3115/1220355.1220361","title":"Improved word alignment using a symmetric lexicon model","year":2004,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Lexicon; Natural language processing; Artificial intelligence; Word (group theory); Task (project management); Speech recognition; German; Linguistics","score_opus":0.024820219888318992,"score_gpt":0.2825969827975287,"score_spread":0.2577767629092097,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2011612519","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02366194,0.00029572332,0.9639385,0.0002620016,0.00012830656,0.000093032715,0.00053763826,0.008050862,0.0030319863],"genre_scores_gemma":[0.32691887,0.0004948781,0.65101904,0.00046673213,0.00020131278,0.00039144087,0.0066326675,0.0024366772,0.011438374],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9982693,0.0005913384,0.00016080569,0.00048692134,0.00037985013,0.00011182878],"domain_scores_gemma":[0.99770063,0.0007494455,0.0001694823,0.0007786243,0.0005460807,0.00005579826],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015368485,0.0012517418,0.0016224092,0.0020971808,0.0007266859,0.0018030968,0.0016543509,0.0012592252,0.0048207776],"category_scores_gemma":[0.004868979,0.00093656156,0.0015730021,0.0026024487,0.000623004,0.0045813955,0.0018272862,0.0012654184,0.007233582],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00052903884,0.00035996622,0.002777762,0.00038811835,0.00036550514,0.0005045445,0.00046893713,0.20287053,0.069979206,0.05567346,0.026383637,0.6396993],"study_design_scores_gemma":[0.000060828825,0.000060326678,0.00055694894,0.000011734039,0.000055206507,0.00014031952,0.000041199608,0.9517967,0.008561371,0.032847654,0.0058254586,0.000042257067],"about_ca_topic_score_codex":0.008051714,"about_ca_topic_score_gemma":0.01706643,"teacher_disagreement_score":0.008051714,"about_ca_system_score_codex":0.00088346243,"about_ca_system_score_gemma":0.0025830101,"threshold_uncertainty_score":0.01612705},"labels":[],"label_agreement":null},{"id":"W2011978665","doi":"10.7202/003822ar","title":"New Trends in Machine Translation","year":2002,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Annotation; Corpus linguistics; Computer science; Perspective (graphical); Machine translation; Natural language processing; Artificial intelligence; Translation (biology); Linguistics; Data science; Philosophy","score_opus":0.048815006783993746,"score_gpt":0.2849067471516878,"score_spread":0.23609174036769404,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2011978665","genre_codex":"review","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004943797,0.55712456,0.09888902,0.209289,0.012415335,0.00007450124,0.00045589084,0.0012308366,0.11557698],"genre_scores_gemma":[0.099537306,0.5792264,0.14869857,0.04795246,0.060875256,0.00042548295,0.0012311673,0.0010321902,0.061021224],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9953473,0.001874306,0.00031763563,0.00082599843,0.0014157215,0.00021904832],"domain_scores_gemma":[0.9888732,0.0073615094,0.00035912596,0.0011577734,0.0017638699,0.00048464854],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00687036,0.00086213404,0.0013047869,0.004508787,0.0015190187,0.007924805,0.0020349177,0.004797616,0.022583265],"category_scores_gemma":[0.012981404,0.0005017914,0.00082329666,0.0059493063,0.0067427955,0.023065917,0.0031087142,0.0058180615,0.010404765],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010175309,0.00007479519,0.0006748672,0.0016800112,0.0000482514,0.00015314447,0.00078956876,0.0007479587,0.0010892007,0.52205265,0.064986266,0.40760157],"study_design_scores_gemma":[0.000016859398,0.00005299556,0.00053244625,0.0005935083,0.000017863988,0.00032824406,0.0005470643,0.0021865217,0.00037523272,0.2843417,0.71097803,0.000029571449],"about_ca_topic_score_codex":0.0011809444,"about_ca_topic_score_gemma":0.0014302445,"teacher_disagreement_score":0.022583265,"about_ca_system_score_codex":0.0025145067,"about_ca_system_score_gemma":0.0022917772,"threshold_uncertainty_score":0.07554859},"labels":[],"label_agreement":null},{"id":"W2012182979","doi":"10.1177/00238309040470010401","title":"Probability in the Grammar of German and Dutch: Interfixation in Triconstituent Compounds","year":2004,"lang":"en","type":"article","venue":"Language and Speech","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; University of Alberta","funders":"","keywords":"German; Linguistics; Probabilistic logic; Grammar; Psychology; Competence (human resources); Natural language processing; Artificial intelligence; Mathematics; Computer science; Social psychology; Philosophy","score_opus":0.015122864627221552,"score_gpt":0.27528773070691176,"score_spread":0.2601648660796902,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2012182979","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9382528,0.00043018628,0.047517836,0.00029725075,0.000011587282,0.000025579751,0.00039494017,0.00009928312,0.012970478],"genre_scores_gemma":[0.99440974,0.00008679161,0.0046192617,0.0000168131,0.000004331051,0.000008648387,0.00017137572,0.000044419252,0.00063853076],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99929225,0.00022939865,0.000084592175,0.0001557692,0.0001945541,0.00004343499],"domain_scores_gemma":[0.99789953,0.0012771729,0.00047079296,0.00008497434,0.00021348249,0.000053948937],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007469495,0.0002049676,0.00021685645,0.00096189353,0.0005025321,0.0015758586,0.00031633404,0.00026096977,0.0021563075],"category_scores_gemma":[0.0034278897,0.00023092116,0.00022620239,0.0010688896,0.0013172809,0.002229362,0.00064749207,0.00042328934,0.00024188295],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00060168083,0.00010700787,0.1252368,0.0006855786,0.000193847,0.002661897,0.027407492,0.009673279,0.15285411,0.5271844,0.0019806628,0.15141325],"study_design_scores_gemma":[0.000105678635,0.00029591142,0.4510345,0.00013108566,0.00023482337,0.008017594,0.011257885,0.0734341,0.06058399,0.3308529,0.0637713,0.0002803104],"about_ca_topic_score_codex":0.0057914075,"about_ca_topic_score_gemma":0.0076640253,"teacher_disagreement_score":0.0057914075,"about_ca_system_score_codex":0.0009991104,"about_ca_system_score_gemma":0.00043228187,"threshold_uncertainty_score":0.011515379},"labels":[],"label_agreement":null},{"id":"W2012480537","doi":"10.7202/019921ar","title":"Terminology and Translation — bringing research and professional training together through technology","year":2009,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Terminology; Computer science; Term (time); Domain (mathematical analysis); Suite; Emphasis (telecommunications); Linguistics; Natural language processing; Artificial intelligence; Political science; Mathematics","score_opus":0.12470142698100803,"score_gpt":0.3799404962396917,"score_spread":0.25523906925868367,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2012480537","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011833023,0.013088579,0.7622298,0.06266932,0.0057126083,0.0005614535,0.00027656578,0.0048384997,0.13879015],"genre_scores_gemma":[0.10701995,0.0100527415,0.824773,0.01256749,0.005108888,0.0007605874,0.00065266027,0.0025902102,0.036474552],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9593765,0.025494909,0.003277255,0.0038526808,0.006984382,0.0010141807],"domain_scores_gemma":[0.9033713,0.055910904,0.0033378005,0.025891282,0.007869532,0.0036192131],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.042529125,0.0012296946,0.0013508275,0.007099557,0.00393614,0.016968703,0.0032864646,0.0037942522,0.014281107],"category_scores_gemma":[0.0837009,0.0010576602,0.00095762464,0.0056135994,0.016234705,0.030079067,0.020328706,0.005229489,0.010017093],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000089364636,0.00014394765,0.0012826096,0.0011540554,0.000060160786,0.00038294628,0.02543617,0.00035711133,0.0050763525,0.22586396,0.047302462,0.6928508],"study_design_scores_gemma":[0.00004637787,0.00014704838,0.0012130567,0.0010376789,0.00003572843,0.0008482949,0.009712615,0.0010356832,0.0033622752,0.3286002,0.6538658,0.00009513598],"about_ca_topic_score_codex":0.0010334483,"about_ca_topic_score_gemma":0.0010312953,"teacher_disagreement_score":0.042529125,"about_ca_system_score_codex":0.0025429332,"about_ca_system_score_gemma":0.009589795,"threshold_uncertainty_score":0.22491819},"labels":[],"label_agreement":null},{"id":"W2012932862","doi":"10.3115/1708087.1708090","title":"Work-in-progress project report","year":2004,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Protocol (science); IBM; Computer science; Task (project management); Reliability (semiconductor); Object (grammar); Software engineering; Multimedia; Engineering; Systems engineering; Artificial intelligence","score_opus":0.016323052254838856,"score_gpt":0.309056049303254,"score_spread":0.2927329970484151,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2012932862","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03222866,0.007165469,0.1797175,0.013752545,0.008733304,0.017674861,0.20643845,0.032812558,0.50147665],"genre_scores_gemma":[0.06018601,0.0029620514,0.12488052,0.0020177988,0.0013298732,0.010071265,0.48780635,0.011941787,0.29880428],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.97197735,0.00817504,0.0016773395,0.0031161385,0.012323991,0.0027301544],"domain_scores_gemma":[0.95988744,0.0036224627,0.0012397195,0.008950454,0.01977391,0.0065260883],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.028469093,0.0026409659,0.0017571445,0.00508512,0.003211512,0.009571073,0.0043630903,0.0036848811,0.14018975],"category_scores_gemma":[0.03084863,0.0008386892,0.0014781057,0.00396429,0.0010306917,0.0057202573,0.006988511,0.002481585,0.14574523],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001466924,0.0018829921,0.003998434,0.0015505402,0.00014067537,0.0005728375,0.0014300391,0.0019380662,0.013323421,0.013393435,0.6596065,0.3006961],"study_design_scores_gemma":[0.00019597258,0.0006410875,0.0030948885,0.00025896446,0.000082293045,0.00026340556,0.0005732336,0.0011038865,0.0109727625,0.0027005493,0.98004675,0.00006627963],"about_ca_topic_score_codex":0.009721863,"about_ca_topic_score_gemma":0.0056392555,"teacher_disagreement_score":0.14018975,"about_ca_system_score_codex":0.0032724116,"about_ca_system_score_gemma":0.0153998425,"threshold_uncertainty_score":0.46898144},"labels":[],"label_agreement":null},{"id":"W2013138719","doi":"10.1145/572020.572056","title":"A character-level error analysis technique for evaluating text entry methods","year":2002,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":147,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Weighting; Character (mathematics); Computer science; Set (abstract data type); Sequence (biology); Character encoding; Error analysis; Algorithm; Data mining; Artificial intelligence; Mathematics","score_opus":0.12161983069548121,"score_gpt":0.4292188861413062,"score_spread":0.307599055445825,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2013138719","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.032910224,0.00047739455,0.95330983,0.00011357924,0.00018982109,0.0012556906,0.0022301332,0.006524261,0.0029889704],"genre_scores_gemma":[0.0986497,0.00013971106,0.89362746,0.000075691154,0.00010593401,0.001631784,0.002655715,0.0012011116,0.0019128307],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9567963,0.012404703,0.006867466,0.003216711,0.019869711,0.00084502937],"domain_scores_gemma":[0.81196684,0.11048,0.015726428,0.015934706,0.04493612,0.00095591985],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.018163975,0.0022760676,0.0013046651,0.011845277,0.0014250353,0.0034612913,0.0016780205,0.0016336909,0.004798247],"category_scores_gemma":[0.120037414,0.00051941996,0.0011779767,0.007963912,0.0008814663,0.0033649139,0.0019088109,0.0025695919,0.002180461],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016827037,0.0005997869,0.023850698,0.0021334859,0.0006532823,0.0004320834,0.002115908,0.012904697,0.06803936,0.008675876,0.014427422,0.8644847],"study_design_scores_gemma":[0.0006273432,0.0046000895,0.07042322,0.0011262548,0.0011683236,0.0040106415,0.002121542,0.49735227,0.32232854,0.027741916,0.067661755,0.000838174],"about_ca_topic_score_codex":0.0011595599,"about_ca_topic_score_gemma":0.001552116,"teacher_disagreement_score":0.018163975,"about_ca_system_score_codex":0.00087547896,"about_ca_system_score_gemma":0.0012858681,"threshold_uncertainty_score":0.09606147},"labels":[],"label_agreement":null},{"id":"W2014131935","doi":"10.3115/1654524.1654534","title":"The acquisition and use of argument structure constructions","year":2005,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Argument (complex analysis); Computer science; Natural language processing; Representation (politics); Artificial intelligence; Verb; Bayesian probability; Linguistics; Theoretical computer science","score_opus":0.008018515312825665,"score_gpt":0.24549090301450197,"score_spread":0.2374723877016763,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2014131935","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.27354255,0.00030153146,0.70443374,0.0011769191,0.000025071618,0.00012991222,0.0004182128,0.00048952346,0.019482613],"genre_scores_gemma":[0.89467704,0.0003270419,0.10149188,0.00016390691,0.0000203286,0.00013328715,0.0005906599,0.00016924184,0.0024265628],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.9951591,0.0020795309,0.0002205306,0.0011744085,0.00107822,0.00028819748],"domain_scores_gemma":[0.9728222,0.017603926,0.002760317,0.004479614,0.0015866426,0.00074725016],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0055463235,0.00046163026,0.0006154786,0.0012907812,0.0007543615,0.0035191535,0.0016207079,0.0022489831,0.003641204],"category_scores_gemma":[0.047736425,0.0013803665,0.0009051597,0.0009856396,0.0030272778,0.0113013545,0.00270652,0.0034024105,0.0008279991],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00039094512,0.0005451653,0.04806665,0.0006019793,0.00026336114,0.00073898135,0.0067169033,0.04159445,0.036969334,0.6050991,0.002751251,0.25626194],"study_design_scores_gemma":[0.000056963003,0.00022464833,0.023827348,0.00012272016,0.000115493014,0.001550656,0.00055458327,0.23609896,0.0119874915,0.71845293,0.006887409,0.00012093387],"about_ca_topic_score_codex":0.0021595357,"about_ca_topic_score_gemma":0.002596564,"teacher_disagreement_score":0.0055463235,"about_ca_system_score_codex":0.0011312191,"about_ca_system_score_gemma":0.0014300907,"threshold_uncertainty_score":0.029332101},"labels":[],"label_agreement":null},{"id":"W2014400548","doi":"10.1109/saso.2014.42","title":"Towards an Agent-Based Simulation Model for Schema Matching","year":2014,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Schema (genetic algorithms); Schema matching; Computer science; Matching (statistics); Schema evolution; Semi-structured model; Schema migration; Theoretical computer science; Data mining; Database schema; Machine learning; Data integration; Mathematics","score_opus":0.03353479941778583,"score_gpt":0.3290216371855063,"score_spread":0.2954868377677205,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2014400548","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0026781703,0.00004160733,0.99508387,0.0001960147,0.000023760922,0.00006794053,0.000047311587,0.00038109094,0.0014801124],"genre_scores_gemma":[0.14141564,0.00029528112,0.85390514,0.00023954456,0.000034091518,0.0004484508,0.00042512006,0.00022406825,0.0030127035],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99821424,0.0007720095,0.00012629827,0.00024266145,0.00055104087,0.000093770665],"domain_scores_gemma":[0.99782854,0.0011607595,0.00015475774,0.00039159847,0.0003317867,0.00013261681],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023660103,0.0005965971,0.0008080636,0.000982262,0.00078962935,0.0027710064,0.0031064379,0.0022030314,0.004457063],"category_scores_gemma":[0.006868652,0.00074890384,0.0018801177,0.0009875816,0.0014700881,0.003093545,0.002666962,0.002892275,0.0011143587],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008967673,0.00008223402,0.00064946373,0.00011466744,0.000094928386,0.00014715198,0.0002986479,0.70500416,0.0034014056,0.2722143,0.0011033483,0.016800087],"study_design_scores_gemma":[0.000015206115,0.000010171118,0.000022254231,0.000010311102,0.000011179027,0.000019522748,0.000016132333,0.9642759,0.00075143285,0.031261392,0.0036001573,0.0000063748075],"about_ca_topic_score_codex":0.006464503,"about_ca_topic_score_gemma":0.006073373,"teacher_disagreement_score":0.006464503,"about_ca_system_score_codex":0.0015858955,"about_ca_system_score_gemma":0.002499866,"threshold_uncertainty_score":0.01491034},"labels":[],"label_agreement":null},{"id":"W2014520247","doi":"10.7202/003372ar","title":"Une analyse terminométrique pour le repérage automatique des descripteurs complexes dans les textes de spécialité","year":2002,"lang":"fr","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Philosophy","score_opus":0.07878136392422941,"score_gpt":0.29831132168063057,"score_spread":0.21952995775640116,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2014520247","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013221512,0.000319132,0.984382,0.00011194992,0.00002186125,0.000072875286,0.0001414698,0.00097179133,0.0007574095],"genre_scores_gemma":[0.11305736,0.0004359166,0.8807115,0.000059099126,0.000036318503,0.00023253672,0.00076066726,0.0005735499,0.0041330303],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9959615,0.0012440289,0.0003501086,0.0010399773,0.0012557437,0.0001486108],"domain_scores_gemma":[0.9876666,0.007568442,0.0007786718,0.0013859039,0.0024190699,0.00018122118],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0041993554,0.0012455733,0.0012657619,0.003888028,0.0013483886,0.004400836,0.0012470293,0.0014796872,0.0056683626],"category_scores_gemma":[0.014038095,0.0007987246,0.0022450674,0.002928249,0.0020232736,0.004135926,0.0016500416,0.0021652314,0.0018515792],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007902369,0.0001548352,0.014890307,0.00204037,0.00047674135,0.00068199873,0.009736358,0.03073917,0.14072835,0.094873615,0.0036023979,0.70128566],"study_design_scores_gemma":[0.00010739934,0.00057261105,0.015923917,0.0007051537,0.00057885546,0.0025357478,0.005561946,0.60467595,0.17123397,0.11343393,0.08436577,0.00030471355],"about_ca_topic_score_codex":0.006903649,"about_ca_topic_score_gemma":0.008515327,"teacher_disagreement_score":0.006903649,"about_ca_system_score_codex":0.0016674172,"about_ca_system_score_gemma":0.0019449717,"threshold_uncertainty_score":0.022208571},"labels":[],"label_agreement":null},{"id":"W2014571024","doi":"10.7202/037219ar","title":"Stratégie pour la détection semi-automatique des néologismes de presse","year":2007,"lang":"fr","type":"article","venue":"TTR traduction terminologie rédaction","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Philosophy; Political science","score_opus":0.054367767270390326,"score_gpt":0.33601653403893245,"score_spread":0.2816487667685421,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2014571024","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012787175,0.0020607295,0.9570351,0.001187031,0.000264918,0.0006055859,0.004390068,0.013695761,0.0079735005],"genre_scores_gemma":[0.067772016,0.0013434296,0.9047571,0.00038014588,0.00016254235,0.0010155652,0.012633353,0.0016368562,0.0102990465],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9920373,0.0021655264,0.00090079074,0.00220961,0.0023534424,0.00033324983],"domain_scores_gemma":[0.9867446,0.007341234,0.0009146609,0.0017236697,0.0030198684,0.00025586295],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0052317395,0.001926013,0.0014450634,0.008195387,0.0012468244,0.0074765817,0.0015568087,0.0019614683,0.011226465],"category_scores_gemma":[0.017371627,0.0014552091,0.0027768724,0.003434456,0.0015154219,0.004034727,0.0026623518,0.0028738067,0.015045752],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00044146905,0.00015399374,0.010073291,0.0028936348,0.00050088076,0.000595863,0.0029957774,0.007014219,0.062686436,0.02342454,0.024896186,0.8643237],"study_design_scores_gemma":[0.00021746712,0.0004479057,0.042402305,0.0019465938,0.00067626586,0.003134679,0.0050880644,0.24456902,0.17320405,0.087853536,0.4399714,0.0004887414],"about_ca_topic_score_codex":0.011076098,"about_ca_topic_score_gemma":0.010433058,"teacher_disagreement_score":0.011226465,"about_ca_system_score_codex":0.0017554417,"about_ca_system_score_gemma":0.0041925455,"threshold_uncertainty_score":0.03755629},"labels":[],"label_agreement":null},{"id":"W2016200093","doi":"10.5539/ijel.v4n6p52","title":"A Corpus-Based Study on Original English Abstracts and Translated English Abstracts: A Case Study of Passive Voice and Pronouns","year":2014,"lang":"en","type":"article","venue":"International Journal of English Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Normalization (sociology); Natural language processing; Linguistics; Corpus linguistics; Artificial intelligence; Part of speech","score_opus":0.013500417039671909,"score_gpt":0.2972570238573678,"score_spread":0.2837566068176959,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2016200093","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.97222024,0.0055687367,0.0064906497,0.00091723626,0.0001324011,0.0010909578,0.0034304834,0.000077382,0.010071901],"genre_scores_gemma":[0.9560726,0.005404373,0.023703134,0.0004916089,0.00021346117,0.002043109,0.0067012175,0.00014548322,0.005224991],"study_design_codex":"qualitative","study_design_gemma":"observational","domain_scores_codex":[0.9924132,0.0028811337,0.0012690305,0.0010959473,0.0020769518,0.000263619],"domain_scores_gemma":[0.9518558,0.03136689,0.003996767,0.002765531,0.009160255,0.0008548069],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0072379527,0.00054907525,0.00072832766,0.012085295,0.0029345376,0.00207966,0.00095833704,0.0011132933,0.0018675321],"category_scores_gemma":[0.027472362,0.00042064258,0.0006402956,0.01710899,0.0025446343,0.0024882357,0.0020229272,0.00071228377,0.00058657303],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009535544,0.002013016,0.17508227,0.013562409,0.00031513665,0.03959142,0.42157203,0.0011024054,0.03291608,0.010662615,0.017500827,0.28472826],"study_design_scores_gemma":[0.00024360832,0.0012260802,0.4340801,0.0022376063,0.0007195711,0.017604312,0.28733155,0.0032915317,0.01814675,0.0024323803,0.23235491,0.00033167162],"about_ca_topic_score_codex":0.008954815,"about_ca_topic_score_gemma":0.01737743,"teacher_disagreement_score":0.012085295,"about_ca_system_score_codex":0.0018937627,"about_ca_system_score_gemma":0.0019042976,"threshold_uncertainty_score":0.0382784},"labels":[],"label_agreement":null},{"id":"W2016238228","doi":"10.1002/meet.2008.1450450357","title":"A semantic interface for post secondary education programs","year":2008,"lang":"en","type":"article","venue":"Proceedings of the American Society for Information Science and Technology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; World Wide Web; Interface (matter); Graph; Semantic Web; User interface; Programming language","score_opus":0.008668668006836782,"score_gpt":0.26977699933115945,"score_spread":0.26110833132432265,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2016238228","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.094883606,0.00023068415,0.75925606,0.001186978,0.00022286251,0.0005042061,0.0032178273,0.097179756,0.043318],"genre_scores_gemma":[0.5069681,0.00033096224,0.43381983,0.0009415262,0.00010141685,0.00071370136,0.007389268,0.006846938,0.042888332],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995327,0.00016148022,0.000047167763,0.00007435096,0.00014413074,0.000040084666],"domain_scores_gemma":[0.99891007,0.00047436205,0.00004961751,0.00014913097,0.00026965397,0.00014718466],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010325881,0.00056674256,0.00037426414,0.0010183413,0.0005175341,0.0016846821,0.0009257887,0.0010180121,0.020221645],"category_scores_gemma":[0.002318192,0.0002005172,0.00046487083,0.0006689838,0.00036235785,0.0032270518,0.0018843091,0.00071204285,0.003667733],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0023020026,0.002024676,0.0084526185,0.0013925869,0.0000797446,0.0019157823,0.010514409,0.006241201,0.08627527,0.13109706,0.12395325,0.6257514],"study_design_scores_gemma":[0.00030436958,0.00051067,0.0043778196,0.00028237267,0.00011372594,0.0012178395,0.0019366135,0.09216722,0.055803023,0.033411413,0.80974543,0.00012962172],"about_ca_topic_score_codex":0.0012918618,"about_ca_topic_score_gemma":0.0014237433,"teacher_disagreement_score":0.020221645,"about_ca_system_score_codex":0.0004829043,"about_ca_system_score_gemma":0.0005664671,"threshold_uncertainty_score":0.06764817},"labels":[],"label_agreement":null},{"id":"W2017180565","doi":"10.3115/974147.974151","title":"Automatic construction of parallel English-Chinese corpus for cross-language information retrieval","year":2000,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":60,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Parallel corpora; Natural language processing; Machine translation; Cross-language information retrieval; Artificial intelligence; Obstacle; Probabilistic logic; Information retrieval; Translation (biology); Language model","score_opus":0.0046891765913976285,"score_gpt":0.26837743116226065,"score_spread":0.26368825457086303,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2017180565","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09120852,0.0018424899,0.8523216,0.0014706773,0.0009623846,0.0020814636,0.015294477,0.014210398,0.020608032],"genre_scores_gemma":[0.2527108,0.00090834056,0.6800015,0.00035911537,0.0003337418,0.0035624667,0.050077442,0.0021531445,0.009893547],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9975631,0.0008533963,0.00030093844,0.0005493049,0.0005759802,0.00015723365],"domain_scores_gemma":[0.99423075,0.0019119872,0.00023273964,0.0009845623,0.0024771402,0.00016282093],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030856496,0.0011277373,0.0014645165,0.003858005,0.0023640022,0.0015324886,0.0012911741,0.0006646579,0.0132571915],"category_scores_gemma":[0.011310821,0.00092336937,0.0009385018,0.004570954,0.0008804863,0.0035822303,0.0023523788,0.00162137,0.0053302734],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011720178,0.0008623335,0.0049288147,0.0022793782,0.00020233054,0.0017004249,0.0020549532,0.022473188,0.10447078,0.052272834,0.120067626,0.6875153],"study_design_scores_gemma":[0.0007680109,0.00062971906,0.0135117965,0.00036231105,0.000583103,0.0020206799,0.001934043,0.4315218,0.19159937,0.049905498,0.30682456,0.00033909263],"about_ca_topic_score_codex":0.009248594,"about_ca_topic_score_gemma":0.010977428,"teacher_disagreement_score":0.0132571915,"about_ca_system_score_codex":0.0015365544,"about_ca_system_score_gemma":0.0047994335,"threshold_uncertainty_score":0.04434979},"labels":[],"label_agreement":null},{"id":"W2017292107","doi":"10.1162/coli_a_00143","title":"Computing Lexical Contrast","year":2012,"lang":"en","type":"article","venue":"Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":100,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; National Research Council Canada","funders":"National Research Council Canada; Natural Sciences and Engineering Research Council of Canada; National Science Foundation","keywords":"Contrast (vision); Computer science; Natural language processing; Artificial intelligence; Word (group theory); Meaning (existential); Focus (optics); Linguistics; Information retrieval; Psychology; Philosophy","score_opus":0.020473053507219972,"score_gpt":0.3092489102287959,"score_spread":0.28877585672157596,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2017292107","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6477583,0.0025458706,0.2844786,0.00091863895,0.000525663,0.00060965854,0.012439431,0.0033634074,0.04736044],"genre_scores_gemma":[0.8689468,0.00030246607,0.11663323,0.00018647958,0.00015153557,0.00036002582,0.011235234,0.0002736791,0.0019105489],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9952253,0.00074719236,0.00068535237,0.0016975742,0.0012604741,0.0003840722],"domain_scores_gemma":[0.9888359,0.0067430777,0.0007787124,0.0010893891,0.002070561,0.00048241916],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023437354,0.00090478634,0.0014648095,0.011683117,0.0016506898,0.004957482,0.0011349526,0.0015334842,0.008647076],"category_scores_gemma":[0.024712952,0.000533111,0.0010112422,0.006615039,0.0011194934,0.008883035,0.0032862942,0.0011340771,0.0026129358],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0027019903,0.0006660275,0.160423,0.0017754966,0.0007319567,0.0016989597,0.002535648,0.013450382,0.0489909,0.08746205,0.023318775,0.6562448],"study_design_scores_gemma":[0.0005158137,0.0014900909,0.1436443,0.0004652763,0.000872473,0.0050581284,0.0063725873,0.28215855,0.049055215,0.41531098,0.094644524,0.00041216676],"about_ca_topic_score_codex":0.0014944297,"about_ca_topic_score_gemma":0.0017379217,"teacher_disagreement_score":0.011683117,"about_ca_system_score_codex":0.0012138683,"about_ca_system_score_gemma":0.0010387083,"threshold_uncertainty_score":0.028927386},"labels":[],"label_agreement":null},{"id":"W2017877614","doi":"10.1017/s0959269507003018","title":"Thematic indirect objects in French","year":2007,"lang":"en","type":"article","venue":"Journal of French Language Studies","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Transitive relation; Argument (complex analysis); Object (grammar); Event (particle physics); Computer science; Linguistics; Causative; Thematic map; Position (finance); Artificial intelligence; Mathematics; Philosophy; Verb; Combinatorics; Physics; Geography","score_opus":0.019672806981421355,"score_gpt":0.3221367907907047,"score_spread":0.3024639838092833,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2017877614","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.400841,0.005474678,0.28954667,0.0040792436,0.0003757346,0.00016116504,0.0010333466,0.0031575416,0.29533058],"genre_scores_gemma":[0.94963723,0.00115142,0.026480226,0.0003901793,0.00017885769,0.00009273167,0.00058358663,0.00056515215,0.020920603],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99854857,0.0005399224,0.00006668445,0.00033695129,0.0002812494,0.0002265688],"domain_scores_gemma":[0.99915326,0.00032816816,0.00011495948,0.00018458578,0.0001837205,0.000035427456],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002164773,0.001086277,0.0005444356,0.0019832482,0.0029571676,0.0059377304,0.0009362641,0.0015829387,0.0062702144],"category_scores_gemma":[0.0023873502,0.0006554274,0.0011155838,0.0014718622,0.004875989,0.006972295,0.002378738,0.0019820367,0.0009956221],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000068256595,0.000015728328,0.0026558205,0.00013573655,0.0000326366,0.0006775998,0.017058976,0.00071803265,0.003781955,0.94471294,0.0015321397,0.02861001],"study_design_scores_gemma":[0.00007891553,0.00019356492,0.013208595,0.0003916097,0.00023359788,0.0031240287,0.01431327,0.0083010895,0.0105301095,0.47126803,0.47817293,0.00018422578],"about_ca_topic_score_codex":0.0275075,"about_ca_topic_score_gemma":0.016947668,"teacher_disagreement_score":0.0275075,"about_ca_system_score_codex":0.006680069,"about_ca_system_score_gemma":0.0018548868,"threshold_uncertainty_score":0.05469483},"labels":[],"label_agreement":null},{"id":"W2017988763","doi":"10.3115/1620754.1620835","title":"Hierarchical search for parsing","year":2009,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Heuristics; Parsing; Computer science; Simple (philosophy); Computation; Artificial intelligence; Theoretical computer science; Machine learning; Algorithm","score_opus":0.02407577681050759,"score_gpt":0.32550815041551034,"score_spread":0.30143237360500275,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2017988763","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017318835,0.00052216224,0.96470827,0.00039989204,0.000053895503,0.00017412489,0.00045652033,0.006104724,0.010261633],"genre_scores_gemma":[0.17070971,0.00024068686,0.8253116,0.00019160831,0.000027061062,0.00017835638,0.0006643271,0.0006056568,0.0020710165],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9977235,0.0011713511,0.0001527123,0.00044826165,0.00034139046,0.00016275872],"domain_scores_gemma":[0.9947477,0.0036734827,0.00020443815,0.0009139532,0.00031862292,0.00014180884],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024553253,0.00094491435,0.00091160735,0.0016945138,0.0014893338,0.0022281415,0.0016220208,0.0016702892,0.010262038],"category_scores_gemma":[0.009510153,0.00070348586,0.0014113467,0.002148016,0.0017224278,0.0042189523,0.0024893414,0.0016553702,0.0018459493],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00031390728,0.00023819256,0.0020394267,0.00080398814,0.00012213535,0.00020612354,0.00081884436,0.2049579,0.010148825,0.34566894,0.02086913,0.41381252],"study_design_scores_gemma":[0.000075208154,0.000080048434,0.0005179894,0.00009170652,0.000057350608,0.00009687358,0.00016880187,0.6993999,0.0061478307,0.2803443,0.012974412,0.000045570134],"about_ca_topic_score_codex":0.0074635814,"about_ca_topic_score_gemma":0.012282799,"teacher_disagreement_score":0.010262038,"about_ca_system_score_codex":0.0015513772,"about_ca_system_score_gemma":0.0033895143,"threshold_uncertainty_score":0.03432995},"labels":[],"label_agreement":null},{"id":"W2018116550","doi":"10.3115/1220355.1220445","title":"A path-based transfer model for machine translation","year":2004,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":56,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Machine translation; Dependency grammar; Dependency (UML); Parsing; Artificial intelligence; Transfer-based machine translation; Path (computing); Natural language processing; Translation (biology); Synchronous context-free grammar; Tree (set theory); Graph; Transfer (computing); Language model; Word (group theory); Set (abstract data type); Example-based machine translation; Theoretical computer science; Programming language; Mathematics","score_opus":0.024130035667295106,"score_gpt":0.2734985460518554,"score_spread":0.24936851038456032,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2018116550","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004314871,0.00035413195,0.9907362,0.000473988,0.0000872761,0.00005483587,0.00028827897,0.0011319589,0.0025585152],"genre_scores_gemma":[0.4078487,0.0022841953,0.5693876,0.0006704129,0.0003323987,0.00093195477,0.0025304854,0.00069807936,0.015316179],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992999,0.00025532083,0.000036848036,0.00019915169,0.00014558845,0.00006323346],"domain_scores_gemma":[0.9989955,0.0005894806,0.000063980646,0.00013733841,0.00017977781,0.000033967583],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011518268,0.00092933135,0.0008681622,0.0010624542,0.00079433253,0.0010434556,0.0018298145,0.0017158116,0.0071351505],"category_scores_gemma":[0.0034177876,0.00057278073,0.0012221204,0.0016119842,0.001071225,0.0045266887,0.0013657445,0.0017373088,0.0033629234],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002544509,0.00016653266,0.0009103255,0.00028233725,0.00020553266,0.00035878757,0.00029215202,0.5701314,0.0052786614,0.18179584,0.012505935,0.227818],"study_design_scores_gemma":[0.000023879382,0.00005318395,0.00011975282,0.000012809217,0.00003048362,0.000083785555,0.000012490806,0.83169156,0.0009630783,0.16251126,0.0044762255,0.000021490287],"about_ca_topic_score_codex":0.0039337534,"about_ca_topic_score_gemma":0.0030170628,"teacher_disagreement_score":0.0071351505,"about_ca_system_score_codex":0.0010510575,"about_ca_system_score_gemma":0.0016131818,"threshold_uncertainty_score":0.023869455},"labels":[],"label_agreement":null},{"id":"W2018179016","doi":"10.7202/016739ar","title":"Grammar and Translation: The Noun + Noun Conundrum","year":2007,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Linguistics; Noun; Grammar; Gerund; Computer science; Rendering (computer graphics); Grammatical category; Commit; Noun phrase; English grammar; Meaning (existential); Proper noun; Nominalization; Natural language processing; Psychology; Artificial intelligence; Philosophy","score_opus":0.033012505089768024,"score_gpt":0.2790394268687934,"score_spread":0.24602692177902538,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2018179016","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07409328,0.026609242,0.6757773,0.096168846,0.0026741799,0.00017693521,0.0002826449,0.0009074417,0.12330999],"genre_scores_gemma":[0.7703466,0.012604275,0.18688677,0.009182191,0.0023763748,0.00028613408,0.0003367929,0.0011395133,0.016841415],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"qualitative","domain_scores_codex":[0.9861639,0.010393351,0.00058119994,0.001049544,0.0014453725,0.00036659895],"domain_scores_gemma":[0.98528486,0.010345528,0.0009998406,0.0016993381,0.0014167337,0.00025365342],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007194765,0.00079921033,0.0009747824,0.0017169553,0.0021486098,0.0061670393,0.0012194128,0.002799598,0.0051694694],"category_scores_gemma":[0.022971284,0.000656912,0.0007598473,0.0020092097,0.018283444,0.015088755,0.0029120385,0.0028884504,0.0014092582],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000922034,0.000035261997,0.0016574201,0.00041215486,0.000035408822,0.0008653922,0.017168397,0.0012856567,0.0016445823,0.8759107,0.006226265,0.09466658],"study_design_scores_gemma":[0.000030412886,0.00007435204,0.0014945006,0.0003781905,0.000019115261,0.0017877212,0.0068148053,0.0027301223,0.001268855,0.8652638,0.12008807,0.000049939787],"about_ca_topic_score_codex":0.0038082278,"about_ca_topic_score_gemma":0.0024525495,"teacher_disagreement_score":0.007194765,"about_ca_system_score_codex":0.002275127,"about_ca_system_score_gemma":0.0031495267,"threshold_uncertainty_score":0.038050056},"labels":[],"label_agreement":null},{"id":"W2018725701","doi":"10.1353/cjl.2011.0009","title":"Mesure de la productivité morphologique des créoles : au-delà des méthodes quantitatives","year":2011,"lang":"fr","type":"article","venue":"The Canadian Journal of Linguistics / La revue canadienne de linguistique","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Humanities; Philosophy","score_opus":0.054009613825122246,"score_gpt":0.3042198745727583,"score_spread":0.250210260747636,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2018725701","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.082264766,0.0016328816,0.90078956,0.00024296694,0.00010967944,0.0003095163,0.001479434,0.004400425,0.008770859],"genre_scores_gemma":[0.34286055,0.0011100626,0.6436095,0.00012545144,0.000066234476,0.0008167883,0.0013473427,0.0016047214,0.008459323],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99595046,0.0011306575,0.00025800336,0.00119985,0.00131368,0.00014735427],"domain_scores_gemma":[0.9806059,0.011763767,0.0011343403,0.0028608702,0.0034670515,0.00016797903],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006636871,0.0014527625,0.0007767055,0.0053570447,0.0006256987,0.0039463006,0.0010548914,0.0012289006,0.0074465824],"category_scores_gemma":[0.025731446,0.000980055,0.0010774067,0.0034996734,0.0014851961,0.0030203536,0.0015327462,0.0015063243,0.0026175752],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00045199104,0.00009268581,0.02454035,0.0017080118,0.0005652308,0.00017734287,0.0034741892,0.005348801,0.17746639,0.009185002,0.0018637956,0.7751262],"study_design_scores_gemma":[0.00016867353,0.0013543693,0.31544334,0.0007470798,0.0009699414,0.0027605745,0.00615818,0.1456326,0.37896708,0.032370005,0.11474248,0.00068570214],"about_ca_topic_score_codex":0.0034380925,"about_ca_topic_score_gemma":0.004195497,"teacher_disagreement_score":0.0074465824,"about_ca_system_score_codex":0.00086503464,"about_ca_system_score_gemma":0.00057178584,"threshold_uncertainty_score":0.035099566},"labels":[],"label_agreement":null},{"id":"W2020932825","doi":"","title":"Comparing Different Units for Query Translation in Chinese Cross-Language Information Retrieval","year":2013,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Bigram; Computer science; Cross-language information retrieval; Natural language processing; Ranking (information retrieval); Artificial intelligence; Machine translation; Word (group theory); Search engine indexing; Translation (biology); Information retrieval; Language model; Linguistics; Trigram","score_opus":0.020636934158821667,"score_gpt":0.2960630798886607,"score_spread":0.27542614572983903,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2020932825","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9560355,0.0029669316,0.032751232,0.00021173403,0.000084646905,0.0002842427,0.0001805716,0.00066858827,0.006816612],"genre_scores_gemma":[0.9750236,0.0004341754,0.02311322,0.00005704633,0.000029739882,0.00012431697,0.00033279398,0.00007693734,0.00080806384],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9960424,0.002438504,0.0004437017,0.00026571815,0.0006247698,0.00018489042],"domain_scores_gemma":[0.9938332,0.0039960938,0.00028539024,0.00054470106,0.0011861381,0.00015449346],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004891056,0.0006660561,0.00075088517,0.002546529,0.00085909915,0.0013855544,0.000539382,0.00053488935,0.0017034383],"category_scores_gemma":[0.013107336,0.00021196135,0.00068573444,0.0021987432,0.00077069737,0.0031764612,0.00080165034,0.00036240416,0.00060419086],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0069692107,0.0014636024,0.04994609,0.0024820073,0.00086195307,0.000433687,0.0011928004,0.08709554,0.10466066,0.010334179,0.0048420336,0.72971827],"study_design_scores_gemma":[0.0005483613,0.004919815,0.057024743,0.00013856762,0.0012216445,0.0005060942,0.0012537041,0.7763931,0.14791255,0.005792852,0.0040909755,0.00019761028],"about_ca_topic_score_codex":0.005505999,"about_ca_topic_score_gemma":0.005458278,"teacher_disagreement_score":0.005505999,"about_ca_system_score_codex":0.0014321997,"about_ca_system_score_gemma":0.0010458744,"threshold_uncertainty_score":0.025866687},"labels":[],"label_agreement":null},{"id":"W2021254338","doi":"10.1145/1871840.1871841","title":"The nature of noise in linguistic corpora","year":2010,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Variety (cybernetics); Computer science; Natural language processing; Artificial intelligence; Linguistics","score_opus":0.005712326292142829,"score_gpt":0.2647338117014542,"score_spread":0.25902148540931136,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2021254338","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.110492505,0.015435168,0.78704333,0.030407246,0.0032767346,0.00044598218,0.00644474,0.004031904,0.042422414],"genre_scores_gemma":[0.755438,0.0071554193,0.18646954,0.016130092,0.0045362753,0.0014266558,0.014155231,0.002376624,0.012312196],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.922444,0.029345678,0.0043667844,0.011796104,0.030604638,0.0014427784],"domain_scores_gemma":[0.71488965,0.22967355,0.009101704,0.024675442,0.020864159,0.0007954656],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.043275625,0.00084181695,0.0021954493,0.009427086,0.0031091163,0.008865211,0.0034557395,0.003095215,0.002960949],"category_scores_gemma":[0.2956412,0.0017417406,0.0008492388,0.0130553795,0.0068957577,0.009413852,0.0048499526,0.004131609,0.0024207733],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007268527,0.00023437003,0.054041695,0.0026561834,0.0010022849,0.0022704348,0.015265956,0.028026642,0.010346827,0.32800508,0.10476662,0.45265695],"study_design_scores_gemma":[0.00012811592,0.00012871098,0.050438426,0.0014722081,0.0003324878,0.0028623384,0.0043364917,0.10813872,0.01029569,0.6399316,0.1816261,0.00030908317],"about_ca_topic_score_codex":0.008701993,"about_ca_topic_score_gemma":0.0096056275,"teacher_disagreement_score":0.043275625,"about_ca_system_score_codex":0.0030539946,"about_ca_system_score_gemma":0.0025098596,"threshold_uncertainty_score":0.2288661},"labels":[],"label_agreement":null},{"id":"W2021294551","doi":"10.3166/isi.8.3.55-70","title":"Text Representation with WordNet Synsets Using Soft Sense Disambiguation","year":2003,"lang":"fr","type":"article","venue":"Ingénierie des systèmes d information","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; WordNet; Representation (politics); Philosophy; Computer science; Artificial intelligence; Political science","score_opus":0.027569267840617005,"score_gpt":0.27707934481395,"score_spread":0.24951007697333302,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2021294551","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.062346384,0.00069341296,0.9292393,0.00070184504,0.00019350795,0.00034289068,0.00066785305,0.0033576237,0.0024571388],"genre_scores_gemma":[0.36060733,0.0004043286,0.6343122,0.00019657747,0.00011380363,0.0003723628,0.0019643141,0.00018381614,0.0018452987],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99823606,0.0007375128,0.00022356576,0.00037339682,0.0003482146,0.00008119959],"domain_scores_gemma":[0.99741143,0.0013073641,0.00041603236,0.0003429806,0.00041558527,0.000106606094],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021951506,0.0009449139,0.0010043489,0.0062870355,0.0011715797,0.0027749976,0.0011628229,0.0012241255,0.0029632235],"category_scores_gemma":[0.0060665607,0.00043451638,0.0011654095,0.0051426818,0.0012138515,0.004937388,0.002277023,0.00092492433,0.0013351116],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013795167,0.00044792565,0.0036052298,0.00072980253,0.00028380746,0.0009262497,0.00164475,0.14477164,0.026053717,0.06842412,0.010005383,0.7417278],"study_design_scores_gemma":[0.00010577815,0.00017488083,0.0011688196,0.00014528511,0.00010859778,0.000349922,0.00064662274,0.88277525,0.019187136,0.083219856,0.01204448,0.00007340718],"about_ca_topic_score_codex":0.0016766677,"about_ca_topic_score_gemma":0.0017618118,"teacher_disagreement_score":0.0062870355,"about_ca_system_score_codex":0.000996237,"about_ca_system_score_gemma":0.0009863789,"threshold_uncertainty_score":0.011609197},"labels":[],"label_agreement":null},{"id":"W2021343534","doi":"10.1109/slt.2014.7078547","title":"Document-based Dirichlet class language model for speech recognition using document-based n-gram events","year":2014,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Institut National de la Recherche Scientifique","funders":"","keywords":"Perplexity; n-gram; Language model; Computer science; Artificial intelligence; Natural language processing; Context (archaeology); Class (philosophy); Part of speech; Word error rate; Latent Dirichlet allocation; Speech recognition; Word (group theory); Gram; Dirichlet distribution; Topic model; Linguistics; Mathematics","score_opus":0.024500394467597175,"score_gpt":0.30760400850018604,"score_spread":0.2831036140325889,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2021343534","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029161584,0.0009155003,0.9656172,0.00071563,0.00023064179,0.00013606166,0.00068597373,0.001022961,0.0015144558],"genre_scores_gemma":[0.7127275,0.0017623616,0.26431346,0.000836943,0.0006816921,0.0010833475,0.0035033377,0.00047605124,0.014615297],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9976833,0.0009259036,0.00016542774,0.0006319988,0.00037338032,0.00022008315],"domain_scores_gemma":[0.9974152,0.0017061807,0.00016686681,0.00026329033,0.0003576731,0.00009080683],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025267235,0.0010925423,0.0020009426,0.0016468641,0.00089493825,0.001704372,0.0034785133,0.0017723814,0.0024734961],"category_scores_gemma":[0.0052532763,0.00070996705,0.002151023,0.0018603869,0.0012924789,0.0031358388,0.0011937727,0.0029892155,0.0017457039],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016238816,0.0004722868,0.005004117,0.00047938808,0.0003905449,0.00037189084,0.0012842972,0.56175894,0.012989756,0.08409448,0.009663212,0.32186714],"study_design_scores_gemma":[0.000024892439,0.00003571514,0.00034592787,0.000012059235,0.00002437845,0.000042541553,0.000023375676,0.9848571,0.0011530264,0.012274429,0.0011737494,0.000032828826],"about_ca_topic_score_codex":0.013200772,"about_ca_topic_score_gemma":0.013183576,"teacher_disagreement_score":0.013200772,"about_ca_system_score_codex":0.002080835,"about_ca_system_score_gemma":0.0016351794,"threshold_uncertainty_score":0.026247859},"labels":[],"label_agreement":null},{"id":"W2021352277","doi":"10.3115/1627306.1627320","title":"Zero to spoken dialogue system in one quarter","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Quarter (Canadian coin); Computational linguistics; Zero (linguistics); Architecture; Linguistics; Spoken language; Open source; Artificial intelligence; Natural language processing; Programming language; History","score_opus":0.020194379024232764,"score_gpt":0.2410642183627724,"score_spread":0.22086983933853963,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2021352277","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.093992256,0.002601192,0.6759081,0.004932021,0.005500053,0.0006215439,0.0019904603,0.083128,0.13132642],"genre_scores_gemma":[0.5305099,0.00080413773,0.23410602,0.0026075516,0.0006255993,0.00058604,0.0051792115,0.0066361935,0.21894537],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9983217,0.00058799976,0.00012841876,0.00043929456,0.00033997113,0.00018250245],"domain_scores_gemma":[0.9988103,0.000271945,0.00002739591,0.00034307965,0.00035127398,0.00019608246],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018435892,0.0004553897,0.000708709,0.00046340242,0.00083779375,0.0024352157,0.0010467696,0.0010396116,0.026644455],"category_scores_gemma":[0.0037203566,0.00040083056,0.00037600234,0.00022724153,0.0008047893,0.002243186,0.003317874,0.0015406889,0.011292731],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0025357893,0.00049441395,0.0016877541,0.00043113105,0.00006481213,0.0008007554,0.0044236216,0.00506652,0.047897514,0.18423566,0.16997634,0.5823857],"study_design_scores_gemma":[0.0002976145,0.0007323764,0.0014038894,0.00014537989,0.000061056715,0.00085371686,0.0012365424,0.080454536,0.057059363,0.08131693,0.7763266,0.0001120368],"about_ca_topic_score_codex":0.0016884041,"about_ca_topic_score_gemma":0.0012283132,"teacher_disagreement_score":0.026644455,"about_ca_system_score_codex":0.0009767592,"about_ca_system_score_gemma":0.0011707296,"threshold_uncertainty_score":0.08913463},"labels":[],"label_agreement":null},{"id":"W2021474593","doi":"10.7202/001917ar","title":"Aide au transfert lexical dans une perspective de TAO : expérimentation sur un lexique non-terminologique","year":2002,"lang":"fr","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Philosophy; Humanities","score_opus":0.05333798097277809,"score_gpt":0.296021957663359,"score_spread":0.2426839766905809,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2021474593","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.49271634,0.00029223182,0.4645869,0.00044904594,0.00007566126,0.00035122002,0.00070405204,0.0025938954,0.038230617],"genre_scores_gemma":[0.75630796,0.0003109734,0.22233716,0.00010100856,0.000017385277,0.00018584893,0.0007001251,0.0010477288,0.018991694],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9986602,0.00051697245,0.000103651466,0.00031004727,0.00031639464,0.000092592716],"domain_scores_gemma":[0.99585646,0.0025569329,0.00015733152,0.0008018021,0.0005325478,0.00009489287],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00212915,0.00043913387,0.0004405811,0.00066550885,0.00079418905,0.0030818793,0.0008030652,0.0005602751,0.008459263],"category_scores_gemma":[0.006238409,0.00039154594,0.0004470064,0.0009195887,0.0014955869,0.0045651644,0.0017684983,0.0011518867,0.0019059613],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002055328,0.00041885307,0.021217192,0.001964568,0.000080405596,0.0009289418,0.03682867,0.0085121,0.25461757,0.21927744,0.002583499,0.45151535],"study_design_scores_gemma":[0.00041155197,0.002014843,0.027428942,0.00055000774,0.00042436217,0.0023949647,0.033461608,0.12626202,0.4636783,0.07845048,0.26466253,0.0002603601],"about_ca_topic_score_codex":0.006222059,"about_ca_topic_score_gemma":0.008170117,"teacher_disagreement_score":0.008459263,"about_ca_system_score_codex":0.0013102244,"about_ca_system_score_gemma":0.0018639652,"threshold_uncertainty_score":0.028299093},"labels":[],"label_agreement":null},{"id":"W2021589264","doi":"10.3115/1613692.1613702","title":"Classifying particle semantics in English verb-particle constructions","year":2006,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":36,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Principle of compositionality; Verb; Computer science; Natural language processing; Semantics (computer science); Meaning (existential); Particle (ecology); Artificial intelligence; Feature (linguistics); Component (thermodynamics); Space (punctuation); Word (group theory); Test (biology); Linguistics; Psychology; Programming language; Physics","score_opus":0.011841665162732573,"score_gpt":0.250924347079345,"score_spread":0.23908268191661244,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2021589264","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.94241655,0.0002550238,0.052457225,0.00019588873,0.000044880508,0.000052309646,0.00047255214,0.00032431268,0.0037813028],"genre_scores_gemma":[0.98701364,0.00005700537,0.011899123,0.000018762179,0.000009559344,0.000025085223,0.0006473551,0.000045445217,0.0002841348],"study_design_codex":"design_other","study_design_gemma":"qualitative","domain_scores_codex":[0.9993766,0.00021070277,0.00008885036,0.00015716176,0.000110174165,0.000056489327],"domain_scores_gemma":[0.9969055,0.0020930907,0.0003267006,0.00025519726,0.00032406798,0.00009545361],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012253253,0.0004596217,0.00039133814,0.0018566207,0.00079614826,0.0016562365,0.00044223113,0.00086778705,0.0015900325],"category_scores_gemma":[0.00530591,0.00022308729,0.00065549323,0.001208414,0.0013359465,0.0031756747,0.0010458097,0.0008332561,0.00029812584],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021894143,0.00068333745,0.17850423,0.0010176315,0.00026978803,0.002017086,0.010354533,0.026149562,0.090408646,0.16422053,0.0062408005,0.5179444],"study_design_scores_gemma":[0.00021736755,0.00055758446,0.1583116,0.00027334734,0.00027627815,0.003391329,0.00858192,0.44109806,0.04350717,0.324711,0.01886932,0.00020505674],"about_ca_topic_score_codex":0.0019755429,"about_ca_topic_score_gemma":0.0020858392,"teacher_disagreement_score":0.0019755429,"about_ca_system_score_codex":0.00055185053,"about_ca_system_score_gemma":0.00046895954,"threshold_uncertainty_score":0.006480217},"labels":[],"label_agreement":null},{"id":"W2021739502","doi":"10.5555/1182635.1164157","title":"Multi-column substring matching for database schema translation","year":2006,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":41,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Substring; Computer science; Database schema; Schema (genetic algorithms); Schema matching; Data mining; Algorithm; Artificial intelligence; Set (abstract data type); Database; Theoretical computer science; Database design; Information retrieval; Programming language; Data integration","score_opus":0.031107301051481722,"score_gpt":0.29643508175660666,"score_spread":0.2653277807051249,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2021739502","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006872103,0.0005492311,0.98263377,0.0001868588,0.000079348894,0.0001365535,0.0011171266,0.007229435,0.0011955479],"genre_scores_gemma":[0.03323588,0.00020848615,0.9623159,0.00011557242,0.00003006993,0.0000993361,0.0028128766,0.00031271463,0.00086913776],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99825007,0.00043397537,0.00022285002,0.000460171,0.00054752454,0.000085358944],"domain_scores_gemma":[0.99646425,0.0013165713,0.00030629654,0.0012194347,0.00061759056,0.00007579802],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00151959,0.0008185359,0.0007680375,0.0029520628,0.00075921783,0.0015050597,0.0018325077,0.001187009,0.007215354],"category_scores_gemma":[0.006466491,0.0005198514,0.0009449517,0.0046910904,0.0007735803,0.0033242302,0.001123808,0.0012203524,0.0043734764],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003281153,0.00019040794,0.003326322,0.00074756535,0.00013753628,0.0003808277,0.00046432842,0.009953081,0.04828793,0.025980389,0.021115214,0.88908833],"study_design_scores_gemma":[0.00012736306,0.0003086694,0.0035265968,0.00024541075,0.0002150606,0.0029311222,0.00051619293,0.53292555,0.25075033,0.09080526,0.117477536,0.00017097549],"about_ca_topic_score_codex":0.0018342722,"about_ca_topic_score_gemma":0.0024451157,"teacher_disagreement_score":0.007215354,"about_ca_system_score_codex":0.00061912544,"about_ca_system_score_gemma":0.0014350838,"threshold_uncertainty_score":0.024137735},"labels":[],"label_agreement":null},{"id":"W2021856631","doi":"10.7202/019244ar","title":"New Light Shed on Chinese Word Segmentation in MT by a Language Investigation","year":2008,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Natural language processing; Machine translation; Word (group theory); Artificial intelligence; Text segmentation; Feature (linguistics); Segmentation; Linguistics; Translation (biology); Space (punctuation); Chinese language","score_opus":0.018839004372472448,"score_gpt":0.271211043192125,"score_spread":0.2523720388196526,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2021856631","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5715231,0.036597427,0.07635434,0.15635557,0.00095523556,0.00018904332,0.0002901365,0.00013612917,0.15759896],"genre_scores_gemma":[0.98004305,0.0053623263,0.008112252,0.0021242143,0.0004318864,0.00012270566,0.00007501261,0.000055214907,0.0036734298],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99297476,0.0050751283,0.00027929968,0.00051069167,0.0005981423,0.00056199107],"domain_scores_gemma":[0.9612259,0.03278469,0.0014804455,0.002105387,0.0016630086,0.000740514],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.014225314,0.0008052353,0.0007761894,0.0032353683,0.005622753,0.0061239363,0.0009789981,0.0023314517,0.0044631944],"category_scores_gemma":[0.02053615,0.0005247456,0.0005857786,0.006613109,0.019803418,0.027646136,0.004443272,0.002873793,0.0002645759],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023698805,0.00012298014,0.013635928,0.0006197306,0.000030137151,0.001782719,0.14173321,0.00083646365,0.0028311089,0.69656307,0.0055959756,0.13601169],"study_design_scores_gemma":[0.00004072333,0.00016583149,0.027162628,0.001594107,0.00006137843,0.0013124602,0.27931094,0.0070881234,0.004199314,0.5550096,0.12393055,0.00012433977],"about_ca_topic_score_codex":0.015909288,"about_ca_topic_score_gemma":0.020164091,"teacher_disagreement_score":0.015909288,"about_ca_system_score_codex":0.0053009293,"about_ca_system_score_gemma":0.0058591887,"threshold_uncertainty_score":0.07523155},"labels":[],"label_agreement":null},{"id":"W2022050541","doi":"10.3115/1117794.1117801","title":"A uniform method of grammar extraction and its applications","year":2000,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":48,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Atomic Energy of Canada Limited; University of Pennsylvania","keywords":"Computer science; Rule-based machine translation; Natural language processing; Grammar; Artificial intelligence; Information extraction; Training set; Linguistics","score_opus":0.012604210457693226,"score_gpt":0.3212628948038822,"score_spread":0.308658684346189,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2022050541","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0006181325,0.00019893472,0.9931425,0.00008136073,0.000039689356,0.00013137062,0.00016568266,0.004557463,0.0010648457],"genre_scores_gemma":[0.017578438,0.00037041292,0.9762973,0.0001726462,0.00008866182,0.0004294368,0.00092912227,0.0016536863,0.0024802643],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9897765,0.002515632,0.0014027387,0.0036828471,0.0023060113,0.00031621562],"domain_scores_gemma":[0.98970467,0.0020585817,0.000322845,0.0058122086,0.0019203962,0.00018132447],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005030633,0.0014588715,0.0018110884,0.0047188303,0.0017663663,0.0035032423,0.002871497,0.002393783,0.009858548],"category_scores_gemma":[0.017527409,0.0013986389,0.0018831663,0.005528602,0.002059108,0.004956127,0.006131951,0.0036705486,0.010369547],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011712493,0.00016830543,0.0009898975,0.00054231717,0.00013237889,0.00025293895,0.00035036527,0.004709543,0.03488585,0.033848714,0.012154288,0.91184825],"study_design_scores_gemma":[0.00024504267,0.00032766856,0.0044143787,0.00045914226,0.00029921753,0.0047541284,0.0003879026,0.2631375,0.17063731,0.2003819,0.3546426,0.00031324223],"about_ca_topic_score_codex":0.0009636043,"about_ca_topic_score_gemma":0.0013355546,"teacher_disagreement_score":0.009858548,"about_ca_system_score_codex":0.0005962822,"about_ca_system_score_gemma":0.0021201628,"threshold_uncertainty_score":0.032980144},"labels":[],"label_agreement":null},{"id":"W2022433169","doi":"10.3115/1626355.1626383","title":"Rule-based translation with statistical phrase-based post-editing","year":2007,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":150,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Computer science; Machine translation; Phrase; Natural language processing; Artificial intelligence; Synchronous context-free grammar; Task (project management); Translation (biology); Example-based machine translation; Machine translation software usability; Transfer-based machine translation; Machine translation system; Evaluation of machine translation; Programming language; Engineering","score_opus":0.012734366135268288,"score_gpt":0.27414133325764045,"score_spread":0.26140696712237216,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2022433169","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01011703,0.00016641915,0.9680595,0.00013999232,0.00013906008,0.0002686106,0.0003304783,0.01804526,0.0027336152],"genre_scores_gemma":[0.09605584,0.00018177513,0.896994,0.00021955432,0.00017889569,0.00026115627,0.0016696372,0.0014140725,0.0030250426],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99625236,0.0010589671,0.0004527948,0.00072272687,0.0014003359,0.00011285238],"domain_scores_gemma":[0.99066836,0.0041985814,0.0005184761,0.0019985144,0.0024922383,0.00012386811],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002259215,0.0011937773,0.0013229758,0.001468798,0.0007541246,0.0017254641,0.0019212425,0.0009686342,0.006624129],"category_scores_gemma":[0.009034698,0.0006049581,0.001076192,0.0017417595,0.0007309226,0.0017036556,0.00095120654,0.0016132533,0.008793665],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00042943293,0.00044578032,0.001285457,0.00068218337,0.00027812968,0.000865915,0.00045973618,0.021086946,0.1552315,0.0077775763,0.011011232,0.8004462],"study_design_scores_gemma":[0.00035098108,0.0011670935,0.0033829736,0.00007646446,0.00047537487,0.0031630273,0.00021437954,0.525217,0.37532595,0.026575558,0.06368485,0.00036637633],"about_ca_topic_score_codex":0.0011896902,"about_ca_topic_score_gemma":0.0018054653,"teacher_disagreement_score":0.006624129,"about_ca_system_score_codex":0.00026607345,"about_ca_system_score_gemma":0.0010815039,"threshold_uncertainty_score":0.022159934},"labels":[],"label_agreement":null},{"id":"W2023973156","doi":"10.1016/s0278-2626(01)80066-x","title":"Semantic information is used by a deep dyslexic to parse compounds","year":2001,"lang":"en","type":"article","venue":"Brain and Cognition","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Psychology; Parsing; Natural language processing; Semantics (computer science); Cognitive psychology; Artificial intelligence; Linguistics; Cognitive science; Computer science; Programming language; Philosophy","score_opus":0.009838752397924037,"score_gpt":0.2500880690573021,"score_spread":0.24024931665937804,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2023973156","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.95226336,0.00017627222,0.031182278,0.0010024799,0.0002036509,0.00004283117,0.0010490565,0.0019390712,0.012140898],"genre_scores_gemma":[0.9749307,0.00012521936,0.01816368,0.00025148285,0.000027692795,0.000015380749,0.0007756448,0.00042454773,0.005285558],"study_design_codex":"bench_or_experimental","study_design_gemma":"observational","domain_scores_codex":[0.9995009,0.000070421374,0.000053565887,0.00020032427,0.00012212747,0.0000526505],"domain_scores_gemma":[0.9970939,0.0016126384,0.0003507356,0.00043726992,0.0003790962,0.0001263658],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00043048814,0.0009761233,0.0005016862,0.00068216375,0.0004964508,0.0022823764,0.00063207967,0.0015364146,0.007786473],"category_scores_gemma":[0.0041440707,0.0004903376,0.00048638866,0.00042184445,0.00094390026,0.0030067777,0.0007754379,0.001982516,0.0017541175],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018057206,0.0003184977,0.019369358,0.0005270668,0.00017034919,0.012237312,0.0018105152,0.0053013945,0.76466435,0.031789813,0.01167933,0.15032628],"study_design_scores_gemma":[0.00037579535,0.0021059162,0.047866434,0.0001441282,0.0005129929,0.03267782,0.003179244,0.15106413,0.6389585,0.0969843,0.025834192,0.00029645982],"about_ca_topic_score_codex":0.0026314973,"about_ca_topic_score_gemma":0.0019046374,"teacher_disagreement_score":0.007786473,"about_ca_system_score_codex":0.00045791248,"about_ca_system_score_gemma":0.0006640621,"threshold_uncertainty_score":0.026048362},"labels":[],"label_agreement":null},{"id":"W2024161645","doi":"10.1145/1822327.1822344","title":"Evolution of MARF and its NLP framework","year":2010,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"Mesothelioma Applied Research Foundation","keywords":"Computer science; Pipeline (software); Extensibility; Software; Artificial intelligence; Software engineering; Natural language processing; Programming language","score_opus":0.006060748154826161,"score_gpt":0.2575844887334029,"score_spread":0.25152374057857674,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2024161645","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0009810806,0.0016553636,0.9785387,0.0022723598,0.00030945882,0.00010790564,0.00042020885,0.0105687305,0.005146263],"genre_scores_gemma":[0.021762434,0.0015266029,0.96770734,0.00103303,0.00087603874,0.0002077788,0.0010531007,0.0023064418,0.0035272103],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9902796,0.0031498654,0.0006965045,0.002558069,0.0029511973,0.00036485333],"domain_scores_gemma":[0.9792677,0.010642965,0.0006349943,0.0039044179,0.0050403685,0.00050960464],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.022814343,0.0016483795,0.0025577073,0.0069747716,0.0016443321,0.008644092,0.006709521,0.0032983017,0.0075146733],"category_scores_gemma":[0.034723613,0.0015236071,0.0028519633,0.0034148004,0.004077164,0.013344277,0.0036098852,0.0065336605,0.006832272],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016262247,0.0001471728,0.00066618074,0.00066931604,0.00016306757,0.0003637601,0.00063124293,0.020219134,0.0038995168,0.36271957,0.032945685,0.57741266],"study_design_scores_gemma":[0.00006314892,0.00012373313,0.00050045724,0.00043324966,0.00011176672,0.0013003249,0.00022218119,0.2806738,0.008721032,0.31646323,0.39112177,0.00026527751],"about_ca_topic_score_codex":0.010787233,"about_ca_topic_score_gemma":0.005812377,"teacher_disagreement_score":0.022814343,"about_ca_system_score_codex":0.003402629,"about_ca_system_score_gemma":0.005258794,"threshold_uncertainty_score":0.12065524},"labels":[],"label_agreement":null},{"id":"W2024303237","doi":"10.1109/icdmw.2012.58","title":"An Ensemble-Based Named Entity Recognition Solution for Detecting Consumer Products","year":2012,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"CONTEST; Computer science; Conditional random field; Product (mathematics); Task (project management); Information retrieval; Artificial intelligence; Natural language processing; Data mining; Engineering; Mathematics","score_opus":0.041899253702016324,"score_gpt":0.2993572676404133,"score_spread":0.257458013938397,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2024303237","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024100032,0.0009630731,0.95264184,0.00042119695,0.0002471372,0.00015348011,0.002722161,0.016869076,0.0018819678],"genre_scores_gemma":[0.12223577,0.0005036343,0.85818535,0.00025749597,0.00019814142,0.0001795355,0.013032042,0.00038849222,0.005019577],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99862826,0.00019992009,0.00011284286,0.00052727864,0.00042483787,0.00010677758],"domain_scores_gemma":[0.998102,0.0005540163,0.00016224416,0.00045569596,0.00064814417,0.00007790137],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001867143,0.001182255,0.0011625444,0.0036509114,0.00090121053,0.0009862126,0.0016332134,0.0016889089,0.0015132373],"category_scores_gemma":[0.0033925779,0.00044568614,0.0012458415,0.0032291522,0.00023851464,0.0029460464,0.0012270666,0.0018427304,0.0024722328],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00033311176,0.00027231823,0.0057996,0.00021003245,0.00033593833,0.00034434514,0.00018634411,0.018199274,0.020925812,0.003646712,0.029042665,0.9207038],"study_design_scores_gemma":[0.000040446586,0.00017451368,0.0048289727,0.000047412974,0.0003000936,0.00063024723,0.00016875824,0.92105067,0.035102915,0.010612025,0.026961576,0.000082341],"about_ca_topic_score_codex":0.0044384697,"about_ca_topic_score_gemma":0.0077709192,"teacher_disagreement_score":0.0044384697,"about_ca_system_score_codex":0.00043270242,"about_ca_system_score_gemma":0.00097545964,"threshold_uncertainty_score":0.009874523},"labels":[],"label_agreement":null},{"id":"W2024889439","doi":"10.3758/s13423-011-0092-y","title":"Is more always better? Effects of semantic richness on lexical decision, speeded pronunciation, and semantic classification","year":2011,"lang":"en","type":"article","venue":"Psychonomic Bulletin & Review","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":177,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Psychology; Pronunciation; Lexical decision task; Linguistics; Semantic memory; Cognitive psychology; Natural language processing; Cognition; Neuroscience; Computer science","score_opus":0.028635854432548062,"score_gpt":0.29346984277031046,"score_spread":0.2648339883377624,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2024889439","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99347466,0.002317913,0.0003143357,0.00042780995,0.000051346495,0.000019311467,0.00012761369,0.000014982408,0.003251923],"genre_scores_gemma":[0.9966355,0.0010980447,0.0007800178,0.00028419858,0.0001356611,0.00001746688,0.00018413874,0.000055848195,0.0008091507],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9979705,0.00078014785,0.00016928405,0.0005164462,0.00042610674,0.00013746049],"domain_scores_gemma":[0.91700274,0.06774934,0.008479675,0.002745894,0.0011521643,0.002870116],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005316816,0.0004922626,0.0011110281,0.00061573705,0.00042704362,0.0032152338,0.00079197926,0.001113933,0.00746148],"category_scores_gemma":[0.044724975,0.00060761074,0.00072387076,0.0005591228,0.0018019885,0.0033450446,0.0014334202,0.0017766604,0.00071840174],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.06923436,0.0048640985,0.6566741,0.0017103911,0.0062379115,0.0015138601,0.0073523945,0.0033459885,0.073975675,0.0036784224,0.0027244103,0.16868843],"study_design_scores_gemma":[0.0006795823,0.0030615446,0.97658724,0.00012575122,0.0013538136,0.00065668114,0.0018841036,0.0017417385,0.0024831502,0.010187364,0.0011334071,0.00010550747],"about_ca_topic_score_codex":0.0009964519,"about_ca_topic_score_gemma":0.0015480268,"teacher_disagreement_score":0.00746148,"about_ca_system_score_codex":0.00027940798,"about_ca_system_score_gemma":0.0003712863,"threshold_uncertainty_score":0.028118312},"labels":[],"label_agreement":null},{"id":"W202532017","doi":"10.1007/978-3-642-23211-4_4","title":"The Generative Power of Probabilistic and Weighted Context-Free Grammars","year":2011,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Rule-based machine translation; Probabilistic logic; Context-free grammar; Generative grammar; Tree-adjoining grammar; Context-sensitive grammar; Artificial intelligence; Natural language processing; Programming language; Theoretical computer science","score_opus":0.013371662320256817,"score_gpt":0.2353362324880736,"score_spread":0.22196457016781676,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W202532017","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03416733,0.0013117664,0.9415339,0.0008842732,0.00009527758,0.000029415747,0.00020734308,0.00038223725,0.021388479],"genre_scores_gemma":[0.760517,0.0019468395,0.2176989,0.00049627695,0.0005249453,0.00016291665,0.00062254496,0.00072706916,0.01730354],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9964934,0.0015600654,0.00019778417,0.0005076128,0.0009796047,0.0002615753],"domain_scores_gemma":[0.97706443,0.01960591,0.00057148293,0.0017011539,0.000769827,0.00028731432],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00363073,0.00074093445,0.0010731536,0.0018140664,0.001291345,0.0037469903,0.0024682684,0.0016387413,0.004756536],"category_scores_gemma":[0.022338113,0.0017369493,0.0019476449,0.002018014,0.0049505876,0.007916873,0.002597704,0.0029983646,0.0008269417],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000025850628,0.000013687253,0.00023197904,0.00004300206,0.000021713695,0.000071631795,0.00026097355,0.024311962,0.0006989292,0.9580634,0.00042537422,0.015831467],"study_design_scores_gemma":[0.0000046924188,0.0000027693036,0.00004798264,0.0000073740284,0.000007509034,0.000029104358,0.000008780993,0.03748439,0.0001968694,0.9612821,0.00091975543,0.000008721803],"about_ca_topic_score_codex":0.0026904177,"about_ca_topic_score_gemma":0.0029507438,"teacher_disagreement_score":0.004756536,"about_ca_system_score_codex":0.0013461392,"about_ca_system_score_gemma":0.0012007201,"threshold_uncertainty_score":0.019201338},"labels":[],"label_agreement":null},{"id":"W2025437365","doi":"10.3115/1604263.1604271","title":"Control strategies for parsing with freer word-order languages","year":2006,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Word order; Computer science; Parsing; Word (group theory); Natural language processing; Bounding overwatch; Artificial intelligence; Grammar; Control (management); Top-down parsing; Order (exchange); Dependency grammar; Bottom-up parsing; Linguistics","score_opus":0.006693028682922127,"score_gpt":0.2547171451948669,"score_spread":0.24802411651194475,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2025437365","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0052037793,0.00022628924,0.9891939,0.00017129364,0.00003721616,0.000089703826,0.00010069862,0.0026269583,0.0023500465],"genre_scores_gemma":[0.18556052,0.00037942972,0.80739284,0.00028111425,0.00009347724,0.00037035422,0.0005859374,0.0018804454,0.0034558473],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9937698,0.0021308993,0.00059355045,0.0013117241,0.0016506503,0.0005433796],"domain_scores_gemma":[0.9852727,0.00893672,0.0007804121,0.0034331616,0.0012749765,0.0003020923],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0062684976,0.0017648999,0.0015926061,0.0030161536,0.0011193334,0.004744722,0.003735388,0.0020084516,0.007538538],"category_scores_gemma":[0.015556876,0.0012394381,0.0017717901,0.00198493,0.0040666927,0.007867701,0.0038577067,0.0025311837,0.0016438445],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023693696,0.00018980952,0.001017155,0.0008243063,0.00014093264,0.00048050986,0.0022810805,0.10043435,0.020819977,0.6063982,0.007032755,0.26014394],"study_design_scores_gemma":[0.00010029786,0.000084068946,0.00031914655,0.00012459245,0.000107322674,0.00020600813,0.00020869118,0.5311714,0.023362475,0.42708808,0.017080314,0.0001475919],"about_ca_topic_score_codex":0.0045081065,"about_ca_topic_score_gemma":0.006418413,"teacher_disagreement_score":0.007538538,"about_ca_system_score_codex":0.0019687028,"about_ca_system_score_gemma":0.0022740122,"threshold_uncertainty_score":0.03315139},"labels":[],"label_agreement":null},{"id":"W2026130194","doi":"10.3115/1220575.1220686","title":"Exploiting a verb lexicon in automatic semantic role labelling","year":2005,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":36,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Lexicon; Labelling; Computer science; Predicate (mathematical logic); Semantic role labeling; Natural language processing; Artificial intelligence; Verb; FrameNet; Set (abstract data type); Parsing; Programming language; Sentence","score_opus":0.011268959508530883,"score_gpt":0.25989530512333675,"score_spread":0.24862634561480587,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2026130194","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013402273,0.00008948585,0.97410405,0.0002115402,0.00007448611,0.00017379693,0.0006690995,0.006176972,0.0050983108],"genre_scores_gemma":[0.20953418,0.0002539995,0.77810663,0.00032900478,0.000101630176,0.00041780618,0.0061251214,0.0018564205,0.003275212],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.997163,0.0011222555,0.00017485605,0.0006342703,0.00072845846,0.00017709815],"domain_scores_gemma":[0.9926501,0.004570019,0.00045310825,0.0012182975,0.0009376399,0.00017080642],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032139597,0.001173745,0.0008292347,0.003657458,0.0014481664,0.0030818633,0.0020542773,0.0015856888,0.0068964474],"category_scores_gemma":[0.013303061,0.0012186888,0.0011701521,0.0020801963,0.0015309695,0.008697075,0.0028932928,0.0023442518,0.006059368],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004664234,0.0007487672,0.0050489902,0.0009781998,0.00014650644,0.0009623346,0.0018163351,0.034777567,0.1652751,0.14745526,0.029426545,0.61289805],"study_design_scores_gemma":[0.00012608466,0.00020797363,0.003091155,0.00014973294,0.00013105216,0.0010198327,0.00038324943,0.7067835,0.08151681,0.15365085,0.052749824,0.00018999746],"about_ca_topic_score_codex":0.00332763,"about_ca_topic_score_gemma":0.0071236407,"teacher_disagreement_score":0.0068964474,"about_ca_system_score_codex":0.0010526729,"about_ca_system_score_gemma":0.0023200645,"threshold_uncertainty_score":0.023070872},"labels":[],"label_agreement":null},{"id":"W2026222783","doi":"10.7202/1025047ar","title":"Translation Skill-Sets in a Machine-Translation Age","year":2014,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":241,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Terminology; Computer science; Machine translation; Space (punctuation); Translation (biology); Computer-assisted translation; Function (biology); Natural language processing; Artificial intelligence; Linguistics","score_opus":0.03194641009699978,"score_gpt":0.290480704647114,"score_spread":0.25853429455011423,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2026222783","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17807205,0.026135456,0.29203853,0.14635748,0.0023544482,0.00044727343,0.0005527867,0.0014212099,0.35262075],"genre_scores_gemma":[0.7655885,0.007965048,0.16594945,0.013884112,0.0016939332,0.0007077521,0.0005810594,0.00049251225,0.043137528],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9919612,0.003602993,0.00061588705,0.0016744196,0.001710941,0.0004345664],"domain_scores_gemma":[0.97505885,0.016229954,0.0010386452,0.0029088783,0.003346544,0.001417022],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.014094491,0.0008576661,0.0008013857,0.0036136417,0.0036136261,0.00884664,0.0019843527,0.003577325,0.009901158],"category_scores_gemma":[0.027775936,0.00058114505,0.0005274857,0.0013535854,0.019945705,0.015307756,0.008299931,0.0057994085,0.0044931704],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001540955,0.00026934798,0.0055169202,0.00079273846,0.000046794594,0.00049607764,0.06719696,0.00181592,0.0038616091,0.59980917,0.02039166,0.29964867],"study_design_scores_gemma":[0.00003549929,0.00018155637,0.006855746,0.0007318599,0.000016961263,0.00073940295,0.015448795,0.004000565,0.0022151086,0.7712152,0.19846316,0.00009619231],"about_ca_topic_score_codex":0.001777211,"about_ca_topic_score_gemma":0.0011649955,"teacher_disagreement_score":0.014094491,"about_ca_system_score_codex":0.0037361248,"about_ca_system_score_gemma":0.0045803757,"threshold_uncertainty_score":0.07453966},"labels":[],"label_agreement":null},{"id":"W2026539448","doi":"10.3115/1690299.1690305","title":"<i>Assas-Band</i>, an affix-exception-list based Urdu stemmer","year":2009,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":45,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"International Development Research Centre","keywords":"Affix; Urdu; Computer science; Artificial intelligence; Natural language processing; Root (linguistics); Word (group theory); Linguistics","score_opus":0.014217016648230045,"score_gpt":0.2824050739955239,"score_spread":0.26818805734729384,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2026539448","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09452788,0.0014808329,0.7450096,0.0006063183,0.00072677725,0.0006490322,0.0040302514,0.12674212,0.02622714],"genre_scores_gemma":[0.12877361,0.0004605091,0.8478235,0.00038835156,0.00012802934,0.00014026531,0.0065583885,0.004835204,0.010892031],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9991253,0.00014890514,0.00016700965,0.00021146536,0.00028264962,0.00006473873],"domain_scores_gemma":[0.99740124,0.0005515576,0.00034000346,0.0005409875,0.0010619618,0.000104199076],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010778136,0.0011813174,0.00090116385,0.0016902841,0.001098674,0.0017359979,0.001301501,0.00067166117,0.009385933],"category_scores_gemma":[0.0033128257,0.0004776669,0.0005199634,0.0013050839,0.00065096386,0.0022211296,0.0012232972,0.0007975165,0.010632302],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00087667815,0.00014884956,0.0048774364,0.0011412756,0.00008664914,0.0006667021,0.0010260183,0.0008061259,0.1964218,0.011644203,0.036449485,0.7458548],"study_design_scores_gemma":[0.00014821228,0.0007769235,0.0067400904,0.00019197223,0.00022293818,0.005246057,0.0006449804,0.03801037,0.6236125,0.012415913,0.31175226,0.00023769242],"about_ca_topic_score_codex":0.0008812177,"about_ca_topic_score_gemma":0.0022756713,"teacher_disagreement_score":0.009385933,"about_ca_system_score_codex":0.00034905018,"about_ca_system_score_gemma":0.000946181,"threshold_uncertainty_score":0.03139913},"labels":[],"label_agreement":null},{"id":"W2026719105","doi":"10.1177/1466138107076137","title":"Trans-scription as a social activity","year":2007,"lang":"en","type":"article","venue":"Ethnography","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":51,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Simon Fraser University; Université de Franche-Comté; Coordenação de Aperfeiçoamento de Pessoal de Nível Superior","keywords":"Transcription (linguistics); Sociology; Epistemology; Philosophy; Linguistics","score_opus":0.022760168815290173,"score_gpt":0.31646009067551756,"score_spread":0.29369992186022736,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2026719105","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.33796966,0.0013535222,0.41489258,0.010693735,0.0005881953,0.001304847,0.0009832388,0.0011389302,0.23107533],"genre_scores_gemma":[0.91699463,0.00044825344,0.060284596,0.00048538015,0.000168947,0.0005288475,0.00040770604,0.0005350534,0.020146621],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.95777303,0.034043215,0.0010357372,0.0029470057,0.0035322641,0.00066880893],"domain_scores_gemma":[0.93332046,0.046681654,0.0037658014,0.011419945,0.003566576,0.0012455526],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.018110935,0.0007276117,0.00053108303,0.0028230129,0.0056862873,0.010057281,0.0015695423,0.00135519,0.007960541],"category_scores_gemma":[0.042104863,0.00044407038,0.00049168017,0.0032281976,0.01955872,0.010560171,0.0075843837,0.0024377597,0.0020130246],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009934544,0.00005723331,0.006765073,0.00034317118,0.000034709454,0.0008597122,0.63769275,0.0004465195,0.008237024,0.2592779,0.0038395158,0.08234713],"study_design_scores_gemma":[0.000028039763,0.00017691839,0.008908436,0.000896763,0.0000482605,0.0021540343,0.32456866,0.0037431573,0.010641026,0.23464829,0.41405252,0.00013389671],"about_ca_topic_score_codex":0.0023321672,"about_ca_topic_score_gemma":0.0023853488,"teacher_disagreement_score":0.018110935,"about_ca_system_score_codex":0.0024608558,"about_ca_system_score_gemma":0.0031721503,"threshold_uncertainty_score":0.09578091},"labels":[],"label_agreement":null},{"id":"W2026757421","doi":"10.5539/cis.v2n4p55","title":"Unsupervised Coreference Resolution with HyperGraph Partitioning","year":2009,"lang":"en","type":"article","venue":"Computer and Information Science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"National Natural Science Foundation of China","keywords":"Hypergraph; Computer science; Coreference; Equivalence (formal languages); Artificial intelligence; Unsupervised learning; Resolution (logic); Property (philosophy); Data mining; Machine learning; Pattern recognition (psychology); Mathematics","score_opus":0.009861917736676244,"score_gpt":0.23358226383920083,"score_spread":0.2237203461025246,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2026757421","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008622898,0.00027580457,0.98805565,0.00012085019,0.00002303861,0.00017022513,0.00025637986,0.0013017933,0.0011733348],"genre_scores_gemma":[0.112458505,0.0002997987,0.8800132,0.00020637467,0.00005762972,0.0004216681,0.0029536402,0.00034268183,0.0032464785],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99643964,0.0014629463,0.00020247033,0.0012089445,0.0005236417,0.00016225889],"domain_scores_gemma":[0.99478406,0.00257383,0.0004523429,0.0013194144,0.00078947615,0.00008100689],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020038013,0.0012301046,0.0012182745,0.005812008,0.0015895693,0.0016229221,0.0028834161,0.0018212411,0.0017766587],"category_scores_gemma":[0.007013994,0.000749159,0.001621022,0.0056288447,0.0011238727,0.0035064744,0.0031431918,0.0021155484,0.0011847465],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024423318,0.00033751104,0.002673474,0.00037640048,0.00040365398,0.0004414253,0.001089178,0.0955539,0.019273391,0.026791489,0.013912696,0.8389026],"study_design_scores_gemma":[0.000057726393,0.00006647917,0.0019200237,0.00007141615,0.00014659596,0.00045775497,0.0005631379,0.843759,0.024369128,0.111394316,0.017118517,0.00007596716],"about_ca_topic_score_codex":0.005164467,"about_ca_topic_score_gemma":0.0117255,"teacher_disagreement_score":0.005812008,"about_ca_system_score_codex":0.0010631977,"about_ca_system_score_gemma":0.0018369571,"threshold_uncertainty_score":0.010597229},"labels":[],"label_agreement":null},{"id":"W2027081355","doi":"10.3115/1699648.1699697","title":"Character-level analysis of semi-structured documents for set expansion","year":2009,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":40,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Association of Canadian Universities for Research in Astronomy","keywords":"Set (abstract data type); Seal (emblem); Computer science; Character (mathematics); Independence (probability theory); Artificial intelligence; Character encoding; Natural language processing; Programming language; Mathematics","score_opus":0.022466725177752844,"score_gpt":0.31476229507146813,"score_spread":0.2922955698937153,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2027081355","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.22770236,0.0015378903,0.71133137,0.00075162976,0.0002796509,0.0009884002,0.014336416,0.030292569,0.012779649],"genre_scores_gemma":[0.36228254,0.00043121766,0.6170447,0.00012770276,0.00011168961,0.00035225716,0.01573407,0.0006815684,0.0032342745],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985813,0.00029534573,0.00019393014,0.0003016394,0.0005493925,0.00007844249],"domain_scores_gemma":[0.99182266,0.0041063097,0.0007650748,0.0011644295,0.0019072634,0.00023425947],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012370775,0.00061402674,0.00043967395,0.004192177,0.00051251135,0.0013163101,0.00056558906,0.0005315877,0.0036309545],"category_scores_gemma":[0.008608109,0.00021494039,0.00062175834,0.003208796,0.00037395585,0.002731076,0.00068307354,0.0008580568,0.002967746],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006311464,0.00042651835,0.017921207,0.0012136063,0.0001509837,0.0009167262,0.0012963907,0.015330318,0.09480721,0.012800629,0.030589674,0.8239156],"study_design_scores_gemma":[0.00007149537,0.00039962307,0.030093163,0.0002310354,0.00014866154,0.0020827265,0.0010983291,0.7002153,0.1742351,0.01676229,0.07451085,0.00015145302],"about_ca_topic_score_codex":0.0010409104,"about_ca_topic_score_gemma":0.0018156394,"teacher_disagreement_score":0.004192177,"about_ca_system_score_codex":0.0006606612,"about_ca_system_score_gemma":0.0007703537,"threshold_uncertainty_score":0.012146771},"labels":[],"label_agreement":null},{"id":"W2027097848","doi":"10.3115/1626516.1626533","title":"Creating a comparative dictionary of Totonac-Tepehua","year":2007,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Wenner-Gren Foundation","keywords":"Cognate; Computer science; Language family; Identification (biology); Natural language processing; Artificial intelligence; Indigenous; Linguistics","score_opus":0.02095664878535249,"score_gpt":0.31781124877150946,"score_spread":0.296854599986157,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2027097848","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.52975535,0.0015208446,0.35563794,0.00055062433,0.0005879555,0.00059868564,0.013474675,0.0023108225,0.09556309],"genre_scores_gemma":[0.6097006,0.00077712943,0.36662573,0.000085919084,0.000054545362,0.00036361633,0.013042158,0.00059676083,0.008753491],"study_design_codex":"design_other","study_design_gemma":"design_other","domain_scores_codex":[0.9996674,0.00004744012,0.000045326262,0.00014393286,0.00007280283,0.000023182358],"domain_scores_gemma":[0.9992719,0.00018692012,0.00005852551,0.00019246832,0.00025262526,0.00003751642],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00036282555,0.00030867365,0.00040224585,0.0028784522,0.001316248,0.00081623776,0.00053992716,0.00026131785,0.0072319694],"category_scores_gemma":[0.0022578936,0.00021824145,0.00020866253,0.0023887393,0.00072419154,0.0011972996,0.0010544539,0.00054255483,0.0012449927],"study_design_candidate":"design_other","study_design_consensus":"design_other","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00041143427,0.00014720215,0.016116686,0.00095285586,0.000068910325,0.0025701576,0.009808673,0.003534395,0.056120478,0.09175759,0.016592698,0.80191886],"study_design_scores_gemma":[0.00010286097,0.00045150344,0.07180825,0.0007028253,0.00021073059,0.010507802,0.014907084,0.040832732,0.036378827,0.04517128,0.7787322,0.0001939402],"about_ca_topic_score_codex":0.0040861927,"about_ca_topic_score_gemma":0.0106185675,"teacher_disagreement_score":0.0072319694,"about_ca_system_score_codex":0.00055249315,"about_ca_system_score_gemma":0.0013858924,"threshold_uncertainty_score":0.024193406},"labels":[],"label_agreement":null},{"id":"W2028307959","doi":"10.7202/002489ar","title":"Computerised Terminological Databases for Translators Who Use Word Processors","year":2002,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Glossary; Computer science; Word (group theory); Mode (computer interface); Linguistics; Romanization; Software; Space (punctuation); Word processing; Natural language processing; Database; Programming language; Human–computer interaction","score_opus":0.11657613542621476,"score_gpt":0.3068647469268147,"score_spread":0.19028861150059995,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2028307959","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07077974,0.0009049149,0.8874074,0.0005490867,0.00020691873,0.0010024385,0.0034099566,0.016649673,0.019089878],"genre_scores_gemma":[0.16919944,0.0010110443,0.7998302,0.00021994392,0.00010414355,0.0009897794,0.010149032,0.005180806,0.013315521],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9956917,0.0016872168,0.00095529837,0.0007168917,0.0007695396,0.00017936416],"domain_scores_gemma":[0.98270357,0.009115843,0.00073960633,0.004241382,0.0028448105,0.0003548682],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0056251925,0.00074751023,0.0014196477,0.0031613833,0.0018178868,0.0046702377,0.0020435872,0.0011921074,0.015980748],"category_scores_gemma":[0.022857117,0.0012454736,0.00077736226,0.004343477,0.0013435059,0.008770615,0.003517064,0.0019104758,0.010436795],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0027523497,0.00034220063,0.006281048,0.0019129094,0.00008022787,0.0019626797,0.018553304,0.004854599,0.07761135,0.10079446,0.03036672,0.75448817],"study_design_scores_gemma":[0.0007997743,0.0010174048,0.0055119153,0.00092117,0.00027313468,0.0033462336,0.011690034,0.054988306,0.14986894,0.06609148,0.70511264,0.0003790288],"about_ca_topic_score_codex":0.0010643521,"about_ca_topic_score_gemma":0.0015799309,"teacher_disagreement_score":0.015980748,"about_ca_system_score_codex":0.0013284481,"about_ca_system_score_gemma":0.0020865132,"threshold_uncertainty_score":0.053460956},"labels":[],"label_agreement":null},{"id":"W2028492537","doi":"10.3115/1067807.1067858","title":"Efficient search for interactive statistical machine translation","year":2003,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":55,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Machine translation; Prefix; Task (project management); Translation (biology); Quality (philosophy); Representation (politics); Process (computing); Artificial intelligence; Word (group theory); Natural language processing; Extension (predicate logic); Example-based machine translation; Human–computer interaction; Programming language","score_opus":0.02176082850636463,"score_gpt":0.3361458189816818,"score_spread":0.31438499047531715,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2028492537","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029040582,0.00042260656,0.96013033,0.00030345572,0.000024662932,0.00009372914,0.00033994816,0.0064437627,0.0032009466],"genre_scores_gemma":[0.430443,0.00027004362,0.5610131,0.0001965293,0.00006202869,0.00043978787,0.0024552003,0.0010363297,0.004084048],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99788123,0.0011225381,0.00009398913,0.00023291704,0.00051418244,0.00015513948],"domain_scores_gemma":[0.9953696,0.0035211407,0.00014708615,0.00059066416,0.00027819883,0.000093216404],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014213502,0.00079036626,0.0014512872,0.0019656508,0.0007617489,0.0011585938,0.0017528223,0.0015606491,0.008951058],"category_scores_gemma":[0.011119249,0.0006424182,0.0009048704,0.0038528042,0.00096006744,0.0030142958,0.002547543,0.0010733922,0.002048274],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011432075,0.0002923137,0.0011456051,0.0004205613,0.00015454052,0.0004743909,0.00051917415,0.3738949,0.015608071,0.09766736,0.01973092,0.48894897],"study_design_scores_gemma":[0.000057943107,0.000045020395,0.00015044758,0.000005426481,0.000014737966,0.000050885203,0.00004758508,0.9488458,0.0018665446,0.046478227,0.0024272986,0.000010023695],"about_ca_topic_score_codex":0.0037724106,"about_ca_topic_score_gemma":0.006565304,"teacher_disagreement_score":0.008951058,"about_ca_system_score_codex":0.000872107,"about_ca_system_score_gemma":0.0013591736,"threshold_uncertainty_score":0.0299443},"labels":[],"label_agreement":null},{"id":"W2028803755","doi":"10.1017/s0140525x00423241","title":"Implausibility versus misinterpretation of the FLMP","year":2000,"lang":"en","type":"article","venue":"Behavioral and Brain Sciences","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Merge (version control); Perception; Independence (probability theory); Epistemology; Psychology; Quarter (Canadian coin); Cognitive psychology; Philosophy; History; Computer science; Mathematics","score_opus":0.035737767638336713,"score_gpt":0.3431943815085803,"score_spread":0.3074566138702436,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2028803755","genre_codex":"methods","genre_gemma":"commentary","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"commentary","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.082506515,0.0032215004,0.65860313,0.10558696,0.0025980128,0.00007184287,0.00054498273,0.0033478278,0.14351927],"genre_scores_gemma":[0.9332877,0.00084346544,0.046744104,0.0069532967,0.00126692,0.00008183821,0.00020316867,0.0007103975,0.009909218],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.990743,0.0035959268,0.0007006841,0.0019349756,0.0024834538,0.0005419803],"domain_scores_gemma":[0.9789556,0.010330112,0.0015009097,0.0064807525,0.0023770249,0.00035553187],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0076639885,0.00066935684,0.0007212483,0.0017752453,0.0017042388,0.007255325,0.003374506,0.005132959,0.0078114234],"category_scores_gemma":[0.039528333,0.00075088337,0.0010108901,0.00093030854,0.017809784,0.025395567,0.007136661,0.008832728,0.0015578388],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013874892,0.00002060667,0.0011659247,0.00014199212,0.00003611004,0.00064049044,0.0069584073,0.00050254574,0.0016514125,0.9498816,0.008382615,0.03047954],"study_design_scores_gemma":[0.000016835274,0.00001987818,0.0005205822,0.00007080714,0.000026212738,0.0007712748,0.0014421945,0.004171751,0.0027077014,0.96589744,0.024318386,0.000036924623],"about_ca_topic_score_codex":0.0014177503,"about_ca_topic_score_gemma":0.0009518638,"teacher_disagreement_score":0.0078114234,"about_ca_system_score_codex":0.0018728328,"about_ca_system_score_gemma":0.0010791734,"threshold_uncertainty_score":0.040531576},"labels":[],"label_agreement":null},{"id":"W2029070039","doi":"10.1017/s135132490600444x","title":"A general feature space for automatic verb classification","year":2006,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":88,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canada Research Chairs; University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Lexicon; Artificial intelligence; Feature vector; Natural language processing; Feature (linguistics); Redundancy (engineering); Verb; Support vector machine; Variety (cybernetics); Feature selection; Machine learning; Linguistics","score_opus":0.004724068616820104,"score_gpt":0.235974427016896,"score_spread":0.2312503584000759,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2029070039","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.044369638,0.00027851967,0.9502981,0.00016974821,0.000037572052,0.00010488867,0.000541235,0.0036016258,0.00059860706],"genre_scores_gemma":[0.5600121,0.00014980524,0.43583864,0.00010141822,0.00005618458,0.00043437196,0.0020776591,0.00017356992,0.00115623],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987551,0.0004867452,0.000114606795,0.00024283855,0.00029761318,0.00010305925],"domain_scores_gemma":[0.9972331,0.0015768723,0.00016602242,0.00034270325,0.000616604,0.00006459888],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018648098,0.0009028724,0.0010007242,0.0018399539,0.00045359973,0.0012136168,0.0011235698,0.0009818068,0.0030435824],"category_scores_gemma":[0.00506897,0.00030235937,0.00095736916,0.001673197,0.00062947563,0.0019636622,0.0011074941,0.0010151356,0.0010098986],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00055014394,0.00030495858,0.002765108,0.00020595189,0.00008313221,0.00012717948,0.00009866509,0.10855377,0.020340905,0.009618376,0.007277318,0.8500745],"study_design_scores_gemma":[0.0000258956,0.00010972236,0.0008984255,0.000019653195,0.000013418455,0.00006353913,0.000029687442,0.98262066,0.005804028,0.008624844,0.0017711753,0.000019038669],"about_ca_topic_score_codex":0.0021091036,"about_ca_topic_score_gemma":0.00124882,"teacher_disagreement_score":0.0030435824,"about_ca_system_score_codex":0.00064588286,"about_ca_system_score_gemma":0.0007808309,"threshold_uncertainty_score":0.010181844},"labels":[],"label_agreement":null},{"id":"W2029354160","doi":"10.1145/2668260.2668267","title":"A Flexible Approach for Text Processing Engineering","year":2014,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Trois-Rivières","funders":"","keywords":"Computer science; Modular design; Consistency (knowledge bases); Text processing; Set (abstract data type); Rule-based machine translation; Chain (unit); Programming language; Sequence (biology); Theoretical computer science; Natural language processing; Artificial intelligence","score_opus":0.012068898417659324,"score_gpt":0.2481579020891778,"score_spread":0.23608900367151847,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2029354160","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0010978858,0.00009323275,0.9961312,0.00020838334,0.000024313098,0.000070489026,0.00002463834,0.00035354425,0.0019963074],"genre_scores_gemma":[0.040890936,0.00025250905,0.9526336,0.00021744527,0.00007916891,0.000289183,0.00016817922,0.00034257802,0.005126335],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99397635,0.0016780548,0.00046093864,0.0014110035,0.0021987802,0.00027493367],"domain_scores_gemma":[0.99454826,0.0016515289,0.00032607923,0.0024642206,0.0008008359,0.00020907295],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004204216,0.0010134663,0.0006387566,0.0025173093,0.0018744102,0.004908104,0.0036949764,0.0017956593,0.005798075],"category_scores_gemma":[0.0073921983,0.0010116666,0.0027319267,0.0014707295,0.005186206,0.0066280956,0.0049901074,0.0037304016,0.0025971096],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000053026302,0.000080754035,0.00042104948,0.0003029529,0.0000663105,0.00034813656,0.0013676485,0.014516889,0.014106425,0.85841656,0.0022452814,0.10807499],"study_design_scores_gemma":[0.000038326132,0.00012502736,0.0002606458,0.00018273096,0.00007527948,0.0004969723,0.00034354246,0.092573024,0.012718437,0.7911057,0.101993605,0.00008662516],"about_ca_topic_score_codex":0.0014418002,"about_ca_topic_score_gemma":0.0013108485,"teacher_disagreement_score":0.005798075,"about_ca_system_score_codex":0.0016312577,"about_ca_system_score_gemma":0.0023479315,"threshold_uncertainty_score":0.022234261},"labels":[],"label_agreement":null},{"id":"W2029364148","doi":"10.1109/ifsa-nafips.2013.6608433","title":"Fuzzy semantic similarity in linked data using wikipedia infobox","year":2013,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Semantic similarity; Computer science; RDF; Information retrieval; Semantic Web; Similarity (geometry); Linked data; Ontology; Schema (genetic algorithms); Fuzzy logic; Data mining; Artificial intelligence","score_opus":0.05206380192315492,"score_gpt":0.3124514275020841,"score_spread":0.2603876255789292,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2029364148","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19819269,0.00095922605,0.79328245,0.00015610179,0.00008718178,0.0002782206,0.0012247639,0.0013943958,0.004425059],"genre_scores_gemma":[0.651076,0.0003322462,0.34588107,0.000033706612,0.000047019013,0.00020127874,0.0015480191,0.00008364629,0.00079696474],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9973188,0.00077378517,0.0003289791,0.00039465394,0.0011101817,0.000073636045],"domain_scores_gemma":[0.99637705,0.001737736,0.0004910769,0.0004355383,0.00083134056,0.00012737123],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018878075,0.00041026148,0.0006468423,0.009799068,0.0006819987,0.0024607026,0.0005611518,0.0006897177,0.00092256686],"category_scores_gemma":[0.010486063,0.00018794395,0.0006526828,0.005225311,0.0004744,0.003828264,0.0015657232,0.00033735152,0.0003201325],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011055062,0.00046165072,0.019365693,0.00112634,0.0006328977,0.0009235828,0.0022348075,0.07218259,0.044836823,0.06506204,0.0046156337,0.78745246],"study_design_scores_gemma":[0.00004740615,0.0002498947,0.01864081,0.00022697668,0.0002274905,0.00082497625,0.0015040536,0.8117678,0.050590366,0.103097096,0.012645674,0.00017750735],"about_ca_topic_score_codex":0.0020439024,"about_ca_topic_score_gemma":0.0022356468,"teacher_disagreement_score":0.009799068,"about_ca_system_score_codex":0.00084365305,"about_ca_system_score_gemma":0.00048531772,"threshold_uncertainty_score":0.009983838},"labels":[],"label_agreement":null},{"id":"W2029410861","doi":"10.7202/002108ar","title":"Problèmes de traduction automatique dans les sous-langages des bulletins d’avalanches","year":2002,"lang":"fr","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Philosophy; Physics","score_opus":0.05609771423875102,"score_gpt":0.2731172639649021,"score_spread":0.2170195497261511,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2029410861","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15824926,0.0015105469,0.80855775,0.0017768802,0.00040379135,0.0003239625,0.0005830955,0.017898465,0.010696244],"genre_scores_gemma":[0.47667247,0.001018263,0.47301865,0.0004026575,0.00013505016,0.00023369156,0.0013749063,0.004449531,0.042694807],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99534106,0.001308379,0.0005561989,0.00094002037,0.001636747,0.00021756337],"domain_scores_gemma":[0.98337793,0.0075392257,0.001403522,0.0043610916,0.0030076627,0.00031047597],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0047150943,0.0009979027,0.0008735587,0.0013046527,0.0015568257,0.0036967583,0.001667587,0.0015082892,0.00637428],"category_scores_gemma":[0.017813422,0.0010137168,0.0010681215,0.0014086135,0.002240479,0.0055362904,0.0023135033,0.0021575235,0.0026952699],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010601709,0.00027433428,0.0163697,0.0015128385,0.00015930411,0.0026980247,0.02410364,0.019810952,0.10091591,0.07461048,0.01338906,0.7450957],"study_design_scores_gemma":[0.00021018455,0.0007041163,0.013526281,0.00056824693,0.00044184926,0.004112724,0.007986593,0.13104387,0.24328998,0.061399974,0.5363099,0.00040635164],"about_ca_topic_score_codex":0.008210314,"about_ca_topic_score_gemma":0.0077290665,"teacher_disagreement_score":0.008210314,"about_ca_system_score_codex":0.0014748626,"about_ca_system_score_gemma":0.0018700697,"threshold_uncertainty_score":0.02493614},"labels":[],"label_agreement":null},{"id":"W2030760474","doi":"10.1145/1076034.1076085","title":"Linear discriminant model for information retrieval","year":2005,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":101,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Discriminative model; Artificial intelligence; Language model; Linear discriminant analysis; Rank (graph theory); Pattern recognition (psychology); Probabilistic logic; Natural language processing; Hidden Markov model; Component (thermodynamics); Machine learning; Mathematics","score_opus":0.02011058129652285,"score_gpt":0.2907212348309982,"score_spread":0.27061065353447533,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2030760474","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004585556,0.002601151,0.98432285,0.0009776849,0.0001993205,0.00008115068,0.00062801,0.00095877703,0.005645494],"genre_scores_gemma":[0.49896407,0.0066614845,0.4366828,0.0012591644,0.0012382858,0.0008837535,0.004341965,0.00050371734,0.04946478],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9980248,0.0007859038,0.00008925288,0.00044720937,0.00052281405,0.00013010464],"domain_scores_gemma":[0.9982487,0.00088776875,0.00014958184,0.000308282,0.00035969127,0.000045949193],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025281732,0.0012829045,0.0016363676,0.0018788278,0.00048545652,0.0017916441,0.0023596585,0.0017506542,0.0059730946],"category_scores_gemma":[0.006736124,0.0004099823,0.0011418071,0.0025403388,0.0008826576,0.0031490389,0.0011516771,0.0020136272,0.006353791],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002984169,0.00022457459,0.0026622086,0.0005005506,0.00025985797,0.00024540923,0.0001722457,0.33776766,0.003218498,0.30259275,0.030097924,0.32195985],"study_design_scores_gemma":[0.000019195511,0.0000429336,0.00035018366,0.000018454792,0.000027103562,0.000112025,0.00001535528,0.91023606,0.00041441765,0.07851442,0.01022045,0.00002940224],"about_ca_topic_score_codex":0.005187355,"about_ca_topic_score_gemma":0.0032062058,"teacher_disagreement_score":0.0059730946,"about_ca_system_score_codex":0.0016476073,"about_ca_system_score_gemma":0.0010032601,"threshold_uncertainty_score":0.01998204},"labels":[],"label_agreement":null},{"id":"W2031287800","doi":"10.7202/002996ar","title":"Machine Translation Research in Czechoslovakia","year":2002,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Czech; Computer science; Parsing; Closeness; Natural language processing; Machine translation; Artificial intelligence; Representation (politics); Word (group theory); Translation (biology); Linguistics; Political science","score_opus":0.13714930948816523,"score_gpt":0.35610632061176795,"score_spread":0.21895701112360272,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2031287800","genre_codex":"review","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10849997,0.52012455,0.06776675,0.029181376,0.0039565326,0.00033325635,0.0018090329,0.002361774,0.26596686],"genre_scores_gemma":[0.6785058,0.16740242,0.08555293,0.0016527075,0.0006772402,0.00036278728,0.0023970988,0.0007712318,0.062677816],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99658394,0.00085134804,0.0005780585,0.0009995701,0.00067262317,0.0003143593],"domain_scores_gemma":[0.9978059,0.000764833,0.00025164575,0.00032239495,0.0006606351,0.00019456081],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004075712,0.0005585795,0.0011614115,0.0039256006,0.0025592525,0.0055904603,0.0009061909,0.0011727823,0.004236589],"category_scores_gemma":[0.0047449116,0.0005738072,0.0010477229,0.005779008,0.0027722558,0.0041870903,0.0018252564,0.0016660492,0.0015583475],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00034874474,0.00011069739,0.0034291882,0.0034383866,0.00015903419,0.0013527642,0.0035984754,0.008110782,0.02486969,0.43839884,0.018771429,0.49741203],"study_design_scores_gemma":[0.000097188095,0.00014764201,0.017897107,0.0014089403,0.00011745659,0.0011376855,0.0011846026,0.004099945,0.010578588,0.047064126,0.91610456,0.0001621017],"about_ca_topic_score_codex":0.038486507,"about_ca_topic_score_gemma":0.029162392,"teacher_disagreement_score":0.038486507,"about_ca_system_score_codex":0.010545145,"about_ca_system_score_gemma":0.015172439,"threshold_uncertainty_score":0.07652497},"labels":[],"label_agreement":null},{"id":"W2031339321","doi":"10.7202/003508ar","title":"Méthode d'accès informatisé aux combinaisons lexicales en langue technique","year":2002,"lang":"fr","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Philosophy","score_opus":0.08404967293760288,"score_gpt":0.3062588380869818,"score_spread":0.2222091651493789,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2031339321","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0030283218,0.00025152363,0.9886635,0.00014953833,0.00007120089,0.0001535965,0.00022419116,0.005571004,0.001887193],"genre_scores_gemma":[0.03016835,0.00036079387,0.9617557,0.00008895632,0.00004453986,0.000344661,0.0007841543,0.00089104444,0.0055617546],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9954768,0.0011096235,0.0005911961,0.0010654894,0.0015075846,0.0002493206],"domain_scores_gemma":[0.99520797,0.0023363044,0.00022980187,0.0008482746,0.0012957243,0.00008183483],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003373533,0.0015239698,0.0017851535,0.005299939,0.0013614369,0.0058648796,0.0017112414,0.0016317376,0.015116445],"category_scores_gemma":[0.012796445,0.001114977,0.0022922591,0.004461267,0.0014896458,0.0047893543,0.0032250953,0.0022357702,0.006793557],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003196681,0.00011999854,0.002086593,0.00083347224,0.00021274135,0.00046025505,0.002876878,0.0035224524,0.0318171,0.050908662,0.008108292,0.89873403],"study_design_scores_gemma":[0.00034608704,0.0004968643,0.0056735436,0.0008711379,0.00088363944,0.0028767888,0.0035563258,0.22583777,0.13472256,0.15951948,0.46477866,0.00043717658],"about_ca_topic_score_codex":0.0072983303,"about_ca_topic_score_gemma":0.005373195,"teacher_disagreement_score":0.015116445,"about_ca_system_score_codex":0.0010352299,"about_ca_system_score_gemma":0.0020370851,"threshold_uncertainty_score":0.050569594},"labels":[],"label_agreement":null},{"id":"W2031535158","doi":"","title":"TransType: text prediction for translators","year":2002,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Context (archaeology); Machine translation; Natural language processing; Artificial intelligence; Translation (biology); Text generation; Machine learning","score_opus":0.017352588440629098,"score_gpt":0.24285459167008214,"score_spread":0.22550200322945305,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2031535158","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.044488516,0.0005512262,0.68830484,0.0016184266,0.0008311813,0.0005718237,0.007084911,0.23612347,0.020425668],"genre_scores_gemma":[0.37617642,0.0004379488,0.552348,0.00087148364,0.0005842021,0.0010398834,0.015266378,0.01358221,0.039693452],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9977234,0.00087503484,0.00016353447,0.0006312929,0.00050251104,0.00010423652],"domain_scores_gemma":[0.9869735,0.006727089,0.0007033033,0.0030688737,0.0020828226,0.00044429835],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031077287,0.0019926813,0.0008354218,0.00093107467,0.0008562343,0.0018233176,0.0022265585,0.0016634214,0.029367846],"category_scores_gemma":[0.018703822,0.00049383444,0.0006127423,0.0009810621,0.00056975253,0.004491074,0.002205943,0.0016827286,0.018828055],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0034332257,0.00045138496,0.009786993,0.0010298169,0.00009307149,0.00083172665,0.0016480426,0.009206686,0.03309112,0.010991028,0.15765718,0.7717797],"study_design_scores_gemma":[0.00044452035,0.001113643,0.004569828,0.00020567363,0.00016481578,0.0015739113,0.00072000624,0.5486878,0.17122394,0.0358232,0.23519047,0.0002822057],"about_ca_topic_score_codex":0.0013337485,"about_ca_topic_score_gemma":0.0019239986,"teacher_disagreement_score":0.029367846,"about_ca_system_score_codex":0.0004472528,"about_ca_system_score_gemma":0.001039024,"threshold_uncertainty_score":0.0982452},"labels":[],"label_agreement":null},{"id":"W2031539887","doi":"10.7202/002337ar","title":"Un dictionnaire électronique pour la reconnaissance des formes dérivées","year":2002,"lang":"fr","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Philosophy; Art","score_opus":0.04871968185388355,"score_gpt":0.27436460211425573,"score_spread":0.2256449202603722,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2031539887","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012564542,0.00052014104,0.9771624,0.00036871634,0.000079390615,0.00007590794,0.0007625185,0.0017314166,0.006734955],"genre_scores_gemma":[0.18531322,0.0011832048,0.79750913,0.00020639617,0.000057292295,0.00016397846,0.0013634509,0.0007885884,0.013414756],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99931276,0.00015839319,0.0000823979,0.00017309077,0.0002373803,0.000035913068],"domain_scores_gemma":[0.99847656,0.00076769106,0.00006550603,0.00038934214,0.00026885842,0.000032108015],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00076810113,0.0007259179,0.000834852,0.0015019759,0.0010668626,0.003295939,0.0012958186,0.0014530421,0.01213664],"category_scores_gemma":[0.00446006,0.00081279676,0.0018102359,0.0012402436,0.0016044,0.004317801,0.0015649974,0.0017137204,0.0029773857],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00034106098,0.00006673776,0.0058550336,0.0011833004,0.0001642329,0.0013945551,0.0024960053,0.14009376,0.022310706,0.55548555,0.008721748,0.26188728],"study_design_scores_gemma":[0.00008756401,0.00012236316,0.002486103,0.00034616716,0.00018357259,0.001991628,0.0008829813,0.62705415,0.029022269,0.18036471,0.15733197,0.00012649107],"about_ca_topic_score_codex":0.009388718,"about_ca_topic_score_gemma":0.012692753,"teacher_disagreement_score":0.01213664,"about_ca_system_score_codex":0.0012181329,"about_ca_system_score_gemma":0.00094619085,"threshold_uncertainty_score":0.040601075},"labels":[],"label_agreement":null},{"id":"W2032008361","doi":"10.1111/j.1540-4781.2007.00593_15.x","title":"<i>Corpus Linguistics: Readings in a Widening Discipline</i> edited by SAMPSON, GEOFFREY, &amp; DIANA MCCARTHY","year":2007,"lang":"en","type":"article","venue":"Modern Language Journal","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Linguistics; Applied linguistics; Corpus linguistics; Sociology; Philosophy; Psychology","score_opus":0.008623675641908787,"score_gpt":0.2762730224021542,"score_spread":0.2676493467602454,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2032008361","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00021162676,0.92255265,0.0039711953,0.029411228,0.014971872,0.00005491911,0.00044133226,0.00023449089,0.02815081],"genre_scores_gemma":[0.006113721,0.87028444,0.007518779,0.012757385,0.019165177,0.00029085684,0.0012230008,0.0007980707,0.08184858],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9971635,0.0010218291,0.00025596734,0.00046275713,0.00095672684,0.00013933156],"domain_scores_gemma":[0.9919337,0.0048737223,0.0006291523,0.00029812302,0.0019406667,0.0003245258],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003363403,0.0024808282,0.002462808,0.0068055526,0.0023614147,0.009700472,0.002483369,0.0026110676,0.032434385],"category_scores_gemma":[0.010733673,0.0016279698,0.00096876064,0.012386432,0.0047188755,0.012094328,0.0037490486,0.00563305,0.017015647],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000016020698,0.000009567136,0.00012662828,0.0014629568,0.000020782014,0.00007169777,0.0014413475,0.000062933526,0.00020538925,0.010315706,0.89573276,0.09053428],"study_design_scores_gemma":[0.000004009434,0.000006691207,0.0004079766,0.0023653456,0.000011351947,0.00022989119,0.00079940615,0.000055075892,0.00009195354,0.008095662,0.9879167,0.000016049442],"about_ca_topic_score_codex":0.0116767865,"about_ca_topic_score_gemma":0.019077504,"teacher_disagreement_score":0.032434385,"about_ca_system_score_codex":0.0041611036,"about_ca_system_score_gemma":0.00695368,"threshold_uncertainty_score":0.10850382},"labels":[],"label_agreement":null},{"id":"W2032473943","doi":"10.3115/1119212.1119216","title":"Towards a framework for learning structured shape models from text-annotated images","year":2003,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Artificial intelligence; Image (mathematics); Set (abstract data type); Natural language processing; Translation (biology); Word (group theory); Structured prediction; Probabilistic logic; Segmentation; Image segmentation; Image translation; Training set; Computer vision; Pattern recognition (psychology); Mathematics","score_opus":0.01857645544661559,"score_gpt":0.287877980988914,"score_spread":0.2693015255422984,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2032473943","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.002170457,0.00025052787,0.9948732,0.00025217485,0.000023156683,0.00009674436,0.0004144656,0.0015082685,0.00041093052],"genre_scores_gemma":[0.05645149,0.0005554719,0.934617,0.00044981987,0.0001367522,0.00077068433,0.0044453037,0.0004264578,0.0021470084],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9978677,0.00066867116,0.00013072688,0.0008402931,0.00038690583,0.00010564024],"domain_scores_gemma":[0.99537563,0.0023167948,0.00053614675,0.0009363908,0.0006193948,0.00021570655],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003762233,0.0021753977,0.0020846832,0.003854501,0.0010434372,0.003507173,0.006663484,0.004365664,0.0037637078],"category_scores_gemma":[0.01021185,0.0018868364,0.004239954,0.0044560796,0.0022764262,0.005073811,0.003503785,0.0042778067,0.002491797],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00031333088,0.00043536254,0.0028616341,0.0007443052,0.0004325133,0.0005837017,0.0007984666,0.4049783,0.0076902583,0.07365108,0.018461686,0.48904926],"study_design_scores_gemma":[0.00003580728,0.00007194269,0.00022424436,0.000051594303,0.00003436571,0.0001131473,0.00007305241,0.93029857,0.0013200819,0.06396463,0.0037841923,0.00002851246],"about_ca_topic_score_codex":0.010021472,"about_ca_topic_score_gemma":0.01702751,"teacher_disagreement_score":0.010021472,"about_ca_system_score_codex":0.0028574034,"about_ca_system_score_gemma":0.002132086,"threshold_uncertainty_score":0.020731986},"labels":[],"label_agreement":null},{"id":"W2032565104","doi":"10.1109/icassp.2014.6854537","title":"Abin-based ontological framework for low-resourcen-gram smoothing in language modelling","year":2014,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Moncton; Institut National de la Recherche Scientifique","funders":"","keywords":"Perplexity; Computer science; Language model; Natural language processing; Smoothing; Word error rate; Baseline (sea); Artificial intelligence; WordNet; Word (group theory); Ontology; Exploit; Speech recognition; Linguistics","score_opus":0.016801421352542992,"score_gpt":0.2869906196054115,"score_spread":0.2701891982528685,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2032565104","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0014022718,0.00006753409,0.9968432,0.00006859917,0.00003050341,0.000021680506,0.00012703477,0.0007679051,0.00067130034],"genre_scores_gemma":[0.12138388,0.00030501888,0.8708124,0.00017361925,0.00011100278,0.0003794588,0.001495401,0.0008782362,0.0044610742],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9981122,0.00086341717,0.0001384015,0.00031989696,0.0004504328,0.00011562059],"domain_scores_gemma":[0.9981086,0.00094023335,0.00013450846,0.00037524727,0.00036752113,0.00007385239],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026752506,0.00095330324,0.0008632574,0.0017204505,0.0011407498,0.0016031486,0.0018375692,0.0009323779,0.004622556],"category_scores_gemma":[0.006537416,0.0006621558,0.0015689571,0.001606755,0.0008153186,0.0025773076,0.0022761782,0.0025533498,0.0033438588],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002295618,0.00015370146,0.0015405396,0.00033636412,0.00023849377,0.00039458423,0.00087291736,0.30665797,0.014691098,0.33210316,0.007936929,0.3348447],"study_design_scores_gemma":[0.0000052505757,0.000014337906,0.00017643353,0.000018368706,0.000016960837,0.000043227003,0.000050125644,0.93443656,0.0016063814,0.056009363,0.0076014875,0.000021459491],"about_ca_topic_score_codex":0.0134095,"about_ca_topic_score_gemma":0.02121704,"teacher_disagreement_score":0.0134095,"about_ca_system_score_codex":0.0013036496,"about_ca_system_score_gemma":0.002137366,"threshold_uncertainty_score":0.026662886},"labels":[],"label_agreement":null},{"id":"W2032956935","doi":"10.3115/1613704.1613710","title":"Pulling their weight","year":2007,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":81,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Security token; Computer science; Natural language processing; Task (project management); Artificial intelligence; Context (archaeology); Expression (computer science); Identification (biology); Literal (mathematical logic); Interpretation (philosophy); Linguistics; Programming language","score_opus":0.010815755652227711,"score_gpt":0.2577908306875516,"score_spread":0.2469750750353239,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2032956935","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.099403925,0.0021706505,0.7909722,0.0034548074,0.004696031,0.0005166247,0.0020331964,0.0042686677,0.09248386],"genre_scores_gemma":[0.55573744,0.0027831357,0.3074356,0.00179272,0.0014974223,0.0005509009,0.003672315,0.004622489,0.12190796],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99762815,0.00024450364,0.00013714233,0.00067321106,0.0010211044,0.00029591937],"domain_scores_gemma":[0.99508107,0.0013267428,0.00026829954,0.001520293,0.0014097866,0.00039376784],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018121075,0.001766147,0.0014941116,0.004016215,0.0018461153,0.0036616272,0.0021900658,0.0019717661,0.037630796],"category_scores_gemma":[0.021634143,0.00086538336,0.0012999604,0.0034265642,0.0012965322,0.007279402,0.0048591164,0.0025193153,0.014407495],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006438215,0.00024968872,0.0074300305,0.0004539602,0.0002560508,0.00046526105,0.0006503551,0.014129866,0.021050919,0.12870163,0.04380162,0.78216684],"study_design_scores_gemma":[0.00014678763,0.00039554914,0.010061915,0.00042778032,0.0004758655,0.0011742255,0.0017138324,0.25294974,0.029149901,0.47103956,0.23224752,0.00021736135],"about_ca_topic_score_codex":0.0037104927,"about_ca_topic_score_gemma":0.0054376917,"teacher_disagreement_score":0.037630796,"about_ca_system_score_codex":0.00085722166,"about_ca_system_score_gemma":0.0014849289,"threshold_uncertainty_score":0.12588757},"labels":[],"label_agreement":null},{"id":"W2033332470","doi":"10.7202/003695ar","title":"Construction d'un dictionnaire : morphologie à deux niveaux pour le français à l'aide de contraintes basées sur les structures de traits typées","year":2002,"lang":"fr","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Physics; Philosophy; Art","score_opus":0.054438824003739894,"score_gpt":0.2526062630941788,"score_spread":0.19816743909043888,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2033332470","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16518447,0.0010754718,0.75767297,0.0017324863,0.0004084718,0.00027006498,0.0030643975,0.00518972,0.065401904],"genre_scores_gemma":[0.45575956,0.00096970604,0.49819422,0.00027238077,0.00005976488,0.00018410459,0.0033111041,0.0023684776,0.038880736],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9993026,0.00015744715,0.000112927475,0.00020178418,0.0001778916,0.000047389538],"domain_scores_gemma":[0.9989832,0.00031086238,0.00007213777,0.00027183155,0.00030426015,0.00005769178],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006620372,0.0005864425,0.0004078448,0.0011359953,0.0014569806,0.0026570016,0.0005461968,0.0006314116,0.010307644],"category_scores_gemma":[0.0021612244,0.0005865828,0.0005803121,0.001398314,0.001259876,0.0023691312,0.0017756571,0.0013597077,0.002833342],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00041133421,0.00006628027,0.018200716,0.0011477369,0.000091164235,0.0013324167,0.014144974,0.007813125,0.07764168,0.39889586,0.019870926,0.4603838],"study_design_scores_gemma":[0.000055662844,0.00015996133,0.02023804,0.00050628436,0.00015413611,0.0024049985,0.0064447024,0.02908811,0.06855154,0.04536876,0.8268099,0.00021793842],"about_ca_topic_score_codex":0.020906541,"about_ca_topic_score_gemma":0.036520645,"teacher_disagreement_score":0.020906541,"about_ca_system_score_codex":0.0017056047,"about_ca_system_score_gemma":0.0022222237,"threshold_uncertainty_score":0.04156971},"labels":[],"label_agreement":null},{"id":"W2034060496","doi":"10.1007/s10590-008-9048-z","title":"METIS-II: low resource machine translation","year":2008,"lang":"en","type":"article","venue":"Machine Translation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Metis; Scope (computer science); Computer science; German; Machine translation; Computational linguistics; Natural language processing; Resource (disambiguation); Linguistics; Artificial intelligence; Programming language; World Wide Web","score_opus":0.020245452710936356,"score_gpt":0.25645432421122505,"score_spread":0.2362088715002887,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2034060496","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012870758,0.0012228006,0.70135915,0.0008727535,0.001713272,0.0008545605,0.02172454,0.22512543,0.034256805],"genre_scores_gemma":[0.06998738,0.00056885765,0.79801285,0.0006351957,0.00066768593,0.0014842872,0.07932874,0.01899421,0.030320846],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9978363,0.00068518263,0.0001951216,0.00051305845,0.00053617544,0.00023412715],"domain_scores_gemma":[0.99799407,0.000573864,0.000111846995,0.00078761653,0.00041169135,0.000121020006],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015195213,0.0029546292,0.002379322,0.0027692416,0.0016712013,0.004063199,0.0028575624,0.0018138033,0.046278283],"category_scores_gemma":[0.004462415,0.0013807301,0.0015108039,0.0027313824,0.00060514867,0.003093654,0.0036861731,0.0024611754,0.049600482],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019324531,0.00044863852,0.0008207777,0.0014416656,0.0003602485,0.00076990045,0.0002785976,0.0044915583,0.056220464,0.03734394,0.35011995,0.54577184],"study_design_scores_gemma":[0.0015717769,0.0013669342,0.0020337417,0.00034819456,0.00042203147,0.0023728646,0.00039066072,0.2766467,0.17834717,0.1008797,0.43531844,0.00030181083],"about_ca_topic_score_codex":0.001074052,"about_ca_topic_score_gemma":0.0018870619,"teacher_disagreement_score":0.046278283,"about_ca_system_score_codex":0.00068113767,"about_ca_system_score_gemma":0.001985114,"threshold_uncertainty_score":0.15481621},"labels":[],"label_agreement":null},{"id":"W2035282759","doi":"10.1016/j.camwa.2009.01.001","title":"Lexical acquisition and clustering of word senses to conceptual lexicon construction","year":2009,"lang":"en","type":"article","venue":"Computers & Mathematics with Applications","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"Mahidol University","keywords":"Lexicon; Word (group theory); Mathematics; Cluster analysis; Natural language processing; Lexical item; Linguistics; Artificial intelligence; Computer science","score_opus":0.011879764611114518,"score_gpt":0.2627749960082452,"score_spread":0.2508952313971307,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2035282759","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.23338614,0.0007578115,0.73990685,0.000794105,0.00025720993,0.00031601416,0.0007507452,0.0018496664,0.021981487],"genre_scores_gemma":[0.7229014,0.0005951737,0.26508507,0.00019793851,0.00010873477,0.00029941884,0.0026246624,0.000991106,0.0071964255],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9969921,0.0012375722,0.0002050656,0.0009758625,0.000320614,0.0002687974],"domain_scores_gemma":[0.99029934,0.005735918,0.0004069596,0.0015882312,0.0016198382,0.00034977478],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020464463,0.0005990669,0.0008802028,0.0032376416,0.0018868186,0.005397546,0.0020449413,0.0011706605,0.009327925],"category_scores_gemma":[0.02178669,0.0014112674,0.0015062117,0.0036101819,0.0027321216,0.007643664,0.0033894707,0.0026137787,0.0025749907],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00066918257,0.0005101784,0.01295912,0.00084451865,0.0001963955,0.00061201205,0.007215059,0.010920028,0.055839323,0.35293373,0.011018758,0.5462817],"study_design_scores_gemma":[0.000073517695,0.00019721466,0.01712612,0.00023717365,0.00021715144,0.0009419162,0.0044328216,0.18793312,0.046916585,0.7174676,0.024283014,0.00017379454],"about_ca_topic_score_codex":0.0053442544,"about_ca_topic_score_gemma":0.0077200965,"teacher_disagreement_score":0.009327925,"about_ca_system_score_codex":0.0017281191,"about_ca_system_score_gemma":0.0023006892,"threshold_uncertainty_score":0.031204998},"labels":[],"label_agreement":null},{"id":"W2036126214","doi":"10.1016/j.csl.2014.10.007","title":"Hybrid Arabic–French machine translation using syntactic re-ordering and morphological pre-processing","year":2014,"lang":"en","type":"article","venue":"Computer Speech & Language","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Machine translation; Natural language processing; Artificial intelligence; Example-based machine translation; Arabic; BLEU; Translation (biology); Verb; Linguistics","score_opus":0.016871099164201,"score_gpt":0.2750535633622381,"score_spread":0.2581824641980371,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2036126214","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08576951,0.0021109465,0.8244672,0.0011028722,0.0020487143,0.00054500985,0.0050079473,0.045092456,0.03385535],"genre_scores_gemma":[0.25203997,0.0008393508,0.71475995,0.0004747667,0.0004178812,0.00025660484,0.0097660925,0.004262144,0.017183226],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990181,0.00028538163,0.000115922136,0.00027023794,0.00019027328,0.00012015111],"domain_scores_gemma":[0.9982003,0.00043876847,0.00008760811,0.00033465074,0.0008772926,0.00006133596],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007085794,0.00222655,0.0013697473,0.0021915743,0.0013276329,0.0029536013,0.0008846805,0.0011389783,0.016757041],"category_scores_gemma":[0.0020843113,0.00059973187,0.0013625467,0.0017032138,0.00043808296,0.0014157034,0.0014758903,0.0013098177,0.01426248],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010801351,0.00034957653,0.0016187918,0.0013590122,0.00035554508,0.0028748347,0.0009090123,0.0108733,0.18602452,0.012821395,0.042896304,0.73883754],"study_design_scores_gemma":[0.0004673254,0.0010935676,0.006548847,0.00034660118,0.0010104023,0.004979133,0.0019689926,0.3327609,0.38330317,0.025200082,0.24184254,0.0004784725],"about_ca_topic_score_codex":0.004816342,"about_ca_topic_score_gemma":0.0077116815,"teacher_disagreement_score":0.016757041,"about_ca_system_score_codex":0.0005694743,"about_ca_system_score_gemma":0.0016667655,"threshold_uncertainty_score":0.05605787},"labels":[],"label_agreement":null},{"id":"W2036179101","doi":"10.7202/001895ar","title":"ETAP-2: The Linguistics of a Machine Translation System","year":2002,"lang":"fr","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Computer science; Art; Philosophy","score_opus":0.06515560011431351,"score_gpt":0.28132024529049665,"score_spread":0.21616464517618314,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2036179101","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05004238,0.0021692484,0.8662319,0.004531803,0.0009286851,0.00064008666,0.0035229453,0.032389406,0.03954357],"genre_scores_gemma":[0.1881312,0.0013153272,0.75965726,0.0012687774,0.00048482095,0.00060118333,0.006600072,0.0045114644,0.03742998],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99744225,0.0011411849,0.0002557036,0.00051658845,0.0005287686,0.000115524475],"domain_scores_gemma":[0.99747354,0.0010079097,0.00021332788,0.0006276085,0.0005652839,0.00011235513],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002461179,0.000888303,0.0007639669,0.0020458486,0.001884852,0.0044988603,0.0014680428,0.0016201247,0.009750449],"category_scores_gemma":[0.006686514,0.0007628356,0.00091773225,0.001983302,0.0012289897,0.006573052,0.003052211,0.0015664904,0.006472529],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001090897,0.00034634955,0.0068734633,0.0017431604,0.0002528339,0.002007529,0.007947981,0.014247976,0.064119674,0.13738813,0.07023252,0.6937495],"study_design_scores_gemma":[0.00022391835,0.00047937944,0.0054664263,0.00041813593,0.00024175236,0.0039448515,0.0018247602,0.14403977,0.08630665,0.105103604,0.6516633,0.00028756674],"about_ca_topic_score_codex":0.0031287365,"about_ca_topic_score_gemma":0.0043385755,"teacher_disagreement_score":0.009750449,"about_ca_system_score_codex":0.0010630863,"about_ca_system_score_gemma":0.0026270985,"threshold_uncertainty_score":0.032618523},"labels":[],"label_agreement":null},{"id":"W2036207056","doi":"10.1016/s0169-023x(01)00018-0","title":"Using information extraction and natural language generation to answer e-mail","year":2001,"lang":"en","type":"article","venue":"Data & Knowledge Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Natural language generation; Question answering; Domain (mathematical analysis); Natural language; Information extraction; Text messaging; Natural language processing; Frequently asked questions; Information retrieval; World Wide Web; Artificial intelligence","score_opus":0.03129771263752603,"score_gpt":0.3164762397476485,"score_spread":0.2851785271101225,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2036207056","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03219078,0.0005265296,0.94171965,0.0012238275,0.00025296063,0.00069764117,0.0014444953,0.016069131,0.005875005],"genre_scores_gemma":[0.16154426,0.00032265333,0.82849073,0.00043246336,0.00016946235,0.0003953925,0.0045519015,0.0004470424,0.0036460576],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99683005,0.0014739912,0.00029262045,0.0005454548,0.0007138678,0.00014412963],"domain_scores_gemma":[0.9907872,0.0064966897,0.0004194024,0.00066234544,0.0015272783,0.00010701821],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027803844,0.0010549312,0.000976931,0.003698568,0.0010324247,0.0019781196,0.0014356142,0.0016543374,0.0058239703],"category_scores_gemma":[0.012890568,0.0005775355,0.0013109179,0.001964611,0.00068010046,0.0032500576,0.001367671,0.0012594027,0.0032265533],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00054631336,0.00072803,0.0039897375,0.0009534042,0.00016388588,0.0010564728,0.0011323817,0.0122111235,0.036254298,0.028645905,0.030619776,0.8836987],"study_design_scores_gemma":[0.00036747498,0.0004230545,0.0025948763,0.0002225447,0.00048681014,0.0013523008,0.0007958126,0.69349056,0.11925326,0.120469965,0.06039759,0.00014577518],"about_ca_topic_score_codex":0.0018805716,"about_ca_topic_score_gemma":0.0025017834,"teacher_disagreement_score":0.0058239703,"about_ca_system_score_codex":0.00070252817,"about_ca_system_score_gemma":0.0014547699,"threshold_uncertainty_score":0.019483149},"labels":[],"label_agreement":null},{"id":"W2036370008","doi":"10.7202/011612ar","title":"English/Arabic/English Machine Translation: A Historical Perspective","year":2005,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Arabic; Machine translation; Globalization; Context (archaeology); Computer science; Perspective (graphical); Linguistics; Artificial intelligence; Political science; History; Law","score_opus":0.02723138744329888,"score_gpt":0.27070758496541103,"score_spread":0.24347619752211214,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2036370008","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020621344,0.51835907,0.014028518,0.035487365,0.0038369065,0.000039946914,0.00029085277,0.0001365578,0.4071995],"genre_scores_gemma":[0.46014827,0.46552098,0.0145762125,0.0075868275,0.0074844887,0.00009621011,0.00033468058,0.00017287179,0.044079363],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9988954,0.0005004861,0.00012706632,0.00019866823,0.00018311272,0.00009522703],"domain_scores_gemma":[0.9981359,0.0011014297,0.00016960653,0.00007454447,0.00040922177,0.0001093577],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023901842,0.0004534556,0.00024046993,0.0051936386,0.0030924839,0.004746952,0.00041838628,0.0018269323,0.0057654483],"category_scores_gemma":[0.0029623501,0.00026893168,0.0001486918,0.0047710976,0.0054896935,0.006676153,0.0015554522,0.0023940292,0.0015047889],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000077412675,0.000052833668,0.0015001938,0.0011766804,0.000016626245,0.0010211804,0.013328874,0.00048050628,0.00095794175,0.70981807,0.018057961,0.25351176],"study_design_scores_gemma":[0.0000033150739,0.000044718472,0.0020423958,0.0010214795,0.000011380781,0.00094557536,0.004794221,0.00031820277,0.0011084508,0.026571086,0.9631229,0.000016288259],"about_ca_topic_score_codex":0.0024372786,"about_ca_topic_score_gemma":0.0028159758,"teacher_disagreement_score":0.0057654483,"about_ca_system_score_codex":0.003214551,"about_ca_system_score_gemma":0.0012693857,"threshold_uncertainty_score":0.023323298},"labels":[],"label_agreement":null},{"id":"W2036430923","doi":"10.3115/1699648.1699670","title":"Real-word spelling correction using Google Web IT 3-grams","year":2009,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":94,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; University of Toronto","keywords":"Spelling; Computer science; Word (group theory); Fraction (chemistry); Set (abstract data type); String (physics); Task (project management); Natural language processing; Artificial intelligence; Precision and recall; Error detection and correction; Matching (statistics); String searching algorithm; Edit distance; Recall; Speech recognition; Pattern matching; Mathematics; Algorithm; Statistics","score_opus":0.023343311340687505,"score_gpt":0.3014352583549979,"score_spread":0.2780919470143104,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2036430923","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2998894,0.0030604082,0.581268,0.00062884303,0.0011443278,0.00061698427,0.017135961,0.089307785,0.006948303],"genre_scores_gemma":[0.37405467,0.0008066817,0.5991301,0.00022314471,0.00024055623,0.00026839634,0.017453624,0.002201784,0.0056209406],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99710923,0.0004432012,0.00043782202,0.0006121758,0.0012262716,0.00017136309],"domain_scores_gemma":[0.9906749,0.0016488304,0.0022318994,0.0019848065,0.0032619643,0.00019758027],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011014258,0.0021252108,0.0011867038,0.006562154,0.0007551523,0.00108522,0.0012179845,0.0010917567,0.0023883658],"category_scores_gemma":[0.009555987,0.0003257885,0.00084696914,0.004974095,0.00040440608,0.001601334,0.0010017868,0.00086685613,0.004089658],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006879267,0.00015390411,0.020555872,0.0008082276,0.00022038743,0.00075092004,0.00040910728,0.0048164846,0.084864736,0.0010019493,0.017929163,0.86780125],"study_design_scores_gemma":[0.00013645695,0.0006988588,0.061095055,0.0002535057,0.0003951387,0.004519304,0.00082565635,0.2924009,0.56254375,0.0074426164,0.06913318,0.00055559253],"about_ca_topic_score_codex":0.007791705,"about_ca_topic_score_gemma":0.014109467,"teacher_disagreement_score":0.007791705,"about_ca_system_score_codex":0.0004554485,"about_ca_system_score_gemma":0.001509346,"threshold_uncertainty_score":0.015492678},"labels":[],"label_agreement":null},{"id":"W2036476111","doi":"10.1558/aleth.v4i2.55","title":"Conference Report. Debating Realism(s): The Fifth Annual IACR Conference, Roskilde 2001","year":2001,"lang":"en","type":"article","venue":"Alethia","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Realism; Philosophy; Epistemology; Sociology","score_opus":0.025310200025995655,"score_gpt":0.29556820782876214,"score_spread":0.2702580078027665,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2036476111","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008340408,0.023920676,0.026773095,0.09360018,0.055718336,0.0012278443,0.015671331,0.0071263816,0.76762176],"genre_scores_gemma":[0.020590384,0.0053759534,0.008225589,0.0026324985,0.0028385564,0.00031048674,0.0101819,0.0017007276,0.94814396],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99795294,0.00032694472,0.00007844129,0.00030726008,0.00093821745,0.00039622575],"domain_scores_gemma":[0.9932979,0.00047546427,0.00019935044,0.0007546539,0.0037940883,0.0014784877],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005245182,0.0013307761,0.0010115023,0.0021643075,0.0031398227,0.005788487,0.0023031214,0.0032566849,0.26869178],"category_scores_gemma":[0.005301714,0.00057585904,0.0009914794,0.0016385075,0.00079522654,0.003267479,0.0032809365,0.004114126,0.13693736],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007038295,0.000057773155,0.00015630724,0.00007178728,0.000004829193,0.000027118798,0.00005063155,0.00010256297,0.00038447967,0.0017956599,0.97125435,0.026024261],"study_design_scores_gemma":[0.000029804729,0.0000358424,0.000685926,0.000073589355,0.000014695837,0.000059769787,0.00014079436,0.00030275254,0.0009734968,0.0006290432,0.9970401,0.000014104614],"about_ca_topic_score_codex":0.020311056,"about_ca_topic_score_gemma":0.039758015,"teacher_disagreement_score":0.26869178,"about_ca_system_score_codex":0.0024955692,"about_ca_system_score_gemma":0.0049334653,"threshold_uncertainty_score":0.89886355},"labels":[],"label_agreement":null},{"id":"W2036519142","doi":"10.1177/171516350413700804","title":"Deactivate Old Prescription to Prevent Error or Duplication","year":2004,"lang":"en","type":"article","venue":"Canadian Pharmacists Journal / Revue des Pharmaciens du Canada","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Medical prescription; Gene duplication; Medicine; Pharmacology; Biology; Genetics","score_opus":0.031661007041838114,"score_gpt":0.29431661632017975,"score_spread":0.26265560927834164,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2036519142","genre_codex":"methods","genre_gemma":"commentary","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"commentary","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.32318053,0.005266727,0.47294167,0.018520935,0.0042667952,0.0026273518,0.00845341,0.08844996,0.07629259],"genre_scores_gemma":[0.7340133,0.0013275939,0.16789746,0.004648938,0.00082271104,0.00028853718,0.002815286,0.0044524292,0.083733775],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.996332,0.0005288829,0.0004852944,0.0006963209,0.0016642772,0.00029326775],"domain_scores_gemma":[0.977389,0.009263682,0.0029874216,0.0051807296,0.004627585,0.00055159617],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021805575,0.0010092552,0.00094698457,0.0017500286,0.0011634529,0.002332509,0.0015438469,0.0019228589,0.013674466],"category_scores_gemma":[0.022193668,0.0006166271,0.00084048643,0.00086786685,0.00063183246,0.0022755165,0.00096670905,0.00183397,0.005776146],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0023899153,0.0013609198,0.042307623,0.00087207166,0.00018842542,0.0038771702,0.001766212,0.004286903,0.052607153,0.01048313,0.08630658,0.7935539],"study_design_scores_gemma":[0.00035601103,0.00089580286,0.04088295,0.0007003494,0.00087531697,0.008313707,0.0015649921,0.10293977,0.33877167,0.022447657,0.48181716,0.00043468247],"about_ca_topic_score_codex":0.008571677,"about_ca_topic_score_gemma":0.01115336,"teacher_disagreement_score":0.013674466,"about_ca_system_score_codex":0.0012162466,"about_ca_system_score_gemma":0.0043469714,"threshold_uncertainty_score":0.04574567},"labels":[],"label_agreement":null},{"id":"W2036606287","doi":"10.7202/002593ar","title":"Une méthodologie d'identification automatique des syntagmes terminologiques : l'apport de la description du non-terme","year":2002,"lang":"fr","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Philosophy; Art","score_opus":0.09535478051685656,"score_gpt":0.31731118802117303,"score_spread":0.22195640750431647,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2036606287","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.002848877,0.00031563317,0.9937099,0.00013635289,0.00007476027,0.000102697966,0.00016443706,0.00118668,0.0014606921],"genre_scores_gemma":[0.029191285,0.00041422935,0.9634968,0.00008991643,0.00004046903,0.00028392457,0.0006004965,0.00066606223,0.0052168407],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9933194,0.0019748155,0.00067750463,0.0015507867,0.002187688,0.00028983384],"domain_scores_gemma":[0.9904565,0.004179251,0.0005086954,0.002004693,0.0027206321,0.00013031924],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004315985,0.0017061821,0.0015235636,0.0036812134,0.0019234252,0.0056641316,0.002191953,0.002425056,0.0066334233],"category_scores_gemma":[0.014177096,0.0009787336,0.0019927288,0.0031006755,0.0025270346,0.0043700375,0.002920113,0.0033774052,0.005434773],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002938428,0.00013208407,0.0039090887,0.002244427,0.0002445411,0.00089647423,0.0063531757,0.0054095876,0.1212348,0.13990973,0.008104255,0.71126795],"study_design_scores_gemma":[0.00016956872,0.00030776666,0.0071909144,0.0011078633,0.0003792015,0.0075236694,0.005293719,0.16596997,0.2140241,0.21667296,0.38083336,0.0005269293],"about_ca_topic_score_codex":0.0045168027,"about_ca_topic_score_gemma":0.004364582,"teacher_disagreement_score":0.0066334233,"about_ca_system_score_codex":0.0012270737,"about_ca_system_score_gemma":0.0035000194,"threshold_uncertainty_score":0.02282542},"labels":[],"label_agreement":null},{"id":"W2037256905","doi":"10.1007/s10579-014-9271-6","title":"A qualitative comparison method for rhetorical structures: identifying different discourse structures in multilingual corpora","year":2014,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":81,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Rhetorical question; Linguistics; Computer science; Natural language processing; Annotation; Translation (biology); Artificial intelligence; Contrastive linguistics; Applied linguistics; Philosophy","score_opus":0.08234689324121308,"score_gpt":0.47851651750124213,"score_spread":0.39616962426002905,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2037256905","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11223794,0.00078777754,0.81558204,0.0011972412,0.00031829166,0.018403498,0.009043007,0.0012204258,0.04120978],"genre_scores_gemma":[0.2523417,0.0002599571,0.70554495,0.0003151655,0.00004421625,0.033671506,0.0027975196,0.0003893218,0.004635634],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.94543654,0.03518339,0.00460223,0.0045378846,0.009237705,0.0010022392],"domain_scores_gemma":[0.8347893,0.109949395,0.006716553,0.008818467,0.038084365,0.0016419892],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.044161096,0.000992284,0.0011659275,0.013162892,0.0041945227,0.004793654,0.0025422813,0.0012595208,0.011526145],"category_scores_gemma":[0.13460061,0.00070950773,0.0011270446,0.01009224,0.0040522013,0.003713692,0.0048719165,0.0014919321,0.0015162635],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0030170546,0.0012662899,0.028478863,0.010443087,0.00059590366,0.00052169367,0.15613385,0.0024991827,0.051448576,0.14783335,0.018912002,0.57885015],"study_design_scores_gemma":[0.0022694399,0.0024644667,0.07732867,0.0052656196,0.0015530172,0.0016223148,0.23636182,0.052617572,0.13563798,0.2517841,0.23204972,0.0010453789],"about_ca_topic_score_codex":0.0046013403,"about_ca_topic_score_gemma":0.009983829,"teacher_disagreement_score":0.044161096,"about_ca_system_score_codex":0.0056877863,"about_ca_system_score_gemma":0.007736226,"threshold_uncertainty_score":0.233549},"labels":[],"label_agreement":null},{"id":"W2037632172","doi":"10.1023/b:coat.0000010116.83274.c3","title":"Construction of a Chinese–English Verb Lexicon for Machine Translation and Embedded Multilingual Applications","year":2002,"lang":"en","type":"article","venue":"Machine Translation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Directorate for Computer and Information Science and Engineering; National Science Foundation","keywords":"Computer science; Lexicon; Natural language processing; Machine translation; Artificial intelligence; Verb; Lexical database; WordNet","score_opus":0.019152476021540424,"score_gpt":0.2812119978952083,"score_spread":0.2620595218736679,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2037632172","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.030654768,0.00036911038,0.92875326,0.00046484714,0.00036887356,0.00086025184,0.005851702,0.01590016,0.016777067],"genre_scores_gemma":[0.213377,0.00052547007,0.74415946,0.00020821163,0.00012201832,0.00093486154,0.026853325,0.003931476,0.009888169],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9992855,0.00015811133,0.00011547865,0.00020550712,0.00015078933,0.00008466229],"domain_scores_gemma":[0.9988625,0.00026616137,0.000069504764,0.00015010408,0.0005819851,0.00006962332],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00071472995,0.0012146692,0.0012228952,0.0035148624,0.0016010817,0.0024594234,0.0011881228,0.00066953426,0.013331406],"category_scores_gemma":[0.0027893134,0.0011106663,0.0011714163,0.003478239,0.0006847969,0.0025388424,0.0022569129,0.001414814,0.007554062],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006964793,0.00043620073,0.0039899074,0.0021261615,0.00019935425,0.0041114977,0.0026954308,0.01460752,0.1502956,0.14907087,0.08710958,0.58466136],"study_design_scores_gemma":[0.00035847997,0.00044405303,0.004691276,0.0004257676,0.0006945234,0.0026408504,0.0021339462,0.3754455,0.19161046,0.0999506,0.3213098,0.00029473362],"about_ca_topic_score_codex":0.0046816003,"about_ca_topic_score_gemma":0.007886292,"teacher_disagreement_score":0.013331406,"about_ca_system_score_codex":0.0012274226,"about_ca_system_score_gemma":0.0039397595,"threshold_uncertainty_score":0.044598043},"labels":[],"label_agreement":null},{"id":"W2038227658","doi":"10.1007/s10994-005-0913-1","title":"Corpus-based Learning of Analogies and Semantic Relations","year":2005,"lang":"en","type":"article","venue":"Machine Learning","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":178,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Analogy; Noun; Artificial intelligence; Natural language processing; Part of speech; Computer science; Class (philosophy); Meaning (existential); Word (group theory); Set (abstract data type); Semantics (computer science); Mathematics; Linguistics; Psychology","score_opus":0.008477493126515724,"score_gpt":0.25691361684639447,"score_spread":0.24843612371987875,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2038227658","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.18203826,0.003535895,0.7603718,0.0028147066,0.001257574,0.0005038508,0.006369608,0.0056866524,0.03742158],"genre_scores_gemma":[0.68943554,0.001698102,0.28494895,0.0005088111,0.00042921887,0.0005797418,0.014125313,0.0006382998,0.0076360246],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9970065,0.0012794461,0.0002220659,0.000867596,0.00054013496,0.00008430841],"domain_scores_gemma":[0.9761973,0.017672593,0.0005741052,0.0032870094,0.001987597,0.00028143838],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027279905,0.00096027687,0.0010294336,0.004508838,0.0012963895,0.0033809752,0.0030994127,0.0017826528,0.01754993],"category_scores_gemma":[0.035209935,0.0007643875,0.0012021472,0.004113241,0.0018018717,0.010541384,0.0028922944,0.0038698965,0.002767854],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009873494,0.0010220485,0.0059676482,0.0012338302,0.00046616874,0.0006467959,0.0009009264,0.042453013,0.008203799,0.12233772,0.031081298,0.78469944],"study_design_scores_gemma":[0.00030398357,0.00024406695,0.0030929493,0.00022986425,0.000275379,0.0006520984,0.00053707743,0.6097551,0.00791055,0.3503452,0.026554888,0.000098815115],"about_ca_topic_score_codex":0.0034459792,"about_ca_topic_score_gemma":0.0067517017,"teacher_disagreement_score":0.01754993,"about_ca_system_score_codex":0.0011354722,"about_ca_system_score_gemma":0.0018637255,"threshold_uncertainty_score":0.058710396},"labels":[],"label_agreement":null},{"id":"W2038533496","doi":"10.1075/sl.30.2.06mit","title":"Grammars and the community","year":2006,"lang":"en","type":"article","venue":"Studies in Language","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Exemplification; Grammar; Terminology; Linguistics; Presentation (obstetrics); Affect (linguistics); Generative grammar; Situated; Speech community; Rule-based machine translation; Computer science; Style (visual arts); Sociology; Artificial intelligence; History","score_opus":0.018550011044997473,"score_gpt":0.31972688781350617,"score_spread":0.3011768767685087,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2038533496","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13996346,0.0077748643,0.02038197,0.17191194,0.000666685,0.00021558302,0.00026294476,0.0005533233,0.65826917],"genre_scores_gemma":[0.90560204,0.0022148094,0.00508992,0.0032885077,0.00023422953,0.00008436607,0.00012458857,0.0002414869,0.08311999],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"qualitative","domain_scores_codex":[0.9936162,0.003502006,0.00013344247,0.0007756413,0.0010373355,0.00093530805],"domain_scores_gemma":[0.9870745,0.0051678596,0.0005710125,0.0013200272,0.0025091507,0.0033574451],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008013684,0.00036825493,0.00043379996,0.0031832883,0.019371653,0.012599492,0.0019532673,0.0030328785,0.019353908],"category_scores_gemma":[0.013353798,0.00029232303,0.00037268974,0.003174286,0.03466464,0.009358794,0.009499518,0.0023423063,0.0013038359],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00002166536,0.000018941753,0.001951313,0.00004516241,0.000004647985,0.00028669837,0.08721084,0.00028391546,0.00021914131,0.8652231,0.019131754,0.025602764],"study_design_scores_gemma":[0.000028346964,0.000012571417,0.0019195696,0.00014820529,0.000005489527,0.00016843487,0.08072304,0.00078818767,0.00008775834,0.33389348,0.58219814,0.000026715332],"about_ca_topic_score_codex":0.3215188,"about_ca_topic_score_gemma":0.3635548,"teacher_disagreement_score":0.3215188,"about_ca_system_score_codex":0.02472612,"about_ca_system_score_gemma":0.02614245,"threshold_uncertainty_score":0.63929474},"labels":[],"label_agreement":null},{"id":"W2039055524","doi":"10.1145/1384271.1384417","title":"Web-based dynamic learning through lexical chaining","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Chaining; Backward chaining; Natural language processing; Artificial intelligence; Information retrieval; World Wide Web; Expert system; Inference engine","score_opus":0.018460306249049086,"score_gpt":0.28148456381977804,"score_spread":0.26302425757072895,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2039055524","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08107667,0.0002735041,0.90020144,0.0003037342,0.00007790608,0.00022121173,0.00027881417,0.00843199,0.009134775],"genre_scores_gemma":[0.6094778,0.00041296743,0.38154733,0.00019999103,0.000046063124,0.0002735615,0.0012089955,0.0004731583,0.0063600983],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991812,0.00026855292,0.00006700052,0.00021099552,0.00022443998,0.000047798243],"domain_scores_gemma":[0.9948949,0.0034966348,0.00018696861,0.0006803113,0.0005723057,0.00016895137],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012593529,0.00045660548,0.0005445482,0.0012291763,0.0007678439,0.0013377399,0.0012400711,0.0006537361,0.00754181],"category_scores_gemma":[0.0062041576,0.0003106952,0.0003284405,0.0016346221,0.0008180513,0.0060154623,0.0021059157,0.00118493,0.0023169476],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00045590955,0.00077122025,0.0033177142,0.00031029887,0.000056179375,0.0004937968,0.00090652274,0.02586584,0.03223347,0.015827445,0.0046934434,0.91506815],"study_design_scores_gemma":[0.00014097757,0.0003432009,0.0020987745,0.00011295901,0.000097007294,0.0005281204,0.0008911395,0.7714032,0.08504218,0.10320978,0.03603214,0.00010050462],"about_ca_topic_score_codex":0.001781662,"about_ca_topic_score_gemma":0.0034825734,"teacher_disagreement_score":0.00754181,"about_ca_system_score_codex":0.0003225887,"about_ca_system_score_gemma":0.00070669816,"threshold_uncertainty_score":0.025229871},"labels":[],"label_agreement":null},{"id":"W2039302098","doi":"10.3115/1654650.1654657","title":"Phrase-based SMT with shallow Tree-Phrases","year":2006,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Phrase; Natural language processing; Noun phrase; Artificial intelligence; Machine translation; Tree (set theory); Dependency (UML); Translation (biology); Dependency grammar; Specifier; Determiner phrase; Phrase structure rules; Simple (philosophy); Rule-based machine translation; Generative grammar; Mathematics","score_opus":0.006299559038372088,"score_gpt":0.22030345350495817,"score_spread":0.21400389446658608,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2039302098","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0049246685,0.00017270814,0.9857466,0.0001409635,0.000085595115,0.00014113475,0.0007058824,0.0050958553,0.0029864907],"genre_scores_gemma":[0.11903098,0.0003051237,0.87293124,0.0001797881,0.00007267554,0.00030390167,0.0023244596,0.0013241967,0.0035276886],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985771,0.00067733193,0.00014762724,0.00022628698,0.0003101448,0.00006160697],"domain_scores_gemma":[0.9976361,0.0013804213,0.00014053499,0.00044718187,0.0003523572,0.00004338385],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015287298,0.0011955685,0.00089046895,0.00082350895,0.00061484013,0.0013404475,0.001325612,0.0011859563,0.008746323],"category_scores_gemma":[0.0055802744,0.00068179984,0.0009929622,0.0018772752,0.0006580424,0.0030872854,0.0017641818,0.0013986377,0.0069809933],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000797334,0.00018415411,0.0012266308,0.00225583,0.0003601174,0.0014405699,0.0016721082,0.086780325,0.07453142,0.13493185,0.029422186,0.6663975],"study_design_scores_gemma":[0.0001485435,0.00028112845,0.0005140262,0.00017141677,0.00019640851,0.0007213723,0.00030647183,0.68013054,0.06593424,0.18421137,0.06726952,0.00011489681],"about_ca_topic_score_codex":0.0009822578,"about_ca_topic_score_gemma":0.0016236115,"teacher_disagreement_score":0.008746323,"about_ca_system_score_codex":0.00044204757,"about_ca_system_score_gemma":0.00082824996,"threshold_uncertainty_score":0.029259384},"labels":[],"label_agreement":null},{"id":"W2039427737","doi":"10.3115/1220575.1220583","title":"Redundancy-based correction of automatically extracted facts","year":2005,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"York University","keywords":"Redundancy (engineering); Pipeline (software); Computer science; Information extraction; Data mining; Event (particle physics); Artificial intelligence; Machine learning","score_opus":0.009424979568080563,"score_gpt":0.2671534743695565,"score_spread":0.2577284948014759,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2039427737","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1236594,0.0015061474,0.8429934,0.00079787005,0.0003566598,0.0003255794,0.0024499441,0.02440553,0.0035054479],"genre_scores_gemma":[0.37215972,0.0006468151,0.61428136,0.0002727752,0.00024840623,0.00014221208,0.0052274913,0.001609047,0.0054122573],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9964563,0.00072213786,0.0003316402,0.00081251014,0.00149839,0.00017895189],"domain_scores_gemma":[0.9696595,0.012659569,0.003186615,0.008963727,0.005370945,0.00015946916],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036753803,0.0014876117,0.0012775558,0.0051197372,0.0008583353,0.0012844984,0.0017910396,0.00094092544,0.0025266875],"category_scores_gemma":[0.027436858,0.0006132062,0.00091816694,0.0029399444,0.0007476852,0.002751686,0.0014988919,0.001268794,0.001913464],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007120101,0.0001363508,0.012128148,0.00096181245,0.00027096973,0.0015297335,0.00113427,0.023337537,0.07950704,0.005404036,0.015734944,0.8591432],"study_design_scores_gemma":[0.000111557,0.0005025513,0.022913694,0.00030663738,0.0008458979,0.003894219,0.0004968329,0.45480412,0.42754197,0.02355841,0.06478304,0.0002410819],"about_ca_topic_score_codex":0.002695794,"about_ca_topic_score_gemma":0.004204972,"teacher_disagreement_score":0.0051197372,"about_ca_system_score_codex":0.00057360815,"about_ca_system_score_gemma":0.0014745248,"threshold_uncertainty_score":0.019437551},"labels":[],"label_agreement":null},{"id":"W2039590542","doi":"10.5539/ells.v3n3p48","title":"Corpus Linguistics and Corpus-Based Research in Hong Kong: A State-of-Art Review","year":2013,"lang":"en","type":"review","venue":"English Language and Literature Studies","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"Chinese University of Hong Kong; Hong Kong Institute of Education","keywords":"Corpus linguistics; Linguistics; Applied linguistics; Computer science; Computational linguistics; Natural language processing","score_opus":0.0713685790660568,"score_gpt":0.4029719709363367,"score_spread":0.3316033918702799,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2039590542","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00026448263,0.99848276,0.0001405277,0.00033142784,0.000111738605,0.000017809172,0.00004253108,0.000007528662,0.00060110947],"genre_scores_gemma":[0.0017057658,0.9975394,0.00038118774,0.000121381694,0.000056932247,0.000024163874,0.000050341816,0.0000035620953,0.00011721008],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.997698,0.0008013085,0.0005809325,0.00030393424,0.0005212145,0.000094644434],"domain_scores_gemma":[0.9804115,0.014754608,0.0013150676,0.000414329,0.0027308175,0.00037372633],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007579612,0.00092848344,0.0024167325,0.012982633,0.0010565448,0.0032924504,0.0015896328,0.0011811845,0.0035995184],"category_scores_gemma":[0.013141088,0.00079929864,0.0008265104,0.018390741,0.002349925,0.0035629086,0.0018665227,0.0010898318,0.0007111063],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007003136,0.000058678856,0.0019933113,0.13399208,0.00036970386,0.00040172486,0.0020481255,0.00031206902,0.00059941853,0.0052375314,0.024820006,0.8300972],"study_design_scores_gemma":[0.000026319318,0.000097833465,0.011875408,0.1045177,0.0012882976,0.0011558946,0.0026579124,0.00023154696,0.0007098332,0.0023097103,0.87505054,0.00007901894],"about_ca_topic_score_codex":0.024237989,"about_ca_topic_score_gemma":0.03574695,"teacher_disagreement_score":0.024237989,"about_ca_system_score_codex":0.0032527666,"about_ca_system_score_gemma":0.015391797,"threshold_uncertainty_score":0.048193812},"labels":[],"label_agreement":null},{"id":"W2041917831","doi":"10.5087/dad.2013.211","title":"Annotation upon Annotation: Adding Signalling Information to a Corpus of Discourse Relations","year":2013,"lang":"en","type":"article","venue":"Dialogue & Discourse","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":35,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Annotation; Rhetorical question; Parsing; Computer science; Natural language processing; Artificial intelligence; Corpus linguistics; Linguistics; Information retrieval","score_opus":0.010666998607709078,"score_gpt":0.27637598440670746,"score_spread":0.2657089857989984,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2041917831","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.046069972,0.00062900287,0.8903099,0.0032771477,0.001403603,0.0014124365,0.011376858,0.009970801,0.03555042],"genre_scores_gemma":[0.11917479,0.0005032485,0.83030444,0.0010474205,0.0005962828,0.0024810361,0.026294572,0.003922916,0.015675314],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9871459,0.005729913,0.0012397064,0.0023712667,0.003095447,0.00041766142],"domain_scores_gemma":[0.9151309,0.05338061,0.0021648733,0.01669,0.011570665,0.001063023],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01289671,0.0013055276,0.0013985052,0.0073261512,0.004346363,0.005791855,0.0031149061,0.0028599603,0.014780366],"category_scores_gemma":[0.06210398,0.001689528,0.0010104188,0.0076138675,0.0032349941,0.010357164,0.007970722,0.006314311,0.005955084],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001222176,0.0006642023,0.005871384,0.0028230578,0.00014211745,0.0022722178,0.032865267,0.0063731875,0.1838109,0.12378882,0.089895405,0.5502713],"study_design_scores_gemma":[0.00018675085,0.00027706134,0.011560656,0.00086252345,0.00029027017,0.0010770028,0.006630833,0.08906416,0.09832444,0.112579204,0.67859983,0.00054723443],"about_ca_topic_score_codex":0.0049963733,"about_ca_topic_score_gemma":0.008377949,"teacher_disagreement_score":0.014780366,"about_ca_system_score_codex":0.0018215108,"about_ca_system_score_gemma":0.004069498,"threshold_uncertainty_score":0.06820518},"labels":[],"label_agreement":null},{"id":"W2043104383","doi":"10.1007/978-1-4020-4746-6_8","title":"Question Answering By Passage Selection","year":2006,"lang":"en","type":"book-chapter","venue":"Text, speech and language technology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":31,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Question answering; Computer science; Selection (genetic algorithm); Redundancy (engineering); Information retrieval; Data mining; Artificial intelligence","score_opus":0.004151132688285129,"score_gpt":0.23233863796639057,"score_spread":0.22818750527810544,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2043104383","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011855671,0.0016694298,0.9396197,0.0009656746,0.0003375037,0.00047204678,0.0019987368,0.015229991,0.027851216],"genre_scores_gemma":[0.1484074,0.0010015707,0.7930056,0.00057886454,0.00047485315,0.0005549208,0.010509513,0.0016197889,0.043847572],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99711263,0.0013473606,0.00015885502,0.000693011,0.00052147143,0.00016670582],"domain_scores_gemma":[0.99662626,0.0022503128,0.00006935987,0.0005036862,0.00045148074,0.000098931734],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023146449,0.0013502571,0.0018588956,0.0031266077,0.0013605652,0.0026379053,0.0030281977,0.0013010541,0.030966613],"category_scores_gemma":[0.0074417237,0.00076655316,0.0020950243,0.002528314,0.0012822566,0.005296805,0.0028493595,0.0016947953,0.015022246],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00041853517,0.00026258425,0.00089455023,0.0008360169,0.0001783574,0.0004172873,0.0011143119,0.0049226317,0.028020704,0.083204016,0.106045395,0.77368575],"study_design_scores_gemma":[0.00017256431,0.00033700114,0.0024045783,0.00018392302,0.00049068435,0.0015463962,0.0011674543,0.30503777,0.06996091,0.3588435,0.25969145,0.00016375374],"about_ca_topic_score_codex":0.002605276,"about_ca_topic_score_gemma":0.003027557,"teacher_disagreement_score":0.030966613,"about_ca_system_score_codex":0.0007086357,"about_ca_system_score_gemma":0.00070636487,"threshold_uncertainty_score":0.10359365},"labels":[],"label_agreement":null},{"id":"W2044746424","doi":"10.1075/lia.1.2.05car","title":"Explaining how learners extract ‘formulae’ from L2 input","year":2010,"lang":"en","type":"article","venue":"Language Interaction and Acquisition","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"Canada Research Chairs","keywords":"Segmentation; Syllable; Computer science; Linguistics; Speech segmentation; Natural language processing; Tracking (education); Speech recognition; Artificial intelligence; Mathematics; Psychology","score_opus":0.00865952947203548,"score_gpt":0.2754773890086907,"score_spread":0.2668178595366552,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2044746424","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19700953,0.001077405,0.6751395,0.0071983384,0.0001890563,0.00020703739,0.00062165916,0.003238812,0.1153187],"genre_scores_gemma":[0.838681,0.000922647,0.14181413,0.00068569294,0.000074550786,0.00019084636,0.0006362523,0.000685171,0.016309695],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.9994036,0.00019744548,0.00003624912,0.00015036856,0.00014848469,0.000063949],"domain_scores_gemma":[0.99847513,0.0009089427,0.00012423615,0.00020314085,0.00022743082,0.000060994582],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011725092,0.0006218872,0.00034673934,0.0008177582,0.00063865766,0.003835244,0.0015616474,0.0016883711,0.008551649],"category_scores_gemma":[0.0070488174,0.00066837075,0.00068248896,0.0005698596,0.0026778996,0.010909019,0.0020794056,0.0019592629,0.0029082252],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020617206,0.00009383731,0.012104799,0.0006150757,0.000055559984,0.0018339903,0.033444498,0.0100386385,0.025552824,0.69936985,0.010439597,0.20624512],"study_design_scores_gemma":[0.0000864926,0.00012121457,0.008222449,0.0002604294,0.000056291527,0.0013214891,0.0038669494,0.11952,0.021102445,0.7810463,0.06426606,0.0001298466],"about_ca_topic_score_codex":0.0027011647,"about_ca_topic_score_gemma":0.0021366146,"teacher_disagreement_score":0.008551649,"about_ca_system_score_codex":0.0010843426,"about_ca_system_score_gemma":0.00062377343,"threshold_uncertainty_score":0.028608143},"labels":[],"label_agreement":null},{"id":"W2044810048","doi":"10.1002/asi.22696","title":"Mining a multilingual association dictionary from<scp>W</scp>ikipedia for cross‐language information retrieval","year":2012,"lang":"en","type":"article","venue":"Journal of the American Society for Information Science and Technology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Computer science; Cross-language information retrieval; Information retrieval; Variety (cybernetics); Association (psychology); Natural language processing; Graph; Artificial intelligence; Machine translation; Quality (philosophy); Query expansion; Meaning (existential); Filter (signal processing)","score_opus":0.008604640807155386,"score_gpt":0.29438623223224775,"score_spread":0.28578159142509235,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2044810048","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14067271,0.0024138866,0.80266774,0.0014923309,0.0005314418,0.00086394284,0.024512943,0.006198061,0.020646928],"genre_scores_gemma":[0.28511786,0.0016552809,0.6440409,0.00052781426,0.00020801234,0.0007919393,0.058734585,0.0008033426,0.008120308],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99866486,0.00024203632,0.00026694374,0.0003882899,0.00033793316,0.000099973],"domain_scores_gemma":[0.9964154,0.0011860431,0.0004624274,0.0007630453,0.0009867747,0.00018639753],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00093900267,0.0006499853,0.00074878044,0.009555426,0.0013935189,0.0017800541,0.001046986,0.00076826185,0.004865533],"category_scores_gemma":[0.0062396657,0.00042591564,0.0009102265,0.009223473,0.00071070355,0.003859299,0.0027040476,0.0011443981,0.003570253],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00033185823,0.00045701084,0.014063964,0.0018961884,0.00027533615,0.0020833912,0.0020133364,0.0067130355,0.057764098,0.029769462,0.06539703,0.8192352],"study_design_scores_gemma":[0.00017161116,0.0005528037,0.030444484,0.0006165406,0.00045491604,0.0053127767,0.006260121,0.21241097,0.06054725,0.08914154,0.59377193,0.00031516704],"about_ca_topic_score_codex":0.003291576,"about_ca_topic_score_gemma":0.009737184,"teacher_disagreement_score":0.009555426,"about_ca_system_score_codex":0.00063749065,"about_ca_system_score_gemma":0.0021992915,"threshold_uncertainty_score":0.016276836},"labels":[{"model":"gemma","categories":[],"domain":null,"study_design":"bench_or_experimental","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"high"},{"model":"gpt","categories":[],"domain":null,"study_design":"design_other","genre":"methods","about_ca_system":false,"about_ca_topic":false,"confidence":"high"}],"label_agreement":"split"},{"id":"W2046145328","doi":"10.1075/term.8.1.05bar","title":"Hierarchical refinement and representation of the causal relation","year":2002,"lang":"en","type":"article","venue":"Terminology International Journal of Theoretical and Applied Issues in Specialized Communication","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Categorization; Relation (database); Computer science; Formalism (music); Certainty; Natural language processing; Representation (politics); Artificial intelligence; Semantic relation; Theoretical computer science; Data mining; Epistemology; Psychology; Philosophy; Cognition","score_opus":0.018109761798460123,"score_gpt":0.3143732752555234,"score_spread":0.29626351345706325,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2046145328","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0068897572,0.00028733263,0.98718065,0.0003544807,0.000030399628,0.00012842321,0.00039892067,0.00048163254,0.0042483886],"genre_scores_gemma":[0.16231735,0.00054550957,0.832127,0.000125015,0.000056984518,0.0002495446,0.0014027337,0.00014132104,0.0030345505],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9968442,0.0011091371,0.00032588022,0.00079760526,0.00075959915,0.00016370417],"domain_scores_gemma":[0.9918566,0.004167884,0.00082112395,0.0019241488,0.001108787,0.000121352685],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037810276,0.00055286527,0.00056465773,0.0037761775,0.0012890766,0.0026165196,0.0014768906,0.0010027359,0.0056465277],"category_scores_gemma":[0.015061121,0.0007001948,0.0016816187,0.003333919,0.0026966147,0.006379096,0.0019780216,0.0019735696,0.0011274957],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000067342546,0.0000305275,0.0007775709,0.00025502048,0.000036724367,0.0002367349,0.0020987086,0.029195152,0.0031383839,0.8775139,0.0018872031,0.084762774],"study_design_scores_gemma":[0.000025324232,0.00003466903,0.00049904815,0.00014169408,0.00008006273,0.00019250059,0.00032720919,0.15274394,0.0034365035,0.8062453,0.03622475,0.00004889928],"about_ca_topic_score_codex":0.011653747,"about_ca_topic_score_gemma":0.011661692,"teacher_disagreement_score":0.011653747,"about_ca_system_score_codex":0.0019304666,"about_ca_system_score_gemma":0.002270745,"threshold_uncertainty_score":0.023171842},"labels":[],"label_agreement":null},{"id":"W2046384065","doi":"10.3115/1117586.1117593","title":"TransType","year":2000,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":79,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Translation (biology); Machine translation; Embedding; Translation system; Natural language processing; Artificial intelligence","score_opus":0.007148780858881537,"score_gpt":0.24483461589041255,"score_spread":0.237685835031531,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2046384065","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011979947,0.0010159964,0.6094858,0.0027299302,0.0033743654,0.00056501955,0.020134293,0.13333884,0.21737579],"genre_scores_gemma":[0.14079383,0.0023159755,0.3432589,0.006788435,0.001518341,0.0009855984,0.05427809,0.12690794,0.323153],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99643886,0.0008136194,0.00036055953,0.00088134175,0.001210527,0.00029517617],"domain_scores_gemma":[0.9941636,0.0015351798,0.0002451434,0.0022234656,0.0016175561,0.00021513017],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027250242,0.0011845846,0.0009638384,0.0015721521,0.0014364135,0.003834177,0.002273067,0.0013804744,0.08748687],"category_scores_gemma":[0.006251557,0.0010074808,0.0013413336,0.0013616885,0.0010857034,0.007119595,0.0053799674,0.0025531093,0.06316903],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010271227,0.00020930044,0.004032739,0.0013744906,0.000126128,0.0015279219,0.002801042,0.0014415793,0.019627806,0.25665694,0.39797175,0.3132032],"study_design_scores_gemma":[0.000027596574,0.00005038396,0.00035748247,0.00007637119,0.000024104571,0.0007134341,0.00020966685,0.0018049548,0.011373389,0.01952223,0.96578896,0.000051384603],"about_ca_topic_score_codex":0.0017821963,"about_ca_topic_score_gemma":0.0031830342,"teacher_disagreement_score":0.08748687,"about_ca_system_score_codex":0.0011250885,"about_ca_system_score_gemma":0.0014308986,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W2046436395","doi":"10.1162/089120100561755","title":"The Rhetorical Parsing of Unrestricted Texts: A Surface-based Approach","year":2000,"lang":"en","type":"article","venue":"Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":221,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; University of Toronto","keywords":"Rhetorical question; Computer science; Automatic summarization; Parsing; Natural language processing; Artificial intelligence; Linguistics; Context (archaeology); Natural language understanding; Natural language; Philosophy; History","score_opus":0.020461250025670744,"score_gpt":0.28202723553398357,"score_spread":0.26156598550831284,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2046436395","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013765123,0.00039678995,0.98039573,0.00040418204,0.000034177407,0.0001192183,0.00020515008,0.0017450151,0.0029347439],"genre_scores_gemma":[0.16334963,0.0005380128,0.8322905,0.000107219574,0.000097889475,0.00015960836,0.0011505153,0.00082482444,0.0014818683],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9977932,0.0011568957,0.00012182246,0.0003718965,0.00048252993,0.00007371001],"domain_scores_gemma":[0.98816663,0.008296403,0.00068294245,0.0015015631,0.0012270652,0.00012527993],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00255761,0.00088418415,0.0010734204,0.004258771,0.0014150916,0.0050429683,0.0016372411,0.0013281349,0.003956576],"category_scores_gemma":[0.01841367,0.0008874147,0.0011244622,0.0025472958,0.003182783,0.0067814128,0.0017282186,0.0022745458,0.0019295325],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001796173,0.00017936296,0.0027020914,0.0012935248,0.000116805226,0.00040812246,0.00441657,0.045707464,0.037285108,0.29270095,0.005571573,0.6094388],"study_design_scores_gemma":[0.000049783514,0.00019009541,0.0021428436,0.0002429165,0.0001090402,0.00046397164,0.0013345274,0.5034836,0.03129079,0.42767268,0.032928783,0.00009093446],"about_ca_topic_score_codex":0.0006661028,"about_ca_topic_score_gemma":0.00093304086,"teacher_disagreement_score":0.0050429683,"about_ca_system_score_codex":0.00086548354,"about_ca_system_score_gemma":0.0013694259,"threshold_uncertainty_score":0.013526082},"labels":[],"label_agreement":null},{"id":"W2046670498","doi":"10.3115/1118771.1118773","title":"An intelligent terminology database as a pre-processor for statistical machine translation","year":2002,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Terminology; Machine translation; Natural language processing; Artificial intelligence; Architecture; Translation (biology); Database; Linguistics","score_opus":0.041965576070698736,"score_gpt":0.3429050682630908,"score_spread":0.3009394921923921,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2046670498","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01593986,0.0013441268,0.9652542,0.0005708002,0.00022138629,0.0002495628,0.0013219282,0.0076722996,0.007425887],"genre_scores_gemma":[0.13530448,0.001045831,0.8549255,0.0003715796,0.00020635998,0.0002526115,0.003953214,0.00056544394,0.0033748925],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9984616,0.0004093059,0.00023796837,0.00026255575,0.00055629143,0.00007223314],"domain_scores_gemma":[0.99656266,0.0011839619,0.0001851358,0.0009387994,0.0010239319,0.00010557988],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003136948,0.0004887289,0.00096864667,0.0037240859,0.00083251286,0.003656765,0.0020520927,0.00079281145,0.006229915],"category_scores_gemma":[0.0054054493,0.00048745755,0.0007592027,0.004123023,0.00075120863,0.004842665,0.0017257524,0.0012969798,0.0046394165],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008081077,0.00018471407,0.0036469563,0.00066067104,0.00016250715,0.000572855,0.0010782443,0.008673292,0.07059596,0.1026514,0.017678136,0.79328716],"study_design_scores_gemma":[0.00024625406,0.00075752,0.004484369,0.00026542338,0.000685094,0.0029358196,0.0013222999,0.30324283,0.18807642,0.1639126,0.33383393,0.00023742285],"about_ca_topic_score_codex":0.0011634512,"about_ca_topic_score_gemma":0.0014543667,"teacher_disagreement_score":0.006229915,"about_ca_system_score_codex":0.00066942925,"about_ca_system_score_gemma":0.0011969678,"threshold_uncertainty_score":0.020841122},"labels":[],"label_agreement":null},{"id":"W2047032662","doi":"10.3115/1072228.1072372","title":"Concept discovery from text","year":2002,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":185,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"WordNet; Cluster analysis; Computer science; Set (abstract data type); Centroid; Cluster (spacecraft); Task (project management); Similarity (geometry); Artificial intelligence; Data mining; Feature vector; Information retrieval; Quality (philosophy); Natural language processing; Image (mathematics)","score_opus":0.01340747817539674,"score_gpt":0.23816629815319207,"score_spread":0.22475881997779534,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2047032662","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.073447265,0.0041699237,0.8865746,0.002543055,0.00043787548,0.0015869435,0.014296713,0.005854557,0.011089047],"genre_scores_gemma":[0.11556147,0.001600348,0.8588792,0.0003973458,0.00030335764,0.0010392917,0.018392669,0.0002734238,0.0035527726],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9941982,0.0013270051,0.0004219269,0.0015705824,0.0022478134,0.00023448131],"domain_scores_gemma":[0.9889746,0.0073676268,0.0005880472,0.0010114608,0.0017291376,0.0003290827],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032874658,0.00165189,0.0017212082,0.013007714,0.0018003208,0.0035423774,0.003028785,0.0017266851,0.005754899],"category_scores_gemma":[0.021284387,0.0006149966,0.001793809,0.009066447,0.0011206586,0.006504131,0.0035054868,0.0017732836,0.0033923844],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00054016063,0.0004128398,0.006377529,0.0021684784,0.00025081358,0.001558125,0.0019397265,0.0104662,0.012992998,0.033615936,0.031566918,0.89811033],"study_design_scores_gemma":[0.00025494993,0.0004071789,0.007760765,0.0008696645,0.00046022952,0.00364187,0.0040781023,0.4170431,0.05412566,0.30120543,0.20990895,0.00024406059],"about_ca_topic_score_codex":0.002191297,"about_ca_topic_score_gemma":0.0030676343,"teacher_disagreement_score":0.013007714,"about_ca_system_score_codex":0.0013399043,"about_ca_system_score_gemma":0.0023186211,"threshold_uncertainty_score":0.019252002},"labels":[],"label_agreement":null},{"id":"W2047497400","doi":"10.1109/icassp.2010.5495037","title":"Approaches to automatic lexicon learning with limited training examples","year":2010,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":31,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Lexicon; Computer science; Natural language processing; Artificial intelligence; Linguistics; Speech recognition","score_opus":0.09435167382084958,"score_gpt":0.25840816905193337,"score_spread":0.1640564952310838,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2047497400","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005896707,0.00051457284,0.9876894,0.00026538776,0.000030851454,0.00014512097,0.00025671048,0.003242108,0.0019590883],"genre_scores_gemma":[0.106064916,0.00091877836,0.88384527,0.00034129055,0.00012949352,0.0007896124,0.0027396348,0.0006452627,0.0045256987],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99719197,0.0010923434,0.0002689772,0.000678984,0.00059879915,0.00016897605],"domain_scores_gemma":[0.98549205,0.009368178,0.00046338755,0.0032531067,0.001254389,0.00016897806],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034174286,0.0015903299,0.0018860195,0.0026868212,0.0012101964,0.0029459835,0.006080011,0.0021958975,0.007511168],"category_scores_gemma":[0.017745532,0.0019490673,0.0016973588,0.002691142,0.0018755927,0.008219973,0.0045961426,0.0032571866,0.0043781423],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021785777,0.00027001367,0.0012720512,0.00074687455,0.0001830492,0.00025924653,0.00043590015,0.07711841,0.0075831753,0.03738315,0.011112793,0.8634175],"study_design_scores_gemma":[0.000104625564,0.00008310732,0.00036317794,0.00009933032,0.00006834788,0.00024124277,0.0001837916,0.84596777,0.008425374,0.13333185,0.011073308,0.000058016823],"about_ca_topic_score_codex":0.0031499714,"about_ca_topic_score_gemma":0.008852253,"teacher_disagreement_score":0.007511168,"about_ca_system_score_codex":0.0012250318,"about_ca_system_score_gemma":0.0020165069,"threshold_uncertainty_score":0.025127351},"labels":[],"label_agreement":null},{"id":"W2047851837","doi":"10.1017/s1351324901002716","title":"Scalable generation of texts using causal and temporal expansions of sentences","year":2001,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Lethbridge","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Sentence; Artificial intelligence; Natural language processing; Scalability; Kernel (algebra); Process (computing); Theoretical computer science; Information retrieval; Programming language","score_opus":0.015037052180989056,"score_gpt":0.2628490486382299,"score_spread":0.24781199645724084,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2047851837","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.028933601,0.00013701833,0.9600132,0.00018481242,0.000041924028,0.00023923807,0.00026349412,0.007584094,0.0026025737],"genre_scores_gemma":[0.19716692,0.00014056252,0.7973164,0.00008856987,0.00007117372,0.00024081435,0.0011795848,0.0008036014,0.002992435],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986628,0.00048575984,0.000084530504,0.0003033127,0.0004088757,0.000054754197],"domain_scores_gemma":[0.9941606,0.0041515124,0.0002458232,0.0008522181,0.00050336425,0.00008648098],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017411901,0.0006828542,0.00052166014,0.00089965854,0.0005392498,0.0009581512,0.0012793147,0.00053715194,0.00507691],"category_scores_gemma":[0.009235212,0.00045821044,0.0008970586,0.0005578607,0.0007526568,0.0024847954,0.0016697994,0.00087999477,0.0012886979],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007519955,0.00030475823,0.0017102312,0.0006967605,0.00011228829,0.0009818812,0.0019867902,0.06566448,0.10696226,0.052806538,0.009141457,0.75888056],"study_design_scores_gemma":[0.00017894166,0.00026997412,0.0013482964,0.000073764386,0.00017507224,0.0005812069,0.00039662662,0.8070766,0.09435052,0.06295892,0.03250411,0.00008593037],"about_ca_topic_score_codex":0.00076267467,"about_ca_topic_score_gemma":0.0012331394,"teacher_disagreement_score":0.00507691,"about_ca_system_score_codex":0.0004043153,"about_ca_system_score_gemma":0.00047381842,"threshold_uncertainty_score":0.016983986},"labels":[],"label_agreement":null},{"id":"W2048359833","doi":"10.1007/s10579-007-9017-9","title":"Automatically learning semantic knowledge about multiword predicates","year":2007,"lang":"en","type":"article","venue":"Computers and the Humanities","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canada Research Chairs; University of Toronto","funders":"","keywords":"Computer science; Noun; Natural language processing; Linguistics; Artificial intelligence; Focus (optics); Verb","score_opus":0.012579192003722123,"score_gpt":0.2518575762560514,"score_spread":0.23927838425232925,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2048359833","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.31872794,0.0023401713,0.64254147,0.0032104095,0.00057071284,0.00026909966,0.0070211426,0.010201391,0.0151175475],"genre_scores_gemma":[0.7577632,0.0014023358,0.22144356,0.0005328339,0.00031498936,0.00013338239,0.014966596,0.00035392516,0.0030890708],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991027,0.00016406598,0.00009082161,0.00038163934,0.0001794795,0.00008122855],"domain_scores_gemma":[0.99607986,0.002882479,0.00024921316,0.0003604561,0.00032549186,0.0001025091],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00070422306,0.0009707012,0.00091447076,0.002753978,0.00096407084,0.0020783343,0.001492165,0.0015401754,0.0053423536],"category_scores_gemma":[0.0059923492,0.00069759384,0.0013693066,0.002121081,0.00090883294,0.011037956,0.0019193197,0.0026199927,0.001389223],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010024586,0.0011194048,0.011437666,0.0012505769,0.00031660846,0.0010939437,0.00086928066,0.017995123,0.030688617,0.07024643,0.03196274,0.8320171],"study_design_scores_gemma":[0.00018680787,0.00033550718,0.0073740873,0.00024714458,0.0003966995,0.00089342555,0.0011543718,0.5751946,0.029274127,0.3618402,0.023015358,0.00008772567],"about_ca_topic_score_codex":0.0028854485,"about_ca_topic_score_gemma":0.00499276,"teacher_disagreement_score":0.0053423536,"about_ca_system_score_codex":0.0009572328,"about_ca_system_score_gemma":0.0015155778,"threshold_uncertainty_score":0.017871976},"labels":[],"label_agreement":null},{"id":"W2048474089","doi":"10.3115/1564535.1564536","title":"Prosodic correlates of rhetorical relations","year":2006,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"European Commission","keywords":"Rhetorical question; Pairwise comparison; Computer science; Artificial intelligence; Relation (database); Natural language processing; Contrast (vision); Class (philosophy); Support vector machine; Elaboration; Speech recognition; Linguistics; Data mining; Humanities","score_opus":0.00934440435470923,"score_gpt":0.2514985633700256,"score_spread":0.24215415901531634,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2048474089","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.95988214,0.0017917468,0.0128313815,0.0003483154,0.00010865628,0.00005664638,0.0005691128,0.00020332565,0.024208656],"genre_scores_gemma":[0.99659145,0.00022387235,0.0021519104,0.000021733213,0.000073306524,0.000021282085,0.00030124484,0.000033644243,0.0005814981],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9991375,0.00028408263,0.0000741491,0.00017836978,0.00024918275,0.00007658905],"domain_scores_gemma":[0.9881627,0.007293689,0.002614126,0.0003971842,0.0011637144,0.00036849498],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011914144,0.0002795047,0.00023747899,0.0018265245,0.0005803644,0.0018110712,0.000308014,0.0004614642,0.0029807861],"category_scores_gemma":[0.019779038,0.0002667748,0.00014937363,0.0008731584,0.0006698552,0.00123911,0.0008881282,0.0007241423,0.0008621045],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002286928,0.00020405767,0.23822264,0.0016953935,0.00020247661,0.0017084291,0.038308352,0.0019944666,0.29715517,0.012428853,0.0023613845,0.40343186],"study_design_scores_gemma":[0.000049641283,0.0005175482,0.9553173,0.00019335811,0.00012833442,0.0013902904,0.007244982,0.004844905,0.015209042,0.0060910936,0.00890454,0.0001089239],"about_ca_topic_score_codex":0.0003067451,"about_ca_topic_score_gemma":0.0005347097,"teacher_disagreement_score":0.0029807861,"about_ca_system_score_codex":0.00019330936,"about_ca_system_score_gemma":0.0002078066,"threshold_uncertainty_score":0.009971738},"labels":[],"label_agreement":null},{"id":"W2048542337","doi":"10.7202/037683ar","title":"When Idioti (Idiotic) Becomes “Fluffy”: Translation Students and the Avoidance of Target-language Cognates","year":2009,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Cognate; Translation (biology); Linguistics; Psychology; Word (group theory); Computer science; Philosophy","score_opus":0.02280330181559537,"score_gpt":0.2900488272617159,"score_spread":0.2672455254461205,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2048542337","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9965778,0.00003768205,0.00026718856,0.00034273908,0.00000953174,0.0000052497994,0.0000050269377,0.0000097281945,0.0027451608],"genre_scores_gemma":[0.9979164,0.00005420932,0.0002679744,0.00031827058,0.000004659622,0.0000073743827,0.000015134541,0.000009304454,0.0014066936],"study_design_codex":"observational","study_design_gemma":"qualitative","domain_scores_codex":[0.9974611,0.00096944463,0.00018565416,0.00031625922,0.0006483793,0.0004191229],"domain_scores_gemma":[0.9865254,0.004611383,0.0049178167,0.001225744,0.0014463492,0.0012733238],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035104225,0.0004206482,0.0005012359,0.00102037,0.0016370136,0.00483761,0.000620989,0.001853159,0.0041936487],"category_scores_gemma":[0.02048775,0.00038212354,0.0002900425,0.0005704704,0.0043367925,0.0023689293,0.0021938828,0.002898853,0.0012105071],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000896494,0.0022835461,0.69935125,0.00018608784,0.000092905204,0.003917869,0.20493609,0.00029598188,0.020306572,0.0057802075,0.0024677888,0.059485182],"study_design_scores_gemma":[0.00011564542,0.0017141062,0.66445506,0.00029131037,0.00023849502,0.010480355,0.26211005,0.0034941393,0.02211816,0.019570602,0.015037835,0.00037426542],"about_ca_topic_score_codex":0.0027712674,"about_ca_topic_score_gemma":0.0036071865,"teacher_disagreement_score":0.00483761,"about_ca_system_score_codex":0.00070345134,"about_ca_system_score_gemma":0.0010146106,"threshold_uncertainty_score":0.018565059},"labels":[],"label_agreement":null},{"id":"W2048760284","doi":"10.1162/089120104323093267","title":"Inferable Centers, Centering Transitions, and the Notion of Coherence","year":2004,"lang":"en","type":"article","venue":"Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Cohesion (chemistry); Coherence (philosophical gambling strategy); Computer science; Indeterminacy (philosophy); Transition (genetics); Linguistics; Identity (music); Natural language processing; Epistemology; Mathematics; Philosophy","score_opus":0.010828978407771098,"score_gpt":0.2574355551626614,"score_spread":0.24660657675489034,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2048760284","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.29876566,0.0013270038,0.68793035,0.00093887834,0.000047340032,0.00013398506,0.0008495854,0.0006238596,0.009383372],"genre_scores_gemma":[0.8705785,0.00029496822,0.12709217,0.00008487521,0.000056939767,0.00014810053,0.00076863996,0.00013825725,0.0008375297],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99642074,0.0015021523,0.0003399249,0.00100609,0.00047025646,0.00026075973],"domain_scores_gemma":[0.9881436,0.0073672133,0.0014393365,0.0017110524,0.0010162203,0.0003224969],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032854413,0.00042450163,0.00054367224,0.0045529054,0.0029158858,0.0031759797,0.0011655859,0.0010069606,0.0021527277],"category_scores_gemma":[0.01347225,0.0005575725,0.0006674048,0.004312677,0.006104614,0.010486696,0.0032482196,0.001722391,0.00021345177],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023307354,0.000047587033,0.005720591,0.00021523016,0.000027056672,0.0003328026,0.011027159,0.005095951,0.005330502,0.9142862,0.0010077823,0.056676026],"study_design_scores_gemma":[0.000039421582,0.00010226263,0.010437445,0.00010002269,0.00012667953,0.00035090232,0.004512637,0.06351352,0.014849172,0.8836205,0.022237455,0.00010993581],"about_ca_topic_score_codex":0.0029404827,"about_ca_topic_score_gemma":0.003577682,"teacher_disagreement_score":0.0045529054,"about_ca_system_score_codex":0.0015245918,"about_ca_system_score_gemma":0.0014450935,"threshold_uncertainty_score":0.01737529},"labels":[],"label_agreement":null},{"id":"W2048782995","doi":"10.1142/s1793840609002147","title":"Correcting Arabic OCR Errors Using Improved Topic-Based Language Models","year":2009,"lang":"en","type":"article","venue":"International Journal of Computer Processing Of Languages","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure","funders":"Université de Montréal","keywords":"Computer science; Natural language processing; Artificial intelligence; Language model; Word error rate; Arabic; Word (group theory); Error detection and correction; Proofreading; Process (computing); Speech recognition; Linguistics; Algorithm; Programming language","score_opus":0.01669986960573842,"score_gpt":0.3202112514965974,"score_spread":0.30351138189085897,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2048782995","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06028176,0.001192583,0.93059474,0.00038943483,0.000272754,0.00015079795,0.0004441397,0.005445194,0.0012285752],"genre_scores_gemma":[0.58627814,0.0014288356,0.4003135,0.00024793143,0.0006000324,0.00044664144,0.0018986498,0.0011346231,0.0076516466],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9982912,0.0004871092,0.00017355283,0.00045489127,0.00046528404,0.00012799035],"domain_scores_gemma":[0.99520653,0.0019304947,0.0004337557,0.00037923083,0.0019723775,0.000077617995],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021677127,0.0013080456,0.0013227572,0.0025431472,0.00058290333,0.0014282968,0.001567411,0.0013327434,0.0011174286],"category_scores_gemma":[0.0068544135,0.0005936844,0.0019128842,0.0014453139,0.00053116225,0.0026386145,0.0008473401,0.0012715244,0.0013999889],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00087582273,0.0003339661,0.007144189,0.00067758677,0.0005146134,0.00047207664,0.0012356506,0.3373089,0.045658384,0.0067115687,0.008586519,0.5904807],"study_design_scores_gemma":[0.00002387715,0.00005483766,0.0010560964,0.0000139890635,0.000090600086,0.00014226654,0.000050387367,0.98606944,0.0091160415,0.0014365542,0.0019066368,0.000039316],"about_ca_topic_score_codex":0.01135796,"about_ca_topic_score_gemma":0.011167284,"teacher_disagreement_score":0.01135796,"about_ca_system_score_codex":0.0008567248,"about_ca_system_score_gemma":0.0012251548,"threshold_uncertainty_score":0.022583723},"labels":[],"label_agreement":null},{"id":"W2050242310","doi":"10.3115/1118905.1118909","title":"Statistical translation alignment with compositionality constraints","year":2003,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Principle of compositionality; Computer science; Constraint (computer-aided design); Translation (biology); Natural language processing; Task (project management); Artificial intelligence; Word (group theory); Machine translation; Linguistics; Mathematics; Engineering","score_opus":0.016564239628769385,"score_gpt":0.2762380755591233,"score_spread":0.25967383593035387,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2050242310","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0018730391,0.00012144107,0.99553204,0.00009060851,0.000060149116,0.000046808364,0.00009417438,0.0010213951,0.0011602159],"genre_scores_gemma":[0.07734599,0.00033724398,0.9166076,0.00016748066,0.00025326715,0.00037893606,0.0013694542,0.0010113707,0.0025286027],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99375737,0.0026664156,0.0004648995,0.0010589199,0.0018662122,0.00018627653],"domain_scores_gemma":[0.9924287,0.0034833134,0.0007074361,0.0020044835,0.0012448964,0.0001310583],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0039235502,0.0014159024,0.0013299449,0.0019752232,0.0015967687,0.0022298056,0.0019230281,0.0013230867,0.004961567],"category_scores_gemma":[0.016303686,0.0012750474,0.0015058919,0.0030966394,0.0015242558,0.0041191224,0.0031933298,0.0027627859,0.0038881737],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004109614,0.00018423652,0.0010521057,0.0009024241,0.00038031847,0.0005540431,0.00085138605,0.10334362,0.0570518,0.23476653,0.011754945,0.58874756],"study_design_scores_gemma":[0.00011573802,0.00019719005,0.00048386288,0.00010381686,0.00013896101,0.00048814804,0.00021520177,0.63742036,0.044228166,0.26828548,0.048215374,0.000107694046],"about_ca_topic_score_codex":0.0014512464,"about_ca_topic_score_gemma":0.0037462479,"teacher_disagreement_score":0.004961567,"about_ca_system_score_codex":0.0006307623,"about_ca_system_score_gemma":0.0032353464,"threshold_uncertainty_score":0.020749986},"labels":[],"label_agreement":null},{"id":"W2050369426","doi":"10.3115/1118735.1118737","title":"Induction of classification from lexicon expansion","year":2002,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Academia Sinica; National Science Council","keywords":"WordNet; Computer science; Natural language processing; Lexicon; Artificial intelligence; Categorization; Information retrieval; Taxonomy (biology); Semantic similarity; Hierarchy; Lexical database","score_opus":0.04685674669557996,"score_gpt":0.26995799552991673,"score_spread":0.22310124883433677,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2050369426","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04060521,0.00069230323,0.9264418,0.00092887884,0.00040071257,0.0008380753,0.0012666417,0.008341415,0.020485047],"genre_scores_gemma":[0.3104982,0.00083984056,0.6513108,0.000739556,0.0004997076,0.0015089661,0.0150447115,0.0011020204,0.018456213],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9968321,0.0009784203,0.00022242866,0.0008703524,0.00078024546,0.00031651405],"domain_scores_gemma":[0.9940269,0.0031711825,0.0002735158,0.0012098307,0.0011433602,0.00017525202],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00160451,0.0012413552,0.0010930898,0.004117124,0.0013518204,0.0022316861,0.0019457318,0.00119844,0.008821956],"category_scores_gemma":[0.010975757,0.00062836736,0.00129116,0.0025688158,0.0011394655,0.004453531,0.0037928137,0.0020826666,0.0075321444],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001813122,0.00030807676,0.0038746595,0.00046264916,0.000053688742,0.00029086895,0.0003606314,0.009557262,0.01345055,0.04055034,0.026496738,0.9044132],"study_design_scores_gemma":[0.00014153864,0.0002785598,0.004938749,0.00029083193,0.00016781589,0.00079175906,0.00060499075,0.65636295,0.036742777,0.21419345,0.08539422,0.00009235557],"about_ca_topic_score_codex":0.0017148792,"about_ca_topic_score_gemma":0.004050135,"teacher_disagreement_score":0.008821956,"about_ca_system_score_codex":0.0013353124,"about_ca_system_score_gemma":0.0019170784,"threshold_uncertainty_score":0.029512405},"labels":[],"label_agreement":null},{"id":"W2050712820","doi":"10.1145/775047.775138","title":"Discovering word senses from text","year":2002,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":597,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Word (group theory); Computer science; Cluster analysis; Precision and recall; Set (abstract data type); Similarity (geometry); Centroid; Artificial intelligence; Natural language processing; Feature (linguistics); Feature vector; Element (criminal law); Space (punctuation); Cluster (spacecraft); Domain (mathematical analysis); Recall; Information retrieval; Mathematics; Linguistics; Image (mathematics)","score_opus":0.016333199237866095,"score_gpt":0.23864345945404686,"score_spread":0.22231026021618078,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2050712820","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.420488,0.0035553544,0.5399799,0.000856318,0.00035313706,0.0010403069,0.016503455,0.0074890736,0.009734431],"genre_scores_gemma":[0.356296,0.0014379752,0.6149664,0.00021272995,0.0001546019,0.0005830666,0.023580195,0.0005921966,0.0021767763],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99626213,0.0008185006,0.0005463578,0.0012786031,0.00092691457,0.00016744949],"domain_scores_gemma":[0.99162227,0.0045166654,0.00093389774,0.0010519669,0.0016583177,0.00021679392],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017522232,0.0012192827,0.0011702464,0.013209365,0.0013878815,0.0022629232,0.001090056,0.0011108535,0.0016529968],"category_scores_gemma":[0.014694427,0.0005952127,0.0013945047,0.0058559733,0.00096472184,0.004321162,0.0021658372,0.0008536202,0.0016581811],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007029556,0.00038751872,0.05773765,0.0024504296,0.00048251796,0.0023541562,0.004287329,0.0110694375,0.061914556,0.014102001,0.021430872,0.82308054],"study_design_scores_gemma":[0.00025519237,0.00080326607,0.08306502,0.001238367,0.0009964285,0.011259894,0.012552222,0.43329802,0.15335666,0.13278788,0.1698433,0.0005437078],"about_ca_topic_score_codex":0.0019350456,"about_ca_topic_score_gemma":0.004160596,"teacher_disagreement_score":0.013209365,"about_ca_system_score_codex":0.00066466734,"about_ca_system_score_gemma":0.0014212176,"threshold_uncertainty_score":0.009266794},"labels":[],"label_agreement":null},{"id":"W2051349122","doi":"10.3115/1118853.1118874","title":"Letter level learning for language independent diacritics restoration","year":2002,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":35,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Romanian; Czech; Computer science; Natural language processing; Artificial intelligence; Linguistics","score_opus":0.03785590327783317,"score_gpt":0.288233978363281,"score_spread":0.25037807508544785,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2051349122","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021092871,0.00032212623,0.9634866,0.00013781236,0.00022739927,0.000069236616,0.00012050922,0.011950917,0.0025925543],"genre_scores_gemma":[0.2803091,0.00025338624,0.7044407,0.00023151062,0.00017075098,0.00010490229,0.0007267995,0.000901279,0.012861525],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991898,0.00014298597,0.000055418124,0.0003018399,0.00022732592,0.00008273515],"domain_scores_gemma":[0.9980566,0.000654636,0.00018542622,0.00048562518,0.00052216277,0.00009559053],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011877228,0.000937952,0.0009228521,0.0015250942,0.00083732133,0.0014275258,0.001975632,0.0012557271,0.0046522655],"category_scores_gemma":[0.0034881053,0.0003949463,0.0006876073,0.00090674247,0.0009528002,0.0022849652,0.0014209462,0.0022814416,0.005411844],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029756918,0.00015871991,0.0009281401,0.00021822448,0.00004603515,0.00012374128,0.00018311667,0.011272581,0.060628146,0.0066437135,0.005352852,0.914147],"study_design_scores_gemma":[0.00007496425,0.00019335095,0.001409845,0.000048281752,0.00006447948,0.0004781659,0.00019873054,0.78812593,0.1502796,0.0274714,0.031576604,0.0000786591],"about_ca_topic_score_codex":0.0011058715,"about_ca_topic_score_gemma":0.0017046479,"teacher_disagreement_score":0.0046522655,"about_ca_system_score_codex":0.0005680123,"about_ca_system_score_gemma":0.0010221603,"threshold_uncertainty_score":0.015563369},"labels":[],"label_agreement":null},{"id":"W2051593977","doi":"10.1162/coli_a_00002","title":"Generating Phrasal and Sentential Paraphrases: A Survey of Data-Driven Methods","year":2010,"lang":"en","type":"article","venue":"Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":302,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Atomic Energy of Canada Limited; National Science Foundation","keywords":"Paraphrase; Computer science; Natural language processing; Task (project management); Artificial intelligence; Parallel corpora; Field (mathematics); Linguistics; Machine translation","score_opus":0.06410955906269462,"score_gpt":0.40156873393396203,"score_spread":0.3374591748712674,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2051593977","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012093974,0.030935701,0.94376403,0.0013705837,0.00012204945,0.00095785473,0.0023462516,0.0042715934,0.0041379067],"genre_scores_gemma":[0.05460917,0.019108256,0.9149152,0.00045814825,0.00016279898,0.0008902143,0.0072048553,0.00081288326,0.0018385034],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.98880017,0.0052458486,0.0014358081,0.0014577333,0.002913964,0.0001464791],"domain_scores_gemma":[0.94835085,0.03835462,0.0018615134,0.0047159623,0.006364801,0.00035224247],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0104788225,0.0017598983,0.0026343726,0.007297795,0.00095831417,0.0034425997,0.005062104,0.0019202672,0.0039218846],"category_scores_gemma":[0.03765097,0.0011637366,0.0016905803,0.009864944,0.001401756,0.0058853915,0.0022303658,0.0020935056,0.0034577977],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017806429,0.00031503785,0.0012340406,0.0037613797,0.00010972399,0.00010865229,0.0006247784,0.0040202504,0.007104069,0.006241766,0.0054965927,0.9708057],"study_design_scores_gemma":[0.0006051477,0.0016917941,0.011819505,0.0041715205,0.00088692,0.0060038255,0.004444236,0.4604189,0.13922341,0.12083806,0.24929361,0.0006031058],"about_ca_topic_score_codex":0.0017435033,"about_ca_topic_score_gemma":0.0023429238,"teacher_disagreement_score":0.0104788225,"about_ca_system_score_codex":0.00094619987,"about_ca_system_score_gemma":0.0018568035,"threshold_uncertainty_score":0.055417955},"labels":[],"label_agreement":null},{"id":"W2051999777","doi":"10.3758/bf03194965","title":"Priming relations in ambiguous noun-noun combinations","year":2002,"lang":"en","type":"article","venue":"Memory & Cognition","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":73,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Psychology; Noun phrase; Noun; Priming (agriculture); Schema (genetic algorithms); Interpretation (philosophy); Phrase; Linguistics; Cognitive psychology; Key (lock); Social psychology; Computer science","score_opus":0.023418261104271868,"score_gpt":0.25007029289518984,"score_spread":0.22665203179091797,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2051999777","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.87659764,0.0018900832,0.042292815,0.001115749,0.0016654764,0.00049347116,0.0009859127,0.0012237803,0.073734984],"genre_scores_gemma":[0.9698278,0.00066663884,0.021146495,0.0005800454,0.00052133243,0.00040817872,0.00056763197,0.0011284716,0.0051533985],"study_design_codex":"bench_or_experimental","study_design_gemma":"observational","domain_scores_codex":[0.9975393,0.0006619354,0.00021250028,0.0007965342,0.00059973367,0.00019009982],"domain_scores_gemma":[0.97402406,0.020124417,0.0016987285,0.0021996067,0.001133995,0.0008192046],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028107879,0.001220868,0.0014662301,0.00090630876,0.00137798,0.0027848538,0.0014229646,0.0017972651,0.027410869],"category_scores_gemma":[0.035831604,0.0014373598,0.0006720223,0.0010768812,0.0012095654,0.007748729,0.0027960395,0.0022567045,0.0031138728],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.006508297,0.000526829,0.005806618,0.002400215,0.00010442138,0.001648895,0.0050614956,0.0005948773,0.85926414,0.055249833,0.0034131715,0.05942122],"study_design_scores_gemma":[0.0022335798,0.0037084115,0.10997889,0.0007420679,0.0011820782,0.013193217,0.0036912018,0.016705595,0.3465746,0.4696034,0.031957276,0.00042965886],"about_ca_topic_score_codex":0.0007278312,"about_ca_topic_score_gemma":0.00083307707,"teacher_disagreement_score":0.027410869,"about_ca_system_score_codex":0.0007045657,"about_ca_system_score_gemma":0.0008672948,"threshold_uncertainty_score":0.09169847},"labels":[],"label_agreement":null},{"id":"W2052227241","doi":"10.1145/1289600.1289602","title":"Adaptive text correction with Web-crawled domain-dependent dictionaries","year":2007,"lang":"en","type":"article","venue":"ACM Transactions on Speech and Language Processing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Trigram; Vocabulary; Natural language processing; Domain (mathematical analysis); Information retrieval; Word (group theory); Artificial intelligence; Web page; Crawling; Punctuation; Linguistics; World Wide Web","score_opus":0.008003122257323006,"score_gpt":0.25074320704803005,"score_spread":0.24274008479070705,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2052227241","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2529732,0.0011843689,0.65903944,0.0003326262,0.000270491,0.0008032728,0.0020172053,0.07972155,0.003657857],"genre_scores_gemma":[0.30058587,0.0005381117,0.6807449,0.00019249314,0.000079642625,0.000465995,0.0064838016,0.0036890793,0.007220119],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99799275,0.0006028052,0.0002432243,0.00052234647,0.00055748207,0.000081434046],"domain_scores_gemma":[0.9814299,0.008723803,0.0011015758,0.0050277743,0.003455435,0.00026141197],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001528184,0.0012171477,0.0012574226,0.0031464489,0.00065421703,0.0012170492,0.0013626122,0.0008773308,0.0017663983],"category_scores_gemma":[0.021536484,0.00070223695,0.000598648,0.002912188,0.0005772642,0.0025585066,0.00148824,0.0010218711,0.0028891729],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007466272,0.00073962036,0.013504569,0.0009826246,0.00026950525,0.0007490371,0.001371365,0.028967073,0.09859189,0.0018542091,0.012497717,0.8397259],"study_design_scores_gemma":[0.000277876,0.0006226463,0.012742062,0.00012757239,0.0003289746,0.0016742971,0.0008417614,0.68009555,0.26705912,0.003618448,0.032420315,0.00019149201],"about_ca_topic_score_codex":0.0047981422,"about_ca_topic_score_gemma":0.0073138997,"teacher_disagreement_score":0.0047981422,"about_ca_system_score_codex":0.0004148117,"about_ca_system_score_gemma":0.0011409895,"threshold_uncertainty_score":0.009540439},"labels":[],"label_agreement":null},{"id":"W2052696540","doi":"10.1017/s1351324908005044","title":"A corpus-based analysis of argument realization by preposition structures","year":2009,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"FrameNet; Computer science; Realization (probability); Argument (complex analysis); Natural language processing; Artificial intelligence; Semantics (computer science); Semantic role labeling; Linguistics; Frame (networking); Parsing; Programming language; Sentence; Mathematics; Philosophy","score_opus":0.0024766510758832297,"score_gpt":0.231932056132034,"score_spread":0.22945540505615075,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2052696540","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8877897,0.0017132261,0.0697191,0.00050305197,0.00019702935,0.00027065453,0.017517302,0.0006122798,0.021677623],"genre_scores_gemma":[0.90894353,0.0008246829,0.06358326,0.000055181226,0.000052255196,0.00036083843,0.023630204,0.00024011979,0.002309964],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9984251,0.000729441,0.00013763069,0.00028667058,0.0003763405,0.000044818284],"domain_scores_gemma":[0.98306227,0.012691675,0.00079582515,0.0018417927,0.001497097,0.00011136005],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020926541,0.0002466572,0.00037110352,0.0053412947,0.0014625423,0.001467859,0.0004965686,0.0005144405,0.004040629],"category_scores_gemma":[0.012617771,0.00036250486,0.00031033225,0.007999713,0.0011669007,0.0018978646,0.0011262504,0.0009275149,0.00076431973],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002974237,0.0010458942,0.08902512,0.00493756,0.00053556677,0.004548921,0.037271217,0.01724724,0.11396074,0.15616016,0.045812294,0.52648103],"study_design_scores_gemma":[0.00031679028,0.00055795006,0.37615746,0.0010032183,0.00052121986,0.0050923782,0.022471808,0.09639845,0.098318085,0.046141587,0.35265526,0.00036582153],"about_ca_topic_score_codex":0.003706906,"about_ca_topic_score_gemma":0.0068263696,"teacher_disagreement_score":0.0053412947,"about_ca_system_score_codex":0.0008323531,"about_ca_system_score_gemma":0.0007376922,"threshold_uncertainty_score":0.013517261},"labels":[],"label_agreement":null},{"id":"W2052806359","doi":"10.3115/1613715.1613843","title":"Computing word-pair antonymy","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":93,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada; University of Maryland; National Science Foundation","keywords":"Computer science; Word (group theory); Natural language processing; Set (abstract data type); Artificial intelligence; Measure (data warehouse); Linguistics; Data mining; Programming language","score_opus":0.017528222054990956,"score_gpt":0.26493948513170673,"score_spread":0.24741126307671577,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2052806359","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7177338,0.0018135523,0.2557027,0.00039179376,0.00024681876,0.00032786993,0.0033370606,0.0029011206,0.017545212],"genre_scores_gemma":[0.8130674,0.0004123402,0.17848518,0.000061514285,0.00014447817,0.0002644212,0.005971866,0.00030984217,0.0012829454],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9910953,0.0020906255,0.0014880534,0.0023591209,0.0025753144,0.00039157431],"domain_scores_gemma":[0.97063035,0.016144589,0.0036970507,0.00323805,0.005652448,0.0006374238],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005596765,0.00087207253,0.0014180918,0.016803566,0.0024546012,0.0032020207,0.0010762572,0.0012837475,0.004634322],"category_scores_gemma":[0.0544956,0.0004533214,0.0009266099,0.008775363,0.0014684827,0.0062952805,0.004670094,0.0010213344,0.0024278492],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015528215,0.00049479085,0.21878497,0.0022208635,0.0010103107,0.0015754522,0.008657901,0.008561607,0.07324377,0.059047624,0.013422564,0.6114273],"study_design_scores_gemma":[0.00025045045,0.0010100756,0.35036656,0.000506197,0.00081774197,0.009230948,0.011604271,0.2599488,0.09250112,0.20429417,0.06891653,0.00055316044],"about_ca_topic_score_codex":0.0012067403,"about_ca_topic_score_gemma":0.0017834855,"teacher_disagreement_score":0.016803566,"about_ca_system_score_codex":0.00074844726,"about_ca_system_score_gemma":0.0014089295,"threshold_uncertainty_score":0.029598892},"labels":[],"label_agreement":null},{"id":"W2053090762","doi":"10.7202/003739ar","title":"Tools for Machine-Aided Translation: The CMU TWS","year":2002,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Machine translation; Workstation; Computer-assisted translation; Variety (cybernetics); Interface (matter); Translation (biology); Artificial intelligence; Human–computer interaction; Rule-based machine translation; Feature (linguistics); User interface; Synchronous context-free grammar; Grammar; Natural language processing; Example-based machine translation; Programming language; Operating system","score_opus":0.09181187059876134,"score_gpt":0.29494814760724053,"score_spread":0.2031362770084792,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2053090762","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004236292,0.0013745667,0.8137724,0.0024151052,0.0009664121,0.0005542143,0.0025199202,0.12541789,0.04874322],"genre_scores_gemma":[0.01806687,0.001180482,0.8869885,0.000624863,0.00046628204,0.0007467777,0.008725547,0.01956478,0.06363589],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99803907,0.00047114826,0.00018534978,0.00040153036,0.00077665143,0.00012621279],"domain_scores_gemma":[0.99630725,0.0009844791,0.00011066452,0.0011936357,0.0008575596,0.0005464216],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036397097,0.0014133003,0.00090835855,0.003516693,0.0011996684,0.0030669519,0.003221051,0.001713208,0.06082255],"category_scores_gemma":[0.009207665,0.0014236213,0.0010107851,0.004648513,0.001093257,0.0055064275,0.0044143694,0.0031495374,0.04111708],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002583902,0.0001563488,0.0003947261,0.00032037005,0.000032519627,0.00053776364,0.0009976657,0.0021096405,0.010946388,0.03614339,0.2782509,0.66985196],"study_design_scores_gemma":[0.00015097242,0.00008708032,0.00077092,0.00027110038,0.000026721049,0.0010379859,0.00014628968,0.02622777,0.009663783,0.024255142,0.9372734,0.00008887221],"about_ca_topic_score_codex":0.003984663,"about_ca_topic_score_gemma":0.00363603,"teacher_disagreement_score":0.06082255,"about_ca_system_score_codex":0.0012560287,"about_ca_system_score_gemma":0.0023862703,"threshold_uncertainty_score":0.20347166},"labels":[],"label_agreement":null},{"id":"W2053183061","doi":"10.7202/001904ar","title":"Unification and Machine Translation","year":2002,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Unification; Rotation formalisms in three dimensions; Computer science; Context (archaeology); Machine translation; Translation (biology); Software engineering; Artificial intelligence; Management science; Programming language; Engineering; Mathematics; Chemistry; History","score_opus":0.05670963729019329,"score_gpt":0.26905635692021573,"score_spread":0.21234671963002244,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2053183061","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0095432745,0.23610008,0.43141684,0.033805847,0.00727751,0.00011957872,0.00045222175,0.0012611302,0.2800235],"genre_scores_gemma":[0.4443583,0.12629534,0.26448393,0.0085702585,0.016325297,0.00058002124,0.0020588988,0.00089796865,0.13643008],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99618065,0.0017821969,0.00025964167,0.0007761743,0.0007644732,0.0002369443],"domain_scores_gemma":[0.9978769,0.0012914869,0.00017101827,0.00042011813,0.00018683104,0.000053624502],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028380125,0.001034762,0.0014539418,0.0031029654,0.002506548,0.005411587,0.001842991,0.0036616796,0.01700411],"category_scores_gemma":[0.007149505,0.00074018916,0.0012937001,0.004369114,0.008151162,0.013273676,0.004285841,0.0040324777,0.006756513],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000016344688,0.000008593158,0.00008545933,0.00019824482,0.000021852178,0.00008770782,0.00028766764,0.0010263891,0.00019014304,0.9461698,0.0085918475,0.043315843],"study_design_scores_gemma":[0.0000041701296,0.0000043548857,0.00006804986,0.00007212381,0.0000052733026,0.000087285225,0.0000472395,0.0012289396,0.00015742927,0.9561026,0.042213604,0.000008986339],"about_ca_topic_score_codex":0.0026986417,"about_ca_topic_score_gemma":0.0015198144,"teacher_disagreement_score":0.01700411,"about_ca_system_score_codex":0.0027563677,"about_ca_system_score_gemma":0.0015259912,"threshold_uncertainty_score":0.056884408},"labels":[],"label_agreement":null},{"id":"W2053613609","doi":"10.7202/003778ar","title":"Traduction automatique et classes d'objets : le problème de « porter un vêtement » en français et en coréen","year":2002,"lang":"fr","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Philosophy","score_opus":0.035095988215545465,"score_gpt":0.28711899735257806,"score_spread":0.25202300913703257,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2053613609","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15716262,0.0027691585,0.77699465,0.0062841433,0.00048092473,0.00021070203,0.0011845927,0.004669721,0.05024354],"genre_scores_gemma":[0.6091121,0.001846095,0.3149932,0.0008221098,0.00016235969,0.00016243418,0.0018790863,0.0039108433,0.06711182],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99624336,0.0009416099,0.00032269442,0.0010628296,0.001122019,0.0003074609],"domain_scores_gemma":[0.99394155,0.0022338412,0.0005186929,0.0016896542,0.0014154137,0.00020078204],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004207355,0.00075991167,0.0006501504,0.0020867346,0.002827048,0.0069794236,0.0012381237,0.0015583653,0.00622625],"category_scores_gemma":[0.007823306,0.00084895565,0.0010139119,0.0031477127,0.0042907395,0.008488585,0.0023884485,0.002384701,0.0021808937],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022380959,0.000071060706,0.01016266,0.00044177184,0.000069347756,0.00055137475,0.018184906,0.0034482207,0.014695675,0.5039229,0.014929491,0.43329865],"study_design_scores_gemma":[0.0000470344,0.00011223489,0.009336925,0.00031980342,0.00010334852,0.00096132123,0.009001762,0.021503072,0.024659142,0.18296258,0.75084907,0.0001437346],"about_ca_topic_score_codex":0.04912296,"about_ca_topic_score_gemma":0.035534143,"teacher_disagreement_score":0.04912296,"about_ca_system_score_codex":0.0043714484,"about_ca_system_score_gemma":0.0039988346,"threshold_uncertainty_score":0.09767407},"labels":[],"label_agreement":null},{"id":"W2053918322","doi":"10.5539/res.v2n2p223","title":"Lexical Computational Models: The Case of the Byzantine Greek Language","year":2010,"lang":"en","type":"article","venue":"Review of European Studies","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Byzantine architecture; Computer science; Natural language processing; Representation (politics); Artificial intelligence; Metric (unit); Greek language; Language model; Linguistics; History; Archaeology; Engineering","score_opus":0.0379944468238408,"score_gpt":0.3380055003269573,"score_spread":0.3000110535031165,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2053918322","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19811417,0.007273432,0.634414,0.0108944075,0.00031739727,0.00021616188,0.0012189479,0.000633352,0.14691806],"genre_scores_gemma":[0.789646,0.0030899185,0.18571201,0.00053959206,0.00024445774,0.00028064894,0.0008857702,0.0002518134,0.019349745],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991769,0.00043643994,0.000048927523,0.0001493681,0.00014727804,0.00004111585],"domain_scores_gemma":[0.99757665,0.001882716,0.00010466183,0.00023995998,0.00015167304,0.0000444024],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011889713,0.0004691833,0.000695948,0.0013782725,0.001108765,0.0050175563,0.0011277249,0.0012666621,0.0062391474],"category_scores_gemma":[0.0064323186,0.0003251859,0.00082282454,0.0019182548,0.0023453494,0.006273852,0.0017998421,0.0011961324,0.0012348578],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000037576825,0.000029881307,0.00095417444,0.00019380313,0.000039270064,0.00057164946,0.002145251,0.025492715,0.0004983957,0.94528717,0.0021240572,0.022626027],"study_design_scores_gemma":[0.000015414957,0.00001700322,0.0007696211,0.00010517859,0.000019860168,0.0003036404,0.0011584093,0.25463116,0.00034638552,0.71772724,0.024881378,0.00002471162],"about_ca_topic_score_codex":0.0043923426,"about_ca_topic_score_gemma":0.004641747,"teacher_disagreement_score":0.0062391474,"about_ca_system_score_codex":0.0012235519,"about_ca_system_score_gemma":0.000906434,"threshold_uncertainty_score":0.020871997},"labels":[],"label_agreement":null},{"id":"W2056030495","doi":"10.3166/isi.15.2.49-71","title":"Approche bayésienne de la composition sémantique dans les systèmes de dialogue oral","year":2010,"lang":"fr","type":"article","venue":"Ingénierie des systèmes d information","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Philosophy","score_opus":0.01163000562191922,"score_gpt":0.25408434081251197,"score_spread":0.24245433519059276,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2056030495","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004228437,0.00033728953,0.9936526,0.00025536082,0.00003419797,0.000045126428,0.00004898946,0.0003373759,0.0010606633],"genre_scores_gemma":[0.16939269,0.0007122911,0.8253615,0.00020354014,0.00016061704,0.000351901,0.00021429303,0.00022576965,0.0033774092],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99393684,0.0023846393,0.0003877886,0.0012615523,0.0018281998,0.00020098896],"domain_scores_gemma":[0.99248403,0.005905002,0.00030302364,0.00040283744,0.0008090188,0.000096097174],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010677883,0.0013683966,0.0014596824,0.0022591066,0.0016023978,0.004206288,0.002188242,0.0019261177,0.0032251978],"category_scores_gemma":[0.019570129,0.001640215,0.0016737165,0.0012385779,0.0027726658,0.0055362387,0.001535293,0.0031997305,0.0010604641],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00068679085,0.00014481648,0.0043039327,0.00062741316,0.00035147092,0.00046531478,0.0027718903,0.167023,0.012602597,0.3268555,0.0027063685,0.48146084],"study_design_scores_gemma":[0.0000677634,0.000047095178,0.00073762547,0.00012462218,0.00010170803,0.00027437453,0.0002404954,0.7189505,0.0054186373,0.26307365,0.0108894175,0.00007412179],"about_ca_topic_score_codex":0.010103297,"about_ca_topic_score_gemma":0.01179252,"teacher_disagreement_score":0.010677883,"about_ca_system_score_codex":0.0026985302,"about_ca_system_score_gemma":0.0026702476,"threshold_uncertainty_score":0.05647075},"labels":[],"label_agreement":null},{"id":"W2056258147","doi":"10.1142/s0218213005002028","title":"APPLICATIVE AND COMBINATORY CATEGORIAL GRAMMAR AND SUBORDINATE CONSTRUCTIONS IN FRENCH","year":2005,"lang":"en","type":"article","venue":"International Journal of Artificial Intelligence Tools","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Trois-Rivières","funders":"","keywords":"Combinatory categorial grammar; Categorial grammar; Combinatory logic; Computer science; Cognitive grammar; Interrogative; Linguistics; Link grammar; Generalization; Grammar; Mildly context-sensitive grammar formalism; Emergent grammar; Head-driven phrase structure grammar; Generative grammar; Natural language processing; Artificial intelligence; Cognition; Programming language; Mathematics; Psychology","score_opus":0.024829668831408476,"score_gpt":0.3139610256161994,"score_spread":0.2891313567847909,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2056258147","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6199811,0.0042478684,0.28291014,0.0016496405,0.000086802575,0.00009650946,0.00036611842,0.00076170586,0.08990016],"genre_scores_gemma":[0.9787208,0.00040248458,0.017133137,0.000090600275,0.000028442111,0.0000282067,0.00010922833,0.00004313878,0.003443891],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99885607,0.0005403673,0.000052499945,0.00019241135,0.00019171879,0.00016693457],"domain_scores_gemma":[0.9991671,0.0003774362,0.00012570484,0.00010681574,0.00017137779,0.000051634892],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011981387,0.00051023846,0.0003551057,0.0021366521,0.0010071449,0.0030505296,0.0003629892,0.0006068829,0.001585565],"category_scores_gemma":[0.0012358269,0.00023580034,0.0009443491,0.00147168,0.0051956913,0.0018391974,0.0008164599,0.0005834208,0.00020181862],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000023499755,0.0000094631105,0.0044211852,0.00003929824,0.000014793616,0.0004174799,0.0032721967,0.0026345397,0.0019744018,0.97095054,0.00038266965,0.015859773],"study_design_scores_gemma":[0.00001737584,0.00005949861,0.024876755,0.000059143786,0.00007403104,0.0017232946,0.0030693857,0.022030158,0.0027648099,0.89156145,0.05367824,0.0000858536],"about_ca_topic_score_codex":0.023464924,"about_ca_topic_score_gemma":0.014299532,"teacher_disagreement_score":0.023464924,"about_ca_system_score_codex":0.002997367,"about_ca_system_score_gemma":0.0012128957,"threshold_uncertainty_score":0.04665667},"labels":[],"label_agreement":null},{"id":"W2056416066","doi":"10.1109/icosc.2015.7050842","title":"Corpus-based analysis of rhetorical relations: A study of lexical cues","year":2015,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Rhetorical question; Elaboration; Computer science; Newspaper; Identification (biology); Relation (database); Linguistics; Corpus linguistics; Natural language processing; Computational linguistics; Artificial intelligence; Sociology; Art","score_opus":0.05319357914890409,"score_gpt":0.3407980723321758,"score_spread":0.2876044931832717,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2056416066","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7714886,0.0068972614,0.19612002,0.00090786204,0.0002631876,0.0006122339,0.0025596635,0.001174364,0.019976806],"genre_scores_gemma":[0.8857978,0.0012935707,0.10950403,0.00010332998,0.00010261213,0.00044042175,0.0016660352,0.00030296805,0.0007892017],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99492776,0.00255715,0.00047814669,0.0007429301,0.001187174,0.000106808366],"domain_scores_gemma":[0.8911775,0.09232462,0.005504731,0.004765982,0.0055939117,0.00063324906],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0046741073,0.00043619136,0.00081714254,0.0062725795,0.001416985,0.0033716946,0.000875147,0.000823941,0.0028099099],"category_scores_gemma":[0.06673152,0.0005301653,0.00032604142,0.0073008076,0.001486413,0.0051133535,0.0018121585,0.0013808971,0.000519671],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018377684,0.0010024406,0.044844642,0.005944591,0.00036907144,0.0022944233,0.04552561,0.0065771793,0.20682336,0.08136412,0.0053795967,0.59803724],"study_design_scores_gemma":[0.0005147157,0.002006094,0.27269223,0.0029632668,0.0009942494,0.005351973,0.0411391,0.25912887,0.16499802,0.11098398,0.13828023,0.00094725296],"about_ca_topic_score_codex":0.0017082763,"about_ca_topic_score_gemma":0.0020847507,"teacher_disagreement_score":0.0062725795,"about_ca_system_score_codex":0.000751682,"about_ca_system_score_gemma":0.0012489163,"threshold_uncertainty_score":0.024719357},"labels":[],"label_agreement":null},{"id":"W2056419757","doi":"10.1145/1113343.1113353","title":"Report on the ACM International Workshop on Methodologies and Evaluation of Lexical Cohesion Techniques in Real-World Applications (ELECTRA 2005) held at SIGIR 2005","year":2005,"lang":"en","type":"article","venue":"ACM SIGIR Forum","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Automatic summarization; Cohesion (chemistry); Natural language processing; Sentence; Computational linguistics; Information retrieval; Artificial intelligence; Question answering; Linguistics","score_opus":0.08345274433676694,"score_gpt":0.4050711201048519,"score_spread":0.32161837576808494,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2056419757","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.043883983,0.06282412,0.73585933,0.03049081,0.022111695,0.010423638,0.012192709,0.008269394,0.07394431],"genre_scores_gemma":[0.078617066,0.02605083,0.68748724,0.00562691,0.0045261565,0.01005371,0.03506746,0.0047069765,0.14786361],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.95829654,0.021535374,0.0025226744,0.0033771545,0.013280871,0.0009873243],"domain_scores_gemma":[0.8917441,0.038573194,0.003018751,0.007135373,0.055418298,0.0041103647],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.06219222,0.0025609867,0.0025587657,0.0054322476,0.0020710437,0.009902578,0.003947977,0.0028729953,0.03306232],"category_scores_gemma":[0.07958161,0.0013931824,0.0017012155,0.0038699992,0.0015241622,0.010841017,0.0058838823,0.0041425517,0.012862264],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001097874,0.001735338,0.0038822773,0.00134284,0.00028509062,0.0002285781,0.0026505878,0.0020903405,0.010323236,0.005342132,0.31014714,0.6608746],"study_design_scores_gemma":[0.0012365464,0.0031591677,0.019110708,0.0028509921,0.0009822458,0.0007214995,0.0044822493,0.025954638,0.038104966,0.020010924,0.8828115,0.00057460775],"about_ca_topic_score_codex":0.010788618,"about_ca_topic_score_gemma":0.010574065,"teacher_disagreement_score":0.06219222,"about_ca_system_score_codex":0.0033998599,"about_ca_system_score_gemma":0.004757213,"threshold_uncertainty_score":0.3289078},"labels":[],"label_agreement":null},{"id":"W2057032961","doi":"10.1007/s10936-011-9175-1","title":"Effects of Grammatical Categories on Letter Detection in Continuous Text","year":2011,"lang":"en","type":"article","venue":"Journal of Psycholinguistic Research","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Percept; Disengagement theory; Function (biology); Reading (process); Psycholinguistics; Computer science; Linguistics; Task (project management); Psychology; Cognitive psychology; Artificial intelligence; Natural language processing; Cognition; Philosophy","score_opus":0.04984775729749586,"score_gpt":0.37826554773260573,"score_spread":0.3284177904351099,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2057032961","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9928219,0.0002727895,0.0024734885,0.00013779572,0.0001847216,0.00005794403,0.00020787356,0.00023658926,0.0036068673],"genre_scores_gemma":[0.9917297,0.0001947589,0.0044469405,0.00022972272,0.00013421402,0.00006605069,0.00038606385,0.00066048745,0.0021521756],"study_design_codex":"bench_or_experimental","study_design_gemma":"observational","domain_scores_codex":[0.9973707,0.001107731,0.0002220098,0.00061150413,0.00046937773,0.00021870754],"domain_scores_gemma":[0.6881893,0.29746774,0.005183573,0.0032078177,0.0034937817,0.0024577654],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034259504,0.00091821497,0.0010830119,0.001011965,0.00058869715,0.0027920112,0.0010007599,0.0019518167,0.011212362],"category_scores_gemma":[0.11713941,0.00091213384,0.00053522777,0.0007186367,0.0008826972,0.0031373715,0.0013958552,0.0018632484,0.0016193289],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.056145765,0.0019157343,0.033955,0.0012017003,0.00027937055,0.0008839041,0.0028240138,0.0032818308,0.767811,0.0013142098,0.0020030623,0.12838446],"study_design_scores_gemma":[0.002085979,0.008698824,0.7314171,0.0002863523,0.0012262997,0.0022953814,0.0016146929,0.08640781,0.15286261,0.010466814,0.0022128846,0.00042517623],"about_ca_topic_score_codex":0.0028893612,"about_ca_topic_score_gemma":0.0023426674,"teacher_disagreement_score":0.011212362,"about_ca_system_score_codex":0.0004657666,"about_ca_system_score_gemma":0.0006440289,"threshold_uncertainty_score":0.037509024},"labels":[],"label_agreement":null},{"id":"W2057604821","doi":"10.1080/15366360802035596","title":"Manifest and Latent Variates","year":2008,"lang":"en","type":"article","venue":"Measurement Interdisciplinary Research and Perspectives","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Statistics; Mathematics; Psychology; Econometrics","score_opus":0.17817029839598855,"score_gpt":0.3724052056309841,"score_spread":0.19423490723499556,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2057604821","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.094847724,0.0025117625,0.8538588,0.009248607,0.00029079363,0.00011615634,0.0015624084,0.0005626118,0.03700113],"genre_scores_gemma":[0.93068737,0.0015448254,0.048656277,0.00041820516,0.00042507786,0.00034194783,0.00085117586,0.00018066894,0.016894478],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9905405,0.0055485247,0.0004984035,0.0015677722,0.0014194527,0.0004253015],"domain_scores_gemma":[0.9518103,0.036147732,0.0030749857,0.005986446,0.002227885,0.0007526075],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011302159,0.0007329667,0.0014436645,0.0028315778,0.0011634723,0.0062181773,0.0010809478,0.0015840539,0.01023006],"category_scores_gemma":[0.05427278,0.00070591696,0.0009888102,0.0036985045,0.006652305,0.013410327,0.0030334839,0.0031363158,0.00083242997],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000011448643,0.000012229092,0.0011702771,0.000027715823,0.000017660868,0.000012053594,0.0002422595,0.0005172611,0.00007053772,0.99233854,0.00036474716,0.0052153477],"study_design_scores_gemma":[0.0000053532995,0.000012295592,0.0013117644,0.00002191987,0.000015962747,0.00003153666,0.00016569553,0.005128999,0.00009376317,0.99159265,0.001611235,0.000008856436],"about_ca_topic_score_codex":0.0013081795,"about_ca_topic_score_gemma":0.0009865039,"teacher_disagreement_score":0.011302159,"about_ca_system_score_codex":0.0020354735,"about_ca_system_score_gemma":0.0017293502,"threshold_uncertainty_score":0.059772253},"labels":[],"label_agreement":null},{"id":"W2057960659","doi":"10.1007/s10579-004-8682-1","title":"Article: Collating Texts Using Progressive Multiple Alignment","year":2004,"lang":"en","type":"article","venue":"Computers and the Humanities","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"Universität Basel","keywords":"Collation; Witness; Computer science; Natural language processing; Phrase; Word (group theory); Base (topology); Point (geometry); Information retrieval; Linguistics; Artificial intelligence; Mathematics","score_opus":0.022612664260272315,"score_gpt":0.2515213186214391,"score_spread":0.2289086543611668,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2057960659","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.041596305,0.004433667,0.8759482,0.0059657213,0.010502478,0.0018402117,0.01446919,0.020394605,0.024849547],"genre_scores_gemma":[0.04767074,0.0021859927,0.89838624,0.0005258347,0.0021103015,0.00053672615,0.023045007,0.004838636,0.020700516],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99515283,0.0013885045,0.0007353519,0.0009913499,0.0015845716,0.00014735144],"domain_scores_gemma":[0.9779417,0.010631083,0.0015953903,0.002570864,0.0067194607,0.0005415593],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003789288,0.0020207893,0.0019651495,0.009922904,0.0036730848,0.0047403877,0.002198759,0.0016574212,0.035863824],"category_scores_gemma":[0.031639248,0.0012716826,0.0012015096,0.009943134,0.001321067,0.0042561223,0.0031663687,0.0030519746,0.021675324],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010620318,0.00027399708,0.0025482676,0.0044469535,0.0004091669,0.003746579,0.0033859222,0.0025875096,0.08633859,0.015150231,0.13578781,0.7442629],"study_design_scores_gemma":[0.00022965683,0.0004982461,0.0049822167,0.0010266311,0.00099723,0.004607878,0.00309049,0.045597646,0.14728145,0.034190927,0.757224,0.00027350092],"about_ca_topic_score_codex":0.0013039247,"about_ca_topic_score_gemma":0.0027330364,"teacher_disagreement_score":0.035863824,"about_ca_system_score_codex":0.00056927075,"about_ca_system_score_gemma":0.003320491,"threshold_uncertainty_score":0.1199764},"labels":[],"label_agreement":null},{"id":"W2058214962","doi":"10.1017/s1351324913000090","title":"On the semantics of noun compounds","year":2013,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Automatic summarization; Computer science; Noun; Natural language processing; Cover (algebra); Artificial intelligence; Proper noun; Semantics (computer science); Subject (documents); Machine translation; Linguistics; Noun phrase; Question answering; Sequence (biology); World Wide Web; Programming language; Philosophy; Chemistry","score_opus":0.0038637183364422247,"score_gpt":0.2078067341440626,"score_spread":0.20394301580762036,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2058214962","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.035982788,0.032373566,0.72077465,0.02252554,0.0024811432,0.00018013013,0.0013525173,0.0010387149,0.18329091],"genre_scores_gemma":[0.6532979,0.023248617,0.2758824,0.006891634,0.004221934,0.00050239795,0.002474962,0.0015295694,0.031950567],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9972754,0.0011368357,0.000332074,0.00054712506,0.00053760293,0.00017103343],"domain_scores_gemma":[0.99465454,0.003259754,0.00031780923,0.00064821675,0.0009365739,0.00018308908],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031815502,0.0010644068,0.0010612903,0.004041028,0.004044357,0.007120886,0.0018640512,0.0024824883,0.0072871526],"category_scores_gemma":[0.008473878,0.0010529693,0.0014643687,0.0042169876,0.01209594,0.028824419,0.0042313486,0.0043413863,0.0022438061],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00002055672,0.000009426374,0.00010721717,0.00006747326,0.000008885433,0.000112901,0.00091268896,0.00057538773,0.00040525864,0.98638505,0.0026913686,0.008703816],"study_design_scores_gemma":[0.0000074059158,0.000008140606,0.00009701839,0.00006213915,0.000008317053,0.00011100302,0.00027357554,0.0019086509,0.00029058105,0.9665482,0.03067159,0.000013444649],"about_ca_topic_score_codex":0.0045205005,"about_ca_topic_score_gemma":0.003227541,"teacher_disagreement_score":0.0072871526,"about_ca_system_score_codex":0.0026225497,"about_ca_system_score_gemma":0.0015644699,"threshold_uncertainty_score":0.024377942},"labels":[],"label_agreement":null},{"id":"W2058621291","doi":"10.1016/j.ipl.2005.09.006","title":"Vertex covering by paths on trees with its applications in machine translation","year":2005,"lang":"en","type":"article","venue":"Information Processing Letters","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Edge cover; Vertex cover; Vertex (graph theory); Path (computing); Combinatorics; Set cover problem; Enhanced Data Rates for GSM Evolution; Tree (set theory); Mathematics; Cover (algebra); Translation (biology); Computer science; Set (abstract data type); Algorithm; Time complexity; Discrete mathematics; Artificial intelligence; Graph","score_opus":0.0062437251733380335,"score_gpt":0.22703625289583707,"score_spread":0.22079252772249902,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2058621291","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.077009775,0.002080202,0.90698737,0.00068717013,0.00018609245,0.0000859795,0.00049627136,0.0011303533,0.011336735],"genre_scores_gemma":[0.39717832,0.002591128,0.5870802,0.00022994836,0.0003868522,0.0002179144,0.0012369286,0.0005866779,0.010491992],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99938285,0.00023227125,0.000035165638,0.00015690082,0.00011522982,0.00007763644],"domain_scores_gemma":[0.9960418,0.0029280693,0.00021566614,0.000420973,0.00023652615,0.00015703203],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006968158,0.00052885705,0.00094060955,0.0016408899,0.0012972696,0.0012400073,0.00093842845,0.0011560555,0.004900102],"category_scores_gemma":[0.0052962094,0.00053973345,0.0010205805,0.0044120564,0.0013139028,0.0026717149,0.001982617,0.0011608773,0.00087678083],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00049196614,0.00013785124,0.0016960308,0.00045281302,0.000071499366,0.0005158956,0.0008119756,0.13044935,0.0065604243,0.5171266,0.012414171,0.32927144],"study_design_scores_gemma":[0.000024790897,0.0000637684,0.0004231952,0.0000378376,0.00003308952,0.00019166376,0.000072075585,0.26379648,0.0019468715,0.7237874,0.009601187,0.000021667882],"about_ca_topic_score_codex":0.0023186046,"about_ca_topic_score_gemma":0.0018643342,"teacher_disagreement_score":0.004900102,"about_ca_system_score_codex":0.00050177367,"about_ca_system_score_gemma":0.0005468774,"threshold_uncertainty_score":0.01639247},"labels":[],"label_agreement":null},{"id":"W2059147389","doi":"10.3115/1609067.1609102","title":"Cube summing, approximate inference with non-local features, and dynamic programming without semirings","year":2009,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Atomic Energy of Canada Limited; National Science Foundation","keywords":"Semiring; Cube (algebra); Residual; Independence (probability theory); Dynamic programming; Computation; Data cube; Pruning; Algorithm; Computer science; Mathematics; Theoretical computer science; Mathematical optimization; Discrete mathematics; Combinatorics; Data mining","score_opus":0.005072351624204556,"score_gpt":0.26259257838324424,"score_spread":0.2575202267590397,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2059147389","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0064306483,0.00023745728,0.9910643,0.0001659337,0.000021999578,0.000019437792,0.00011042484,0.00042929943,0.0015204463],"genre_scores_gemma":[0.17532434,0.0004674813,0.8202077,0.00024813708,0.00012511516,0.0001341921,0.0004662424,0.00037496263,0.0026518737],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9967422,0.00084174494,0.00023105761,0.000692959,0.0012459959,0.00024609166],"domain_scores_gemma":[0.9917482,0.004603971,0.00051636924,0.0020448065,0.0008624336,0.00022411262],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0040978193,0.0008665824,0.0016404126,0.001691486,0.001210751,0.0027391724,0.003052285,0.0010080894,0.0038456605],"category_scores_gemma":[0.01609077,0.000881392,0.0017990292,0.0032459775,0.0028160668,0.009640196,0.0038821765,0.0028432289,0.00079595175],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014151787,0.000053502215,0.00091587997,0.00021588737,0.00007924492,0.00021250166,0.00042830617,0.11149804,0.0032753535,0.6828801,0.0031357629,0.19716395],"study_design_scores_gemma":[0.00001215759,0.000030485724,0.00015974292,0.000024205088,0.00003347056,0.00010791834,0.00004076948,0.32691664,0.0029077092,0.6664471,0.0032939645,0.000025880672],"about_ca_topic_score_codex":0.004321006,"about_ca_topic_score_gemma":0.0063377954,"teacher_disagreement_score":0.004321006,"about_ca_system_score_codex":0.0014740174,"about_ca_system_score_gemma":0.0019755636,"threshold_uncertainty_score":0.021671593},"labels":[],"label_agreement":null},{"id":"W2060185118","doi":"10.7202/002711ar","title":"ELU, un environnement d’expérimentation pour la TA","year":2002,"lang":"fr","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Philosophy; Humanities","score_opus":0.040729829312384364,"score_gpt":0.2747239252889799,"score_spread":0.2339940959765955,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2060185118","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.26545742,0.0011039573,0.6848424,0.0011764,0.00043714372,0.0012665108,0.0016180177,0.011370284,0.032727823],"genre_scores_gemma":[0.50858283,0.0004389083,0.46998274,0.00042867844,0.000058009173,0.0018330801,0.0015058186,0.0019828205,0.015187128],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9883636,0.0060399934,0.0007284716,0.0022674897,0.002066582,0.0005337279],"domain_scores_gemma":[0.9798132,0.012145591,0.00047646032,0.0044392715,0.0027322315,0.00039322546],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009196427,0.001073041,0.00085773476,0.0009797338,0.0013278712,0.0032993427,0.002077813,0.0020403622,0.01220927],"category_scores_gemma":[0.023708122,0.000983143,0.001200253,0.00092558056,0.0019721386,0.003751252,0.0023882156,0.0019815161,0.0036388866],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004811069,0.0024823428,0.012048935,0.0040558088,0.00038066145,0.0012838605,0.022656444,0.050450377,0.38146368,0.074813604,0.010383414,0.43516985],"study_design_scores_gemma":[0.0010156732,0.00950675,0.020295138,0.000991483,0.00052983384,0.0016634782,0.006640521,0.18298908,0.46416977,0.043910507,0.26764953,0.00063817826],"about_ca_topic_score_codex":0.0031952308,"about_ca_topic_score_gemma":0.002784226,"teacher_disagreement_score":0.01220927,"about_ca_system_score_codex":0.0016035852,"about_ca_system_score_gemma":0.0019484644,"threshold_uncertainty_score":0.04863596},"labels":[],"label_agreement":null},{"id":"W2060864821","doi":"10.1300/j104v32n03_08","title":"Word Division in the Transcription of Chinese Script in the Title Fields of Bibliographic Records","year":2001,"lang":"en","type":"article","venue":"Cataloging & Classification Quarterly","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Pinyin; Computer science; Ambiguity; Natural language processing; Information retrieval; Word (group theory); Transcription (linguistics); Orthography; Linguistics; Artificial intelligence; Chinese characters","score_opus":0.026139998092456176,"score_gpt":0.287983754464532,"score_spread":0.26184375637207585,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2060864821","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9643422,0.0007359066,0.018701078,0.00045377383,0.00022391834,0.0005355703,0.004416099,0.0005475804,0.0100439275],"genre_scores_gemma":[0.9513107,0.0004219151,0.037862156,0.00008823931,0.00005361435,0.00025224066,0.0057547954,0.00010766239,0.004148658],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99774736,0.0010218221,0.00033023534,0.00036213166,0.00042766085,0.00011076668],"domain_scores_gemma":[0.9843435,0.010025438,0.0014681934,0.00073198444,0.0031669093,0.00026390166],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019454049,0.00036169676,0.00034797104,0.0020492384,0.00069224345,0.0009953756,0.00023866977,0.00021466536,0.0040868274],"category_scores_gemma":[0.022985004,0.00018303521,0.00020752226,0.003842666,0.0005218812,0.00076327485,0.00057858287,0.0004688603,0.0019058263],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0034713903,0.00020309612,0.07992614,0.0029309038,0.00008363745,0.0015001212,0.035473377,0.002014279,0.15625134,0.003232877,0.014432965,0.7004799],"study_design_scores_gemma":[0.00021644507,0.0014526171,0.59728485,0.0007224117,0.0004287158,0.002021927,0.039789304,0.05318101,0.23303068,0.0038710365,0.06771342,0.00028756223],"about_ca_topic_score_codex":0.008472159,"about_ca_topic_score_gemma":0.016776064,"teacher_disagreement_score":0.008472159,"about_ca_system_score_codex":0.0009877232,"about_ca_system_score_gemma":0.0016701302,"threshold_uncertainty_score":0.016845644},"labels":[],"label_agreement":null},{"id":"W2060953504","doi":"10.3115/1073483.1073500","title":"Word alignment with cohesion constraint","year":2003,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Cohesion (chemistry); Computer science; Constraint (computer-aided design); Syntax; Sentence; Disjoint sets; Word (group theory); Artificial intelligence; Natural language processing; Linguistics; Mathematics","score_opus":0.009935943619383611,"score_gpt":0.240571987813656,"score_spread":0.2306360441942724,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2060953504","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022756428,0.0003308573,0.96672153,0.0003503922,0.000119470504,0.00018219856,0.00065357855,0.0037510016,0.005134689],"genre_scores_gemma":[0.20294684,0.00035070532,0.7864333,0.00032719068,0.00020808348,0.00046234924,0.0038905877,0.001971371,0.003409436],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99492115,0.0020988397,0.00047125027,0.00098708,0.001213694,0.00030806946],"domain_scores_gemma":[0.98703814,0.0058627874,0.0012073305,0.003344899,0.0022548784,0.00029189797],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027598683,0.001186027,0.0014036535,0.0021280714,0.0020448714,0.0021840087,0.002068096,0.0017276355,0.007363343],"category_scores_gemma":[0.018397087,0.00087911397,0.0011801348,0.0042340416,0.0012716377,0.0052681286,0.003391538,0.0021451744,0.0032648104],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009987938,0.00036017387,0.0063340506,0.0017089798,0.00049201207,0.0010159428,0.0016725499,0.09232258,0.097738475,0.16849032,0.041641343,0.5872248],"study_design_scores_gemma":[0.00039754936,0.0004542705,0.0035417292,0.00020340574,0.00039140496,0.0013848959,0.0006446773,0.5956935,0.12645926,0.15518627,0.115363546,0.00027946266],"about_ca_topic_score_codex":0.003387656,"about_ca_topic_score_gemma":0.0052770344,"teacher_disagreement_score":0.007363343,"about_ca_system_score_codex":0.00073278643,"about_ca_system_score_gemma":0.0024037894,"threshold_uncertainty_score":0.024632871},"labels":[],"label_agreement":null},{"id":"W2061023833","doi":"10.3115/1075218.1075232","title":"An unsupervised approach to prepositional phrase attachment using contextually similar words","year":2000,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":77,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Natural language processing; Ambiguity; Artificial intelligence; Phrase; Parsing; Noun phrase; Natural language; Heuristic; Unsupervised learning; Process (computing); Dependency grammar; Noun","score_opus":0.022447445225780505,"score_gpt":0.30407526230633064,"score_spread":0.2816278170805501,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2061023833","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025909912,0.0002521695,0.9668422,0.00014871958,0.000056652196,0.00027883326,0.0005537937,0.0034334222,0.0025242504],"genre_scores_gemma":[0.12859896,0.00028504358,0.863037,0.0001453298,0.00011176655,0.00056345057,0.0031000632,0.00056088384,0.0035974719],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9975981,0.00075412926,0.00016656287,0.0007869767,0.0006048825,0.000089286135],"domain_scores_gemma":[0.9948355,0.0023146656,0.00036410303,0.0010690683,0.0013004824,0.00011620669],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016193775,0.001033178,0.0009892182,0.0032630495,0.0016816669,0.0016008393,0.0019713363,0.0011730274,0.0027142626],"category_scores_gemma":[0.006902191,0.0008053527,0.0010997117,0.0026981707,0.0014006216,0.0024671226,0.001844494,0.0021145972,0.0029373083],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003017799,0.0005657402,0.008216013,0.000504524,0.0002601458,0.00062423415,0.0012082804,0.032365248,0.09477629,0.021906657,0.014005997,0.8252652],"study_design_scores_gemma":[0.00006819927,0.00025144883,0.009390659,0.00009215681,0.00016546895,0.0021348575,0.0005456964,0.84376943,0.06203364,0.046423968,0.034943435,0.00018098888],"about_ca_topic_score_codex":0.0022014545,"about_ca_topic_score_gemma":0.007157985,"teacher_disagreement_score":0.0032630495,"about_ca_system_score_codex":0.0005693392,"about_ca_system_score_gemma":0.0020470337,"threshold_uncertainty_score":0.009080172},"labels":[],"label_agreement":null},{"id":"W2061228083","doi":"10.1075/sl.25.3.11con","title":"Review of Aikhenvald (2000): Classifiers : a typology of noun categorization devices","year":2001,"lang":"en","type":"article","venue":"Studies in Language","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Categorization; Typology; Noun; Computer science; Linguistics; Natural language processing; Artificial intelligence; History; Philosophy; Archaeology","score_opus":0.03259605119647758,"score_gpt":0.3620612829171284,"score_spread":0.32946523172065084,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2061228083","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00049867353,0.96050984,0.00798575,0.00862165,0.0018641793,0.000029706945,0.00017505439,0.00006200203,0.020253107],"genre_scores_gemma":[0.007975664,0.96656203,0.0089895725,0.0049187653,0.0025661779,0.00006594825,0.0003924075,0.0000780892,0.008451429],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.998429,0.00045459234,0.00019279533,0.00022444605,0.000635975,0.00006317649],"domain_scores_gemma":[0.9931672,0.003229644,0.00023763186,0.00021947612,0.002998387,0.00014770341],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035938006,0.000766467,0.0013091111,0.008296281,0.00091448036,0.0036681427,0.0014940888,0.0015233721,0.0055316933],"category_scores_gemma":[0.010611846,0.00051451626,0.00039247106,0.010068691,0.0021285648,0.008518472,0.001310994,0.0023649943,0.0061016236],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000044649612,0.00003054594,0.00059311074,0.0036078622,0.000059317434,0.00007277073,0.0005416711,0.00031648594,0.00045936758,0.04955516,0.22506838,0.7196507],"study_design_scores_gemma":[0.0000056039403,0.000020321006,0.0011443476,0.0020536254,0.000032181015,0.00023858006,0.0002210728,0.00015793742,0.0003426414,0.0152404355,0.9805222,0.000021007352],"about_ca_topic_score_codex":0.0074525825,"about_ca_topic_score_gemma":0.011400414,"teacher_disagreement_score":0.008296281,"about_ca_system_score_codex":0.0027550783,"about_ca_system_score_gemma":0.0030635698,"threshold_uncertainty_score":0.01998955},"labels":[],"label_agreement":null},{"id":"W2061952435","doi":"10.7202/014342ar","title":"Has Computerization Changed Translation?","year":2006,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":63,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Translation (biology); Process (computing); Computer science; Linguistics; Programming language; Philosophy; Chemistry","score_opus":0.0473628939242072,"score_gpt":0.2687487370608563,"score_spread":0.22138584313664908,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2061952435","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0797241,0.05732621,0.061532963,0.5652071,0.023323042,0.00026947216,0.0014107098,0.0022416315,0.20896477],"genre_scores_gemma":[0.68569183,0.048074484,0.06775986,0.11183197,0.015327081,0.00050587684,0.0024228422,0.0040259063,0.06436024],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.973871,0.01575707,0.0016055711,0.004029872,0.0031590946,0.001577442],"domain_scores_gemma":[0.9431726,0.031661905,0.0032012218,0.01097998,0.009974045,0.001010317],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.023686549,0.00082386215,0.0009930194,0.0019781545,0.0024455774,0.010536853,0.0014461654,0.0038674118,0.025491059],"category_scores_gemma":[0.09288175,0.00067421043,0.0010887105,0.006399168,0.009648641,0.019834634,0.0032275843,0.0050558923,0.010020967],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009211483,0.00026979085,0.0075192223,0.0020438712,0.0002171876,0.001105128,0.012176622,0.0012528667,0.00342346,0.33272472,0.07838044,0.5599655],"study_design_scores_gemma":[0.00017810613,0.00024235998,0.0071575637,0.001273051,0.00014033355,0.0019864193,0.0079919705,0.0016859116,0.005499935,0.24601711,0.72766966,0.00015751409],"about_ca_topic_score_codex":0.005432346,"about_ca_topic_score_gemma":0.0034207774,"teacher_disagreement_score":0.025491059,"about_ca_system_score_codex":0.0041154,"about_ca_system_score_gemma":0.0048530516,"threshold_uncertainty_score":0.12526792},"labels":[],"label_agreement":null},{"id":"W2062046281","doi":"10.1007/s10579-009-9083-2","title":"Classification of semantic relations between nominals","year":2009,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":63,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada; University of Ottawa","funders":"","keywords":"Computer science; Task (project management); Natural language processing; SemEval; Sentence; Artificial intelligence; Process (computing); Semantic similarity; Linguistics","score_opus":0.030966356234334785,"score_gpt":0.33024747582626307,"score_spread":0.29928111959192827,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2062046281","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7705426,0.0027312594,0.17110908,0.0016832642,0.00039601838,0.0007887511,0.011246281,0.0040014884,0.037501216],"genre_scores_gemma":[0.9145765,0.0005040888,0.06927261,0.00008638743,0.00009531044,0.00016966376,0.011411955,0.00022259595,0.0036608912],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99486536,0.00153901,0.00056599843,0.00079688133,0.0018820871,0.00035071425],"domain_scores_gemma":[0.9796158,0.0120033035,0.0013066238,0.0017166453,0.004554663,0.0008029657],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0048829443,0.0005685298,0.0007210122,0.007751164,0.001641432,0.0037380795,0.0014066899,0.0011872284,0.0059330375],"category_scores_gemma":[0.022168873,0.00021903966,0.0008952202,0.0033600926,0.0010424724,0.0060146237,0.00175972,0.001125461,0.0016510535],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0036588442,0.0010713787,0.10042476,0.0011622194,0.00023190262,0.00058246404,0.0019509936,0.008358469,0.031434085,0.06389728,0.020121392,0.7671063],"study_design_scores_gemma":[0.00032849438,0.0010272036,0.13452798,0.0006215729,0.0008854677,0.001487108,0.008238938,0.55732405,0.083345555,0.14700766,0.06497576,0.00023021745],"about_ca_topic_score_codex":0.0069790664,"about_ca_topic_score_gemma":0.0071464963,"teacher_disagreement_score":0.007751164,"about_ca_system_score_codex":0.0019897625,"about_ca_system_score_gemma":0.0022175733,"threshold_uncertainty_score":0.025823772},"labels":[],"label_agreement":null},{"id":"W2062246499","doi":"10.5715/jnlp.18.217","title":"Construction of Context Models for Word Sense Disambiguation","year":2011,"lang":"en","type":"article","venue":"Journal of Natural Language Processing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"Ministry of Education, Culture, Sports, Science and Technology","keywords":"Word-sense disambiguation; Word (group theory); Context (archaeology); Computer science; SemEval; Natural language processing; Linguistics; Artificial intelligence; History; Philosophy; Engineering","score_opus":0.02621211769329632,"score_gpt":0.2813093315516963,"score_spread":0.25509721385839995,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2062246499","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010118402,0.0008271072,0.9868897,0.00015150709,0.0000782435,0.000079279744,0.00020049694,0.0009084117,0.0007468697],"genre_scores_gemma":[0.27688676,0.0012692837,0.717652,0.00028654095,0.00021848867,0.0005450605,0.0013690097,0.00051456975,0.0012582691],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99774194,0.0008878513,0.00015347404,0.0007341198,0.0003645629,0.000118162716],"domain_scores_gemma":[0.9962,0.002347955,0.00028600116,0.0005329805,0.00047143686,0.00016156725],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021081134,0.001839285,0.0017825026,0.004371137,0.0016630983,0.001797036,0.0017109954,0.0013379494,0.0017014336],"category_scores_gemma":[0.012522918,0.00147506,0.0022712606,0.0030186754,0.0010987713,0.005472577,0.002680236,0.0025213738,0.0010299447],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00052952784,0.00029059636,0.006747219,0.00062370964,0.00066146714,0.00062687194,0.001250729,0.319558,0.013005146,0.115044855,0.008062129,0.5335999],"study_design_scores_gemma":[0.00003716352,0.000049518883,0.00065780943,0.00006449044,0.000094124414,0.00015254824,0.00010582369,0.8681103,0.0026168097,0.12255928,0.005503129,0.000049081398],"about_ca_topic_score_codex":0.00410116,"about_ca_topic_score_gemma":0.008894399,"teacher_disagreement_score":0.004371137,"about_ca_system_score_codex":0.0009183139,"about_ca_system_score_gemma":0.0017678065,"threshold_uncertainty_score":0.01114887},"labels":[],"label_agreement":null},{"id":"W2063306617","doi":"10.5539/elt.v2n4p33","title":"On Chinese Loan Words from English Language","year":2009,"lang":"en","type":"article","venue":"English Language Teaching","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Loan; China; Linguistics; Psychology; Chinese language; Political science; Business; Finance; Law","score_opus":0.004391729608949878,"score_gpt":0.2590371444645239,"score_spread":0.254645414855574,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2063306617","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.24853782,0.028609408,0.26715183,0.018881174,0.0012043694,0.00017622247,0.00075748767,0.00065294944,0.43402874],"genre_scores_gemma":[0.8900662,0.016556138,0.04692868,0.0013401692,0.00069402123,0.00014691903,0.0006549558,0.00031890208,0.04329385],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99951303,0.00016878179,0.000049980572,0.00007243462,0.00014822233,0.000047615176],"domain_scores_gemma":[0.99908257,0.0005488998,0.000076709766,0.00011927758,0.00015146587,0.0000210286],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00067058596,0.00037193322,0.0003642288,0.0014918525,0.0016852049,0.0016245547,0.00042395183,0.0004311861,0.006914488],"category_scores_gemma":[0.0035935976,0.0002335441,0.00022467072,0.0031649228,0.0031652402,0.006848994,0.001943271,0.0013461604,0.0011222996],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000036178677,0.000012817278,0.0010126497,0.00027879782,0.0000073372016,0.00059394736,0.008030144,0.0006145564,0.001993227,0.84859407,0.0077226907,0.13110353],"study_design_scores_gemma":[0.000020898196,0.00003396048,0.0037766527,0.0003163909,0.000023472376,0.00093501306,0.0046907654,0.007825607,0.006095176,0.7414569,0.23476672,0.000058518395],"about_ca_topic_score_codex":0.004855682,"about_ca_topic_score_gemma":0.0050957818,"teacher_disagreement_score":0.006914488,"about_ca_system_score_codex":0.00095709687,"about_ca_system_score_gemma":0.0009728396,"threshold_uncertainty_score":0.023131311},"labels":[],"label_agreement":null},{"id":"W2063358860","doi":"10.5539/cis.v3n4p187","title":"Hybrid learning of Syntactic and Semantic Dependencies","year":2010,"lang":"en","type":"article","venue":"Computer and Information Science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Parsing; Mutual information; Artificial intelligence; Natural language processing; Dependency grammar; Principle of maximum entropy; Feature selection; Semantic role labeling; Classifier (UML); Dependency (UML); Entropy (arrow of time); Semantic feature; Machine learning; Sentence","score_opus":0.006088217272645731,"score_gpt":0.24055970210185773,"score_spread":0.234471484829212,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2063358860","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021158272,0.00022981809,0.9709171,0.00020308046,0.000036242407,0.00006434946,0.00028770475,0.0055246684,0.0015787059],"genre_scores_gemma":[0.2883159,0.00024219924,0.70240664,0.0003254572,0.00008356364,0.0001945234,0.0018193403,0.00074424426,0.0058680903],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985057,0.00043912837,0.000088150155,0.0004905124,0.0003711193,0.00010533816],"domain_scores_gemma":[0.9979665,0.0011357805,0.00010666909,0.00037293785,0.00035767155,0.000060468443],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020481776,0.0012042192,0.0012748344,0.0017863059,0.0007577113,0.0011953043,0.0016740185,0.001571487,0.0033191137],"category_scores_gemma":[0.004335403,0.0008162709,0.0012902755,0.0016135651,0.00049323204,0.0037120888,0.0017064404,0.00183932,0.001611257],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003431313,0.00044071669,0.0035090162,0.00022303505,0.00029047034,0.00023943736,0.00025848177,0.05928527,0.029652199,0.009307989,0.0074382536,0.8890119],"study_design_scores_gemma":[0.00002663566,0.00011637152,0.0018925024,0.000024661509,0.00011523599,0.00017697572,0.00007449425,0.95077384,0.020692838,0.020276237,0.005782261,0.000047964815],"about_ca_topic_score_codex":0.0029652398,"about_ca_topic_score_gemma":0.0062922123,"teacher_disagreement_score":0.0033191137,"about_ca_system_score_codex":0.0005705876,"about_ca_system_score_gemma":0.0014884206,"threshold_uncertainty_score":0.01110357},"labels":[],"label_agreement":null},{"id":"W2063381274","doi":"10.1007/s10590-006-9017-3","title":"EBMT by tree-phrasing","year":2006,"lang":"en","type":"article","venue":"Machine Translation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"Universität Stuttgart","keywords":"Computer science; Machine translation; Natural language processing; Exploit; Phrase; Artificial intelligence; Example-based machine translation; Machine translation software usability; Translation (biology); Dependency (UML); Tree (set theory); Simple (philosophy); Computational linguistics; Transfer-based machine translation","score_opus":0.0072064830078437106,"score_gpt":0.2440312977121587,"score_spread":0.236824814704315,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2063381274","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006968914,0.00056995585,0.9352546,0.0010362453,0.0008720128,0.00020004506,0.0012564244,0.009676233,0.044165626],"genre_scores_gemma":[0.2386493,0.00089413236,0.71542406,0.0011341742,0.000496116,0.00030158556,0.0031254946,0.0053713038,0.034603868],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9970066,0.0014839896,0.0002632462,0.00048382106,0.0005977761,0.0001645238],"domain_scores_gemma":[0.9954853,0.0017069409,0.00013059935,0.0016640836,0.00095386966,0.00005918067],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017082561,0.00081687,0.0008194618,0.0014363286,0.0012441226,0.0023896361,0.0011233301,0.0016096283,0.0366308],"category_scores_gemma":[0.0104101505,0.00068944495,0.00077844155,0.0025647993,0.0009700379,0.004203098,0.0025987928,0.0016343071,0.019554993],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002367796,0.0001300301,0.0003553996,0.00090869743,0.00008635313,0.00072125555,0.0007983934,0.007300141,0.016712109,0.2378679,0.06973868,0.66514426],"study_design_scores_gemma":[0.00011553734,0.00014799339,0.0005345639,0.00031326842,0.00020296231,0.0018796095,0.00032940108,0.14495915,0.057566646,0.561829,0.23202164,0.00010021472],"about_ca_topic_score_codex":0.0011277238,"about_ca_topic_score_gemma":0.0013488613,"teacher_disagreement_score":0.0366308,"about_ca_system_score_codex":0.00054632383,"about_ca_system_score_gemma":0.001150266,"threshold_uncertainty_score":0.12254226},"labels":[],"label_agreement":null},{"id":"W2063643557","doi":"10.1145/2396761.2398468","title":"Efficient extraction of ontologies from domain specific text corpora","year":2012,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Information retrieval; Domain (mathematical analysis); Ontology; Information extraction; Natural language processing; World Wide Web; Mathematics","score_opus":0.023055874418413017,"score_gpt":0.27734872333576666,"score_spread":0.25429284891735365,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2063643557","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04086347,0.0055751675,0.8340209,0.0037616005,0.00080128666,0.0014523589,0.063425235,0.028886022,0.02121395],"genre_scores_gemma":[0.06756464,0.0038248186,0.7959552,0.00040351218,0.00026243695,0.0007184537,0.12539002,0.0017657023,0.004115345],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9973838,0.00074252655,0.00039968765,0.0005105342,0.0008154732,0.00014787655],"domain_scores_gemma":[0.9894924,0.0057402146,0.0008825933,0.0014462144,0.0022448776,0.00019373427],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002261026,0.0017802556,0.0016871507,0.01351589,0.0017940814,0.0033077223,0.0016271926,0.0015902756,0.004738273],"category_scores_gemma":[0.016390044,0.0011394911,0.0017080642,0.012346858,0.00067465554,0.0052593052,0.0023696905,0.0019699156,0.006855228],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019180174,0.00038693135,0.004192882,0.005095888,0.00039017035,0.0019215316,0.0015408527,0.008990377,0.04785442,0.029302973,0.13375185,0.76638037],"study_design_scores_gemma":[0.00022348619,0.00016828408,0.015029124,0.0015909654,0.00069993,0.0029771149,0.004101219,0.20284641,0.085359156,0.12031496,0.5664003,0.0002891241],"about_ca_topic_score_codex":0.003805302,"about_ca_topic_score_gemma":0.010764061,"teacher_disagreement_score":0.01351589,"about_ca_system_score_codex":0.0012163833,"about_ca_system_score_gemma":0.0041062073,"threshold_uncertainty_score":0.01585108},"labels":[],"label_agreement":null},{"id":"W2064325869","doi":"10.3166/isi.10.2.69-89","title":"Les logiques de description pour le tri sémantique de documents sur le web","year":2005,"lang":"fr","type":"article","venue":"Ingénierie des systèmes d information","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Information retrieval; Web search query; Rank (graph theory); Web query classification; Search engine; Mathematics; Combinatorics","score_opus":0.025745898060258652,"score_gpt":0.2649622357291033,"score_spread":0.23921633766884468,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2064325869","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006050109,0.001224305,0.98356503,0.0012217506,0.00016389097,0.00021937855,0.0014289005,0.0028917375,0.0032348714],"genre_scores_gemma":[0.08250667,0.00223821,0.9021979,0.000416025,0.00037193217,0.00056289707,0.005000641,0.00085650967,0.0058491556],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9885499,0.0038094255,0.0009783965,0.0012964632,0.004994411,0.0003715439],"domain_scores_gemma":[0.9794,0.012927517,0.0012428657,0.0033091335,0.0026894377,0.00043107997],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006588133,0.002101724,0.0014742876,0.008228568,0.0016022284,0.008775566,0.0020168186,0.0022965707,0.0076606893],"category_scores_gemma":[0.031440724,0.0015948412,0.003147318,0.0060805734,0.0031119063,0.018240891,0.0018320044,0.008394707,0.0046932194],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004696266,0.00036354328,0.0058907545,0.001254792,0.0002176608,0.00059251435,0.0012954242,0.014755651,0.012208686,0.48147544,0.0150532825,0.4664227],"study_design_scores_gemma":[0.00020615602,0.0002599377,0.0056409314,0.0005908218,0.00017725195,0.0020429282,0.00153813,0.42867666,0.024061788,0.36950174,0.1670018,0.00030189037],"about_ca_topic_score_codex":0.011118484,"about_ca_topic_score_gemma":0.009024104,"teacher_disagreement_score":0.011118484,"about_ca_system_score_codex":0.0031962104,"about_ca_system_score_gemma":0.0033654738,"threshold_uncertainty_score":0.034841776},"labels":[],"label_agreement":null},{"id":"W2064641255","doi":"10.1109/isda.2010.5687037","title":"Pronominal anaphora understanding","year":2010,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Anaphora (linguistics); Computer science; Covert; Argument (complex analysis); Natural language processing; Coreference; Parsing; Artificial intelligence; Set (abstract data type); Natural (archaeology); Natural language; Resolution (logic); Natural language understanding; Linguistics; Philosophy; Programming language; History","score_opus":0.022918295802881804,"score_gpt":0.2700256137260089,"score_spread":0.24710731792312712,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2064641255","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.055352747,0.0013301506,0.8572495,0.00549146,0.00009091382,0.000084311825,0.00042277377,0.0014785139,0.078499615],"genre_scores_gemma":[0.774941,0.0010019958,0.21285765,0.0013719902,0.00017240857,0.00009006264,0.0006074804,0.00030601956,0.008651347],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9971214,0.0008127757,0.00024023162,0.0006605211,0.0009017062,0.0002632848],"domain_scores_gemma":[0.9864037,0.005916844,0.000996296,0.0052966946,0.0011685988,0.00021785947],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0051507875,0.0005462044,0.00080815173,0.001780876,0.0016017831,0.00634543,0.0024908183,0.0020000935,0.0073916214],"category_scores_gemma":[0.017483195,0.0008657067,0.0010152879,0.0014396008,0.0045044483,0.020573052,0.006559158,0.004586883,0.0020795332],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000047453395,0.000056633053,0.002694246,0.00022793017,0.00003440689,0.0001333491,0.0022765864,0.0032593755,0.008567412,0.9062919,0.0024767814,0.073934026],"study_design_scores_gemma":[0.0000064101564,0.0000114370205,0.00095557974,0.000055045693,0.00002913391,0.00026181786,0.00032104028,0.02169982,0.0073483908,0.95437187,0.014916241,0.000023170185],"about_ca_topic_score_codex":0.0013294618,"about_ca_topic_score_gemma":0.0013052977,"teacher_disagreement_score":0.0073916214,"about_ca_system_score_codex":0.0012454422,"about_ca_system_score_gemma":0.0015067582,"threshold_uncertainty_score":0.027240276},"labels":[],"label_agreement":null},{"id":"W2065236871","doi":"10.1145/1869746.1869766","title":"PlagDetect","year":2010,"lang":"en","type":"article","venue":"ACM Inroads","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Lakehead University","funders":"","keywords":"Computer science; Plagiarism detection; Java; Context (archaeology); Programming language; Measure (data warehouse); Software engineering; Artificial intelligence; Data mining","score_opus":0.0092614352417447,"score_gpt":0.2693851248016546,"score_spread":0.2601236895599099,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2065236871","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.056966815,0.0023869642,0.10416239,0.0045413217,0.0031210706,0.0010489265,0.0391826,0.1317619,0.656828],"genre_scores_gemma":[0.2215591,0.0018986891,0.07385024,0.0017853823,0.0007458664,0.0010394268,0.08090098,0.018721875,0.5994984],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9977962,0.00042371906,0.00013432243,0.0003673833,0.0010609654,0.00021742965],"domain_scores_gemma":[0.99262017,0.0014780865,0.00052276655,0.0023173932,0.0018489944,0.0012126169],"candidate_categories":["research_integrity"],"consensus_categories":[],"category_scores_codex":[0.0023982967,0.0007058453,0.0005849757,0.0033981686,0.0013761701,0.0048554335,0.0016622277,0.0009391668,0.20558788],"category_scores_gemma":[0.011092652,0.0005275287,0.00043670996,0.0026227003,0.00066021906,0.0035268923,0.005021445,0.0015798722,0.119991735],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00060372596,0.00023638226,0.0073083914,0.0007309709,0.000030632495,0.00030905847,0.00067263655,0.0007920155,0.00571906,0.021176228,0.44880623,0.51361465],"study_design_scores_gemma":[0.00006684416,0.00012369046,0.006075689,0.000104456245,0.000011303825,0.00050066033,0.00016701328,0.002320141,0.005086433,0.005304013,0.9802096,0.000030233474],"about_ca_topic_score_codex":0.0012604839,"about_ca_topic_score_gemma":0.0019778984,"teacher_disagreement_score":0.9990608,"about_ca_system_score_codex":0.001342959,"about_ca_system_score_gemma":0.0018904067,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W2066133570","doi":"10.7202/004514ar","title":"A Formal Language to Convey Linguistic Information. A Study in Practical Logic","year":2002,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Linguistics; Rule-based machine translation; Machine translation; Artificial intelligence; Natural language processing; Philosophy","score_opus":0.05097625907166796,"score_gpt":0.32680762969017624,"score_spread":0.27583137061850826,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2066133570","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.003726614,0.015180331,0.8816494,0.031341378,0.0027525513,0.00012128193,0.00059876835,0.0013582173,0.06327156],"genre_scores_gemma":[0.28209326,0.011664372,0.6418567,0.009527286,0.0026444814,0.00058590336,0.0009896418,0.00078103284,0.04985733],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99786997,0.0011959971,0.0002179338,0.00023712462,0.00037691835,0.00010195481],"domain_scores_gemma":[0.99310887,0.005254056,0.0003667635,0.0006121193,0.00049898203,0.0001591624],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0038461941,0.0010325182,0.0006258669,0.003045328,0.002083778,0.0062411292,0.0017828039,0.0023077761,0.0072722244],"category_scores_gemma":[0.011423992,0.00066433015,0.0012248305,0.002615364,0.012451705,0.014911728,0.0024652435,0.0041304026,0.0023383584],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000065137206,0.000005489163,0.00002642355,0.00007329236,0.0000038029357,0.00003553765,0.0004467986,0.0002535605,0.000225044,0.98645616,0.0047239875,0.0077434243],"study_design_scores_gemma":[0.00001598296,0.000014478067,0.000030324545,0.00016155925,0.000010680191,0.0001976992,0.00023022103,0.0025174182,0.0004925421,0.9026737,0.09364272,0.000012661795],"about_ca_topic_score_codex":0.0016535626,"about_ca_topic_score_gemma":0.0011909469,"teacher_disagreement_score":0.0072722244,"about_ca_system_score_codex":0.0023867767,"about_ca_system_score_gemma":0.0018011086,"threshold_uncertainty_score":0.024327993},"labels":[],"label_agreement":null},{"id":"W2066247257","doi":"10.1515/cog-2013-0008","title":"Extracting prototypes from exemplars What can corpus data tell us about concept representation?","year":2013,"lang":"en","type":"article","venue":"Cognitive Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":99,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Abstraction; Categorization; Computer science; Natural language processing; Representation (politics); Cognitive linguistics; Artificial intelligence; Basis (linear algebra); Computational linguistics; Cognition; Cluster analysis; Linguistics; Psychology; Mathematics; Epistemology","score_opus":0.04349772500190396,"score_gpt":0.33517668789813115,"score_spread":0.2916789628962272,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2066247257","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7336144,0.0020603004,0.24187852,0.003717071,0.00021294906,0.00019899108,0.004516592,0.00096036907,0.012840831],"genre_scores_gemma":[0.9058875,0.00051214104,0.08937068,0.00020232663,0.00006675798,0.00022313716,0.0031709936,0.00013981009,0.00042669202],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9942111,0.0035811276,0.00039353035,0.00097107026,0.00071201974,0.0001311978],"domain_scores_gemma":[0.93169713,0.046067685,0.003022975,0.0149816265,0.003749593,0.00048090753],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00939535,0.0005046137,0.000845805,0.004242306,0.0010045082,0.0052621826,0.0018115832,0.001739852,0.0040267077],"category_scores_gemma":[0.09120176,0.00063958997,0.00079200644,0.004631008,0.0033167738,0.009459942,0.0024603012,0.0017945408,0.0009879571],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017932229,0.0005305392,0.21334134,0.0026602987,0.0011797814,0.0010987504,0.014600797,0.043711025,0.023372272,0.122782804,0.015872862,0.55905634],"study_design_scores_gemma":[0.00017300101,0.00042787462,0.0984536,0.0013490579,0.00030205766,0.0012862959,0.012593714,0.24520966,0.016147006,0.5804914,0.04313475,0.0004315123],"about_ca_topic_score_codex":0.0013916417,"about_ca_topic_score_gemma":0.0013162161,"teacher_disagreement_score":0.00939535,"about_ca_system_score_codex":0.0007733311,"about_ca_system_score_gemma":0.00063210435,"threshold_uncertainty_score":0.049687922},"labels":[],"label_agreement":null},{"id":"W2066553323","doi":"10.3115/1631850.1631855","title":"Automatically distinguishing literal and figurative usages of highly polysemous verbs","year":2005,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Principle of compositionality; Literal and figurative language; Computer science; Natural language processing; Semantics (computer science); Meaning (existential); Literal (mathematical logic); Distributional semantics; Artificial intelligence; Verb; Property (philosophy); Linguistics; Polysemy; Programming language; Psychology","score_opus":0.00932501758208123,"score_gpt":0.2733698065967666,"score_spread":0.26404478901468537,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2066553323","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9049428,0.00020067192,0.089276806,0.00013602804,0.000022718155,0.00008036692,0.00045228063,0.00086461764,0.004023594],"genre_scores_gemma":[0.9439922,0.0000847671,0.054068856,0.000026420515,0.000017146638,0.000040882438,0.0010359095,0.00022008312,0.0005138373],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986798,0.0004412407,0.00011369072,0.00038365062,0.0002924335,0.000089185916],"domain_scores_gemma":[0.9874232,0.008722924,0.0016470404,0.0009372117,0.0010279512,0.00024169113],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001575038,0.0004236361,0.00047537574,0.0024353177,0.00062572054,0.002020127,0.0006265147,0.0007295746,0.0014617057],"category_scores_gemma":[0.012837628,0.0003816012,0.0003830719,0.0013083454,0.0007734343,0.0025225445,0.0010868049,0.0008618372,0.00061086414],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010327167,0.00024143783,0.1581482,0.00087000296,0.00012098681,0.0008993134,0.009640862,0.0028811474,0.2713674,0.020669108,0.0027172768,0.5314116],"study_design_scores_gemma":[0.000157496,0.00054072717,0.4243361,0.00022716029,0.0002937792,0.005041749,0.009127413,0.31604576,0.13680464,0.0814656,0.025697215,0.00026233555],"about_ca_topic_score_codex":0.0008328342,"about_ca_topic_score_gemma":0.0024598003,"teacher_disagreement_score":0.0024353177,"about_ca_system_score_codex":0.00030253697,"about_ca_system_score_gemma":0.0003707722,"threshold_uncertainty_score":0.0083296895},"labels":[],"label_agreement":null},{"id":"W2067142167","doi":"10.3115/1614101.1614106","title":"Inductive semi-supervised learning methods for natural language processing","year":2006,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Temporal annotation; Annotation; Artificial intelligence; Natural language processing; Raw data; Semi-supervised learning; Natural language; Supervised learning; Computational linguistics; Face (sociological concept); Machine learning; Language technology; Linguistics; Artificial neural network; Comprehension approach","score_opus":0.01385015109593683,"score_gpt":0.3432689333605641,"score_spread":0.32941878226462723,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2067142167","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00022803037,0.0004980732,0.9976781,0.00014619686,0.00004474752,0.00007510054,0.00013595114,0.00049059984,0.0007031404],"genre_scores_gemma":[0.030904997,0.0018541564,0.95889443,0.00036438325,0.00050107413,0.0013528154,0.0022493338,0.00037812814,0.0035006339],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9857738,0.007287911,0.000980688,0.0015824906,0.004162993,0.00021213156],"domain_scores_gemma":[0.9727864,0.020311406,0.0012038272,0.0031482212,0.0023006576,0.0002494899],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008831321,0.0014521664,0.0016721976,0.0031226566,0.0012942977,0.002695562,0.0046344097,0.0017359417,0.0060883043],"category_scores_gemma":[0.02338689,0.00094979565,0.002136883,0.0031660586,0.003109221,0.003862652,0.0042178384,0.0043282458,0.0052516446],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000107672204,0.00023908383,0.0010503524,0.0023614974,0.0004096658,0.00035293246,0.00067970634,0.074919544,0.003957535,0.32717368,0.029932637,0.55881566],"study_design_scores_gemma":[0.00003791207,0.00005166116,0.0003533741,0.00020873884,0.000043286782,0.00026328088,0.00008469751,0.31515574,0.0032426738,0.64502054,0.035478216,0.000059808783],"about_ca_topic_score_codex":0.0008433665,"about_ca_topic_score_gemma":0.0016589052,"teacher_disagreement_score":0.008831321,"about_ca_system_score_codex":0.001257583,"about_ca_system_score_gemma":0.0024490594,"threshold_uncertainty_score":0.046705067},"labels":[],"label_agreement":null},{"id":"W2067366489","doi":"10.7202/1027474ar","title":"Evidence of Parallel Processing During Translation","year":2014,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":56,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Source text; Reading (process); Natural language processing; Literal translation; Eye tracking; Congruence (geometry); Machine translation; Artificial intelligence; Linguistics; Translation (biology); Danish; Target text; Example-based machine translation; Psychology","score_opus":0.05663297699258963,"score_gpt":0.2992657555134165,"score_spread":0.24263277852082688,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2067366489","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9740097,0.000303545,0.013912778,0.0002284209,0.00003611766,0.00013257509,0.00020789036,0.00019470876,0.010974283],"genre_scores_gemma":[0.9901759,0.00012782897,0.008074903,0.00010785313,0.00001775155,0.00013874008,0.00019554203,0.000083507046,0.0010778875],"study_design_codex":"bench_or_experimental","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9961803,0.0011808212,0.00030798727,0.0014752038,0.00065110996,0.00020461957],"domain_scores_gemma":[0.9681425,0.020302523,0.0035057073,0.0054805465,0.0020641563,0.0005045342],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034942515,0.0004962232,0.00069770124,0.0006326275,0.0006100644,0.0018049062,0.0006208494,0.0011183013,0.0064601437],"category_scores_gemma":[0.04199081,0.0010006365,0.00054732827,0.0005851809,0.0016129621,0.0038338604,0.0015965474,0.0012594855,0.0012739616],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.009504638,0.0016267606,0.08587511,0.002236335,0.0005992922,0.0019246212,0.030976104,0.0023973926,0.64065915,0.01223975,0.0016479098,0.21031287],"study_design_scores_gemma":[0.0015549955,0.0041894726,0.73194003,0.00025378104,0.0005948113,0.004164983,0.0040611667,0.01957839,0.17538796,0.050154686,0.007821791,0.00029800995],"about_ca_topic_score_codex":0.0016876768,"about_ca_topic_score_gemma":0.0016182924,"teacher_disagreement_score":0.0064601437,"about_ca_system_score_codex":0.00051913416,"about_ca_system_score_gemma":0.0006947546,"threshold_uncertainty_score":0.021611273},"labels":[],"label_agreement":null},{"id":"W2067372181","doi":"10.1145/1236181.1236182","title":"Introduction to special issue on reasoning in natural language information processing","year":2006,"lang":"en","type":"article","venue":"ACM Transactions on Asian Language Information Processing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"Engineering and Physical Sciences Research Council","keywords":"Computer science; Artificial intelligence; Natural language processing; Question answering; Perspective (graphical); Heuristic; Natural language; Reasoning system; Analytic reasoning; Opportunistic reasoning; Model-based reasoning; Natural language understanding; Natural (archaeology); Knowledge representation and reasoning","score_opus":0.003288075048748674,"score_gpt":0.24309328166532185,"score_spread":0.23980520661657317,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2067372181","genre_codex":"editorial","genre_gemma":"editorial","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"editorial","genre_consensus":"editorial","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0009117304,0.0810715,0.03955011,0.044743717,0.70071304,0.00038138885,0.0012027854,0.001061981,0.13036375],"genre_scores_gemma":[0.0053715967,0.060242906,0.012450931,0.0190413,0.7206356,0.0002912904,0.0021589992,0.0011241486,0.17868324],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99794215,0.00038562785,0.00023507538,0.0004967509,0.0007853514,0.00015497106],"domain_scores_gemma":[0.9925211,0.0033899082,0.00031578427,0.0007336051,0.0021797905,0.0008598457],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023724998,0.0017146016,0.0024695024,0.0037841666,0.00196325,0.0064348155,0.001983743,0.003422572,0.09189083],"category_scores_gemma":[0.007285249,0.0008302227,0.0023037207,0.0036356605,0.0016441185,0.008157807,0.002613167,0.0075060646,0.04324466],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000026708034,0.000062602994,0.00015510326,0.0004407278,0.00003503314,0.00009898211,0.00006813504,0.00017661227,0.0003563995,0.010173377,0.9300941,0.058312178],"study_design_scores_gemma":[0.000009563313,0.00003391718,0.0003723381,0.0002063692,0.000027135762,0.00028864457,0.00004764814,0.0004398241,0.00013503236,0.011276842,0.9871486,0.000014101052],"about_ca_topic_score_codex":0.0008782181,"about_ca_topic_score_gemma":0.0014402714,"teacher_disagreement_score":0.09189083,"about_ca_system_score_codex":0.0016243398,"about_ca_system_score_gemma":0.0015043812,"threshold_uncertainty_score":0.3074054},"labels":[],"label_agreement":null},{"id":"W2067443652","doi":"10.1075/cf.3.2.03dan","title":"Modification and constructional blends in the use of proper names","year":2011,"lang":"en","type":"article","venue":"Constructions and Frames","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Principle of compositionality; Metonymy; Proper noun; Noun; Linguistics; Context (archaeology); Meaning (existential); Frame (networking); Computer science; Range (aeronautics); Natural language processing; History; Philosophy; Epistemology; Engineering","score_opus":0.06331763105879636,"score_gpt":0.25903190454943037,"score_spread":0.19571427349063403,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2067443652","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5560143,0.0018848875,0.36175525,0.0010846627,0.00010953149,0.000105304265,0.00010387356,0.00056313974,0.078379035],"genre_scores_gemma":[0.9680863,0.00036313152,0.028396733,0.00005217201,0.000028224544,0.000039241542,0.000045103123,0.00015837299,0.0028306104],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"qualitative","domain_scores_codex":[0.9949221,0.0023904978,0.00034676265,0.0010092225,0.0010189234,0.0003124464],"domain_scores_gemma":[0.99568295,0.002251431,0.0004566293,0.0013251686,0.00020402358,0.000079761114],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032188764,0.00056076713,0.0006037045,0.0015301032,0.002458551,0.0041360003,0.00095319655,0.001253489,0.0025091113],"category_scores_gemma":[0.007300732,0.0009150617,0.0006651319,0.001954117,0.010539996,0.01525322,0.0065603945,0.001866887,0.0003054584],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020334563,0.000040695388,0.0041433284,0.00014070411,0.000030451694,0.001503125,0.061572053,0.0008066641,0.019122703,0.8627467,0.00031329412,0.04937693],"study_design_scores_gemma":[0.00005156562,0.00016941207,0.010677723,0.00021913578,0.00015254061,0.0066576917,0.028586846,0.013940528,0.04025097,0.7983656,0.10075154,0.00017634877],"about_ca_topic_score_codex":0.00094387284,"about_ca_topic_score_gemma":0.001253508,"teacher_disagreement_score":0.0041360003,"about_ca_system_score_codex":0.0011136634,"about_ca_system_score_gemma":0.0004918081,"threshold_uncertainty_score":0.017023265},"labels":[],"label_agreement":null},{"id":"W2067460167","doi":"10.1109/fuzz-ieee.2012.6251266","title":"Feature-based similarity assessment in ontology using fuzzy set theory","year":2012,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Information retrieval; Semantic similarity; Ontology; Ontology-based data integration; Upper ontology; Fuzzy set; Ontology Inference Layer; Semantic Web; Ontology alignment; Data mining; Similarity (geometry); RDF; Fuzzy logic; OWL-S; Artificial intelligence; Social Semantic Web","score_opus":0.03293164912137945,"score_gpt":0.3481146563689092,"score_spread":0.31518300724752973,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2067460167","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03822536,0.00014331557,0.9599528,0.000073451,0.000020220228,0.00008922332,0.000045417917,0.00010109497,0.0013491972],"genre_scores_gemma":[0.48609903,0.00015712499,0.51294893,0.000030197814,0.000029226134,0.00015645959,0.000119119475,0.000019683572,0.0004403054],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99682164,0.0008192076,0.00030398916,0.00035709023,0.0016035849,0.000094433155],"domain_scores_gemma":[0.99705684,0.0015339821,0.00030200038,0.00021066424,0.0008185573,0.000077892786],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031738668,0.00045164913,0.0009613691,0.0045090667,0.00088404637,0.0021925243,0.0010437409,0.00083353533,0.000925228],"category_scores_gemma":[0.008413729,0.00021638101,0.0010979825,0.0022007686,0.0010565831,0.002962878,0.0010815823,0.0006840612,0.00014980727],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00054514804,0.0004287678,0.007339939,0.00067351275,0.0003987475,0.0005093987,0.0018654154,0.23917115,0.04424368,0.15572184,0.0018428687,0.54725957],"study_design_scores_gemma":[0.000038994753,0.0001692915,0.0027589982,0.000059853886,0.000085239975,0.00022201268,0.00033849452,0.9061384,0.0090636555,0.07924228,0.0018010994,0.0000816338],"about_ca_topic_score_codex":0.0025393835,"about_ca_topic_score_gemma":0.0018118566,"teacher_disagreement_score":0.0045090667,"about_ca_system_score_codex":0.0014992837,"about_ca_system_score_gemma":0.0008982705,"threshold_uncertainty_score":0.016785264},"labels":[],"label_agreement":null},{"id":"W2068202886","doi":"10.1207/s15327817la0904_02","title":"VP-Ellipsis and Anaphora in Child Language Acquisition","year":2001,"lang":"en","type":"article","venue":"Language Acquisition","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":29,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Max-Planck-Gesellschaft","keywords":"Ellipsis (linguistics); Anaphora (linguistics); Learnability; Linguistics; Computer science; Phrase; Grammar; Verb phrase ellipsis; Natural language processing; Artificial intelligence; Verb; Philosophy; Resolution (logic)","score_opus":0.004717045030404344,"score_gpt":0.24863937709332756,"score_spread":0.24392233206292321,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2068202886","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98063606,0.0013433655,0.002532878,0.0004949465,0.000019115318,0.000027823287,0.00008094757,0.00006904532,0.014795846],"genre_scores_gemma":[0.995535,0.0008789419,0.0019870747,0.00010426292,0.0000074027985,0.000044108896,0.00008365944,0.000025231193,0.001334433],"study_design_codex":"bench_or_experimental","study_design_gemma":"observational","domain_scores_codex":[0.9976688,0.00078795553,0.0001305989,0.0004655715,0.00074949703,0.00019752105],"domain_scores_gemma":[0.9895923,0.0073551326,0.001487579,0.00065298664,0.0005263401,0.00038561743],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030725624,0.0005449121,0.0006915066,0.00075078104,0.0007285043,0.0024416428,0.0006567076,0.0015632466,0.0043741097],"category_scores_gemma":[0.015651684,0.00091468997,0.00022197359,0.00058731035,0.0036702624,0.004921553,0.0017969265,0.0020803052,0.0005663789],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018030093,0.0021406405,0.27570653,0.002497891,0.00018130307,0.008889228,0.14552122,0.0018331693,0.28191,0.101000845,0.003043752,0.17547238],"study_design_scores_gemma":[0.00036705646,0.0018432061,0.718547,0.0006874037,0.00024802622,0.01164112,0.031673465,0.00450508,0.09917456,0.0945852,0.036347516,0.00038030487],"about_ca_topic_score_codex":0.004323397,"about_ca_topic_score_gemma":0.004419889,"teacher_disagreement_score":0.0043741097,"about_ca_system_score_codex":0.0010012581,"about_ca_system_score_gemma":0.0009978215,"threshold_uncertainty_score":0.016249418},"labels":[],"label_agreement":null},{"id":"W2068400772","doi":"10.1002/meet.14504201140","title":"Multilingual digital libraries: Research and practice","year":2005,"lang":"en","type":"article","venue":"Proceedings of the American Society for Information Science and Technology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Digital library; Computer science; World Wide Web; Session (web analytics); Digital collections; State (computer science); Range (aeronautics); Open source; Multimedia; Library science; Linguistics; Engineering","score_opus":0.024323419816688455,"score_gpt":0.34207842935962535,"score_spread":0.3177550095429369,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2068400772","genre_codex":"other","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2281228,0.084102966,0.05708246,0.2873347,0.0011936869,0.0010457775,0.00028339057,0.0017200371,0.33911416],"genre_scores_gemma":[0.9366629,0.024515085,0.022798473,0.0071703643,0.0002752582,0.0003357797,0.0001248183,0.0002196187,0.007897789],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.92605895,0.053219315,0.0034394644,0.0047797156,0.009741302,0.0027611554],"domain_scores_gemma":[0.83916414,0.090159625,0.007774552,0.011293985,0.031600643,0.020007119],"candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.08571574,0.00037833356,0.0007378535,0.0049337302,0.0061774817,0.028417174,0.0047328165,0.004359939,0.01429188],"category_scores_gemma":[0.12723114,0.0007000717,0.0004237306,0.007888304,0.013721972,0.023915274,0.016516624,0.004020864,0.0028189754],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00030114755,0.0018155036,0.021753095,0.0041214526,0.00011713402,0.0008823558,0.081981726,0.0020691066,0.0012346613,0.19738108,0.03515702,0.6531857],"study_design_scores_gemma":[0.00028562106,0.0005800634,0.0142809935,0.013951446,0.00011443702,0.0010709176,0.30618098,0.0048814774,0.0032553053,0.1148277,0.5403706,0.00020047503],"about_ca_topic_score_codex":0.0070145153,"about_ca_topic_score_gemma":0.006613811,"teacher_disagreement_score":0.97158283,"about_ca_system_score_codex":0.0150445085,"about_ca_system_score_gemma":0.030084727,"threshold_uncertainty_score":0.4533136},"labels":[],"label_agreement":null},{"id":"W2068554959","doi":"10.7202/003602ar","title":"Towards a Methodology for the Translation of Japanese Patent Claims","year":2002,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Linguistics; Translation (biology); Style (visual arts); Vocabulary; Object (grammar); Computer science; Natural language processing; Philosophy; Literature; Art","score_opus":0.22679619997255535,"score_gpt":0.3380334926959413,"score_spread":0.11123729272338595,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2068554959","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0028734608,0.00017367395,0.9924263,0.00022649625,0.00010951724,0.00024279767,0.00019302065,0.0016549366,0.0020998132],"genre_scores_gemma":[0.01548349,0.00018834736,0.98122543,0.00006783474,0.0000484372,0.00026877818,0.00052151753,0.00034916532,0.0018469109],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99455607,0.0024348204,0.0008668505,0.0010198398,0.00095069944,0.00017169265],"domain_scores_gemma":[0.99238,0.0025787437,0.00088467903,0.0013425318,0.00264118,0.0001728529],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005472004,0.0011554918,0.00092936814,0.0046513425,0.00231584,0.005601451,0.001559803,0.0015628098,0.008496319],"category_scores_gemma":[0.015916336,0.001303006,0.0018130402,0.003827186,0.0020511672,0.0036706983,0.0030281632,0.0023381093,0.0059565497],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014419634,0.00019282929,0.0019006553,0.0016773386,0.00016500075,0.0010996285,0.008308954,0.007535843,0.037570745,0.20594749,0.013602743,0.72185457],"study_design_scores_gemma":[0.00021342533,0.00051059155,0.0042860303,0.0006526743,0.0004424676,0.00277537,0.006724285,0.23171031,0.06984066,0.3128217,0.3696048,0.00041779355],"about_ca_topic_score_codex":0.003447144,"about_ca_topic_score_gemma":0.004226491,"teacher_disagreement_score":0.008496319,"about_ca_system_score_codex":0.0015017828,"about_ca_system_score_gemma":0.0042804657,"threshold_uncertainty_score":0.028939068},"labels":[],"label_agreement":null},{"id":"W2068868583","doi":"10.1145/2559206.2578862","title":"Mining online software tutorials","year":2014,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Workflow; Task (project management); Software; World Wide Web; Ask price; Automation; Interface (matter); Software engineering; Human–computer interaction; Information retrieval; Data science; Programming language; Database","score_opus":0.014996474080507577,"score_gpt":0.27986393422878003,"score_spread":0.26486746014827245,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2068868583","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6074344,0.0114685735,0.20879245,0.0023727112,0.00047742238,0.0012305167,0.11023028,0.0137952445,0.0441984],"genre_scores_gemma":[0.62642217,0.0042639435,0.15025541,0.00033824611,0.00040793148,0.0009259546,0.19387922,0.0013899828,0.022117041],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9973099,0.0006836708,0.00023039377,0.00055907504,0.0009855747,0.00023136433],"domain_scores_gemma":[0.9845536,0.0075641433,0.0014474061,0.0008486163,0.004780462,0.000805838],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014482057,0.0011966405,0.0006932691,0.0136648165,0.0009226258,0.0020381138,0.0013453985,0.001295325,0.0077387006],"category_scores_gemma":[0.021589443,0.0005137537,0.0008786815,0.006333025,0.00041386965,0.0050682463,0.0015634655,0.001288123,0.0054762717],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007049648,0.0010348796,0.06787865,0.003391433,0.0003023417,0.0031121003,0.0033501796,0.013946304,0.010721321,0.015727425,0.13763423,0.74219614],"study_design_scores_gemma":[0.00016939103,0.00084942713,0.13566606,0.0018347863,0.0006322197,0.003894517,0.00751461,0.31651253,0.028882617,0.03895197,0.46479815,0.00029373117],"about_ca_topic_score_codex":0.004994255,"about_ca_topic_score_gemma":0.0086253155,"teacher_disagreement_score":0.0136648165,"about_ca_system_score_codex":0.0011899926,"about_ca_system_score_gemma":0.0014503814,"threshold_uncertainty_score":0.025888562},"labels":[],"label_agreement":null},{"id":"W2069565131","doi":"10.4018/ijcallt.2014100106","title":"Constructing a Data-Driven Learning Tool with Recycled Learner Data","year":2014,"lang":"en","type":"article","venue":"International Journal of Computer-Assisted Language Learning and Teaching","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria; Simon Fraser University","funders":"","keywords":"TUTOR; ADDIE Model; Computer science; Variety (cybernetics); German; Process (computing); Instructional design; Multimedia; Data collection; Artificial intelligence; Programming language; Linguistics","score_opus":0.017128277174990304,"score_gpt":0.30474059054243186,"score_spread":0.28761231336744153,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2069565131","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0205898,0.000082400045,0.8875327,0.0004287873,0.0001087411,0.0011989793,0.0048358585,0.08394023,0.0012823968],"genre_scores_gemma":[0.07087981,0.00007101,0.91070634,0.00023982656,0.000038843085,0.0019544088,0.008850557,0.0050166813,0.0022425419],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9881164,0.0035192005,0.0013945075,0.003678785,0.0029843776,0.00030667492],"domain_scores_gemma":[0.9255764,0.05408898,0.0021656824,0.010215038,0.0066078217,0.0013460718],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.019793317,0.0019477061,0.0018683402,0.007297439,0.0014973361,0.0071795937,0.0047011157,0.00222827,0.003988325],"category_scores_gemma":[0.07098529,0.0019631549,0.00206218,0.003298443,0.0017722221,0.008000864,0.0068588653,0.0035267097,0.0032333604],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017711129,0.0022618114,0.02760149,0.0025604565,0.00064606,0.0028785185,0.017152525,0.047523245,0.035950672,0.019627219,0.03549431,0.8065326],"study_design_scores_gemma":[0.00058933155,0.0009648018,0.0069605284,0.0006858065,0.00027471382,0.0015090671,0.004252918,0.5877682,0.1511057,0.04760554,0.19761288,0.00067039847],"about_ca_topic_score_codex":0.0015638288,"about_ca_topic_score_gemma":0.002348556,"teacher_disagreement_score":0.019793317,"about_ca_system_score_codex":0.0014748878,"about_ca_system_score_gemma":0.002981173,"threshold_uncertainty_score":0.10467833},"labels":[],"label_agreement":null},{"id":"W2070011962","doi":"10.7202/1001055ar","title":"Vers une théorie du récit automatique","year":2011,"lang":"fr","type":"article","venue":"Cinémas Revue d études cinématographiques","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Humanities; Art","score_opus":0.03022788387466484,"score_gpt":0.2529643888318923,"score_spread":0.22273650495722747,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2070011962","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0057640905,0.00077005883,0.9635184,0.001397713,0.0000931221,0.000121535326,0.00013722279,0.00043020718,0.027767565],"genre_scores_gemma":[0.20637505,0.0026635213,0.7576055,0.0004932418,0.00014442096,0.00055813935,0.0006264029,0.0006850954,0.030848615],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99464756,0.0022822162,0.0002735399,0.001172923,0.0013626273,0.00026112466],"domain_scores_gemma":[0.9917595,0.0050418386,0.00041521867,0.0017667692,0.00086560013,0.00015109002],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005771019,0.0015557706,0.00089230435,0.004177165,0.0019981044,0.009756901,0.0034722164,0.0035383804,0.028708087],"category_scores_gemma":[0.020391887,0.0015973145,0.003246937,0.0031658902,0.008586719,0.019314425,0.0032729108,0.003723265,0.0050985143],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003576926,0.000036161364,0.0004812897,0.00026777104,0.000033351684,0.00015217101,0.0016539744,0.014911731,0.0009382837,0.9511275,0.0013852217,0.028976768],"study_design_scores_gemma":[0.00005293601,0.00007825943,0.00069510884,0.00051553786,0.000072515846,0.0006065999,0.0017362671,0.15272637,0.0058354535,0.7049832,0.13260144,0.00009638898],"about_ca_topic_score_codex":0.008614147,"about_ca_topic_score_gemma":0.005979502,"teacher_disagreement_score":0.028708087,"about_ca_system_score_codex":0.0037429782,"about_ca_system_score_gemma":0.002763293,"threshold_uncertainty_score":0.09603816},"labels":[],"label_agreement":null},{"id":"W2070354752","doi":"10.1075/ijcl.11.2.04cla","title":"Discovering and organizing noun-verb collocations in specialized corpora using inductive logic programming","year":2006,"lang":"en","type":"article","venue":"International Journal of Corpus Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Realization (probability); Natural language processing; Computer science; Noun; Artificial intelligence; Verb; Inductive logic programming; Relevance (law); Meaning (existential); Linguistics; Parsing; Mathematics; Psychology; Philosophy","score_opus":0.02060452257872276,"score_gpt":0.30126745915532455,"score_spread":0.2806629365766018,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2070354752","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06067627,0.00046446256,0.93075734,0.000292106,0.00003204104,0.00052631483,0.0016708298,0.0025123602,0.0030682606],"genre_scores_gemma":[0.10700477,0.00028543753,0.8864659,0.00005797893,0.000026815334,0.00056155486,0.0045327935,0.00032902858,0.0007356994],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99500763,0.0017726908,0.00052985846,0.0015108399,0.0009851452,0.00019384522],"domain_scores_gemma":[0.98570824,0.01050051,0.001267628,0.0014077834,0.0009452682,0.00017057027],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0044154343,0.0012979142,0.0012039939,0.011496893,0.0021523815,0.0035808843,0.002249573,0.000922631,0.0030846407],"category_scores_gemma":[0.017856807,0.0012026608,0.0015702619,0.009586083,0.0018817802,0.0049134935,0.0027926445,0.0015225419,0.0014274293],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036643082,0.00049766153,0.02068061,0.0022174774,0.0003470303,0.0019078843,0.008371279,0.031367745,0.043496225,0.042582653,0.006236067,0.841929],"study_design_scores_gemma":[0.00029071816,0.00034124078,0.026212242,0.00066783983,0.0005401587,0.0024011186,0.01073497,0.59670764,0.060471132,0.20692341,0.0943717,0.00033787233],"about_ca_topic_score_codex":0.004254276,"about_ca_topic_score_gemma":0.008355218,"teacher_disagreement_score":0.011496893,"about_ca_system_score_codex":0.0015253889,"about_ca_system_score_gemma":0.0024337,"threshold_uncertainty_score":0.023351371},"labels":[],"label_agreement":null},{"id":"W2070558146","doi":"10.1006/jath.2002.3693","title":"On the Completeness of the System {Zτ} in L2","year":2002,"lang":"en","type":"article","venue":"Journal of Approximation Theory","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Mathematics; Completeness (order theory); Calculus (dental); Discrete mathematics; Pure mathematics; Algebra over a field; Mathematical analysis","score_opus":0.016668667781482672,"score_gpt":0.22417266242835784,"score_spread":0.20750399464687516,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2070558146","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.37745854,0.0010569706,0.57807875,0.008803285,0.00015311695,0.00020055797,0.0032449733,0.0016801219,0.029323686],"genre_scores_gemma":[0.92223054,0.00042562524,0.06585735,0.0009958264,0.00021035786,0.00023462367,0.002859739,0.00050757127,0.006678468],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99216485,0.0029154874,0.00044356714,0.0012573089,0.0013794317,0.0018393274],"domain_scores_gemma":[0.9374503,0.04654779,0.002445023,0.005234457,0.006171239,0.0021512634],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008500449,0.0011921455,0.0030286126,0.0024534888,0.0053637717,0.0074396385,0.0037701246,0.003178698,0.0084501],"category_scores_gemma":[0.037366673,0.0017432219,0.003038137,0.0022275555,0.0072300597,0.015917405,0.0071109137,0.006513378,0.0010338672],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013253625,0.00013579866,0.0038232207,0.00038819818,0.00017167706,0.00022867593,0.0019990124,0.026175754,0.002330431,0.9462338,0.0048457975,0.012342314],"study_design_scores_gemma":[0.00017197734,0.00008591682,0.0010222062,0.000043114353,0.000069657435,0.00009795731,0.0003532928,0.06589179,0.0017355932,0.9284325,0.0020493297,0.000046631954],"about_ca_topic_score_codex":0.012313381,"about_ca_topic_score_gemma":0.008862798,"teacher_disagreement_score":0.012313381,"about_ca_system_score_codex":0.004162774,"about_ca_system_score_gemma":0.0051261047,"threshold_uncertainty_score":0.044955194},"labels":[],"label_agreement":null},{"id":"W2070720620","doi":"10.4312/ala.4.2.37-51","title":"Construction of a Learner Corpus for Japanese Language Learners: Natane and Nutmeg","year":2014,"lang":"en","type":"article","venue":"Acta Linguistica Asiatica","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Tellabs (Canada)","funders":"Ministry of Education, Culture, Sports, Science and Technology","keywords":"Nutmeg; Computer science; Identification (biology); Register (sociolinguistics); Variety (cybernetics); Collocation (remote sensing); Natural language processing; Reading (process); Active listening; Linguistics; Artificial intelligence; World Wide Web; Psychology; Communication; Biology; Medicine; Traditional medicine","score_opus":0.007812596003544454,"score_gpt":0.2621353032343071,"score_spread":0.25432270723076267,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2070720620","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.80195665,0.000780465,0.07643241,0.00076936994,0.00050356763,0.006377821,0.073194325,0.006366994,0.03361838],"genre_scores_gemma":[0.5076801,0.00051306444,0.25163272,0.0004450254,0.00012087651,0.012528252,0.2016109,0.0032482164,0.022220803],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99760216,0.0008530713,0.00037468615,0.00062391825,0.00041315792,0.00013295645],"domain_scores_gemma":[0.98826116,0.004426135,0.00041719485,0.0021749705,0.0040966207,0.00062396395],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003808579,0.000866731,0.0008459178,0.0037616754,0.0019549709,0.0013900019,0.0011825929,0.0010986913,0.015376889],"category_scores_gemma":[0.011559206,0.0007388963,0.00043101824,0.0022783785,0.0012613408,0.0025076582,0.0048949495,0.0016534801,0.006657595],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021211018,0.0030129778,0.077675864,0.0062214215,0.00016265402,0.0059051523,0.07717669,0.0038945342,0.15348454,0.009025595,0.11658023,0.5447392],"study_design_scores_gemma":[0.0007386831,0.0012301129,0.19983658,0.0009864835,0.000284915,0.005413601,0.04302177,0.0211189,0.08414271,0.0033716266,0.6394783,0.0003763447],"about_ca_topic_score_codex":0.0059309425,"about_ca_topic_score_gemma":0.013773008,"teacher_disagreement_score":0.015376889,"about_ca_system_score_codex":0.00075981324,"about_ca_system_score_gemma":0.0026807166,"threshold_uncertainty_score":0.051440835},"labels":[],"label_agreement":null},{"id":"W2070852132","doi":"10.3115/1225733.1225734","title":"Automatic detection of translation errors","year":2005,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"National Research Council Canada; Université de Montréal","funders":"","keywords":"Computer science; Quality assurance; Machine translation; Translation (biology); Section (typography); Natural language processing; Information retrieval; Library science; Engineering","score_opus":0.013372412111749844,"score_gpt":0.2677342358224453,"score_spread":0.2543618237106955,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2070852132","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13127449,0.0015753828,0.72218484,0.0012594458,0.00079614803,0.00045736815,0.0037541506,0.12514976,0.013548398],"genre_scores_gemma":[0.4122841,0.0004957781,0.5588073,0.0005311228,0.00014524521,0.00017034866,0.007388192,0.0064183758,0.01375955],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99542385,0.0007926063,0.0003942795,0.0011848911,0.001958902,0.00024547757],"domain_scores_gemma":[0.9868924,0.0033708091,0.0015880258,0.0031246832,0.0047891135,0.0002350164],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021020153,0.0012587808,0.0010987625,0.0026227427,0.0010136367,0.0019697477,0.0011349639,0.0014197668,0.009431499],"category_scores_gemma":[0.01690524,0.00046505212,0.0005180217,0.0016913296,0.0006125123,0.0014540574,0.0018257804,0.0012300706,0.00831686],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008076351,0.00013934344,0.00723376,0.00067038945,0.00010021878,0.0016308767,0.0010092874,0.003603052,0.18736959,0.0043791905,0.046425648,0.7466311],"study_design_scores_gemma":[0.00018903197,0.00043793832,0.01310117,0.00022932586,0.00018235703,0.006711363,0.00081102113,0.18730451,0.6772929,0.009737263,0.103776984,0.00022615625],"about_ca_topic_score_codex":0.0021071606,"about_ca_topic_score_gemma":0.0029118669,"teacher_disagreement_score":0.009431499,"about_ca_system_score_codex":0.0004063077,"about_ca_system_score_gemma":0.0011200064,"threshold_uncertainty_score":0.03155154},"labels":[],"label_agreement":null},{"id":"W2071294154","doi":"10.7202/004608ar","title":"Basic Concepts of MT","year":2002,"lang":"fr","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Philosophy","score_opus":0.06451204117336859,"score_gpt":0.3027918405419113,"score_spread":0.23827979936854268,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2071294154","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005591109,0.01404189,0.65336275,0.008586718,0.001975771,0.00057514635,0.0019717005,0.0022148034,0.31168017],"genre_scores_gemma":[0.23842749,0.016781071,0.603352,0.0037605437,0.005004831,0.0022141056,0.0030576899,0.0010671928,0.12633514],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9980445,0.0006144156,0.00027661523,0.0004793542,0.00045698983,0.00012816877],"domain_scores_gemma":[0.998271,0.0007683657,0.00022266839,0.00037355322,0.00027396283,0.00009045266],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016821177,0.0010062179,0.0008757076,0.0044311304,0.0021569203,0.0065092756,0.0024515942,0.0029015073,0.029412769],"category_scores_gemma":[0.0051260605,0.00051247765,0.0011747003,0.004392695,0.00631004,0.010500594,0.0026503874,0.0023483727,0.011922339],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000015745927,0.000009141629,0.00015579238,0.00020315536,0.000012716028,0.00020013603,0.0005334872,0.0006541782,0.00047121453,0.9491303,0.0066391276,0.041975018],"study_design_scores_gemma":[0.000009426657,0.00002040557,0.000213467,0.00016325181,0.000011555434,0.0006843692,0.00016322114,0.0027948916,0.0004063583,0.76495993,0.23055485,0.000018258339],"about_ca_topic_score_codex":0.0021140752,"about_ca_topic_score_gemma":0.0013824372,"teacher_disagreement_score":0.029412769,"about_ca_system_score_codex":0.0017842487,"about_ca_system_score_gemma":0.0014512484,"threshold_uncertainty_score":0.09839553},"labels":[],"label_agreement":null},{"id":"W2071643874","doi":"10.1109/slt.2014.7078552","title":"Incremental translation using hierarchichal phrase-based translation system","year":2014,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Machine translation; Computer science; Decoding methods; Speech translation; Transfer-based machine translation; Translation (biology); Phrase; Natural language processing; Artificial intelligence; Speech recognition; Synchronous context-free grammar; Example-based machine translation; Algorithm","score_opus":0.029870486058793368,"score_gpt":0.2730560117679839,"score_spread":0.24318552570919053,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2071643874","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.045096103,0.00066988857,0.92381185,0.00021346488,0.0001780337,0.00028539164,0.0010120356,0.01939915,0.009334133],"genre_scores_gemma":[0.24050085,0.00036997828,0.74861985,0.00020039451,0.00014023761,0.00027685225,0.0039033382,0.0005296355,0.0054589114],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99928075,0.00018292306,0.000069162445,0.00019113053,0.00022027962,0.00005572633],"domain_scores_gemma":[0.99909794,0.00027714865,0.00006457849,0.00023460994,0.00028694092,0.000038844973],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00064820953,0.00064214366,0.0007749503,0.0010302273,0.0005766521,0.0009066638,0.0011315817,0.00081261626,0.004705294],"category_scores_gemma":[0.0017349652,0.00032439132,0.0006879531,0.001401782,0.00038510954,0.0010892873,0.0010731862,0.0008186072,0.0035023852],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00058428023,0.00025815165,0.0016450883,0.0005856593,0.00012014891,0.0009089378,0.0005332074,0.04422931,0.15542448,0.011594668,0.024842966,0.75927305],"study_design_scores_gemma":[0.00021498116,0.0005604277,0.0027732302,0.000051277708,0.00029610388,0.0016547809,0.00022264018,0.82709086,0.11161374,0.013787047,0.04159376,0.00014106772],"about_ca_topic_score_codex":0.0023220035,"about_ca_topic_score_gemma":0.0030057014,"teacher_disagreement_score":0.004705294,"about_ca_system_score_codex":0.0003929128,"about_ca_system_score_gemma":0.0010992613,"threshold_uncertainty_score":0.015740752},"labels":[],"label_agreement":null},{"id":"W2071786876","doi":"10.4000/corela.574","title":"La modification adjectivale en arabe à la lumière de la grammaire adaptative","year":2005,"lang":"fr","type":"article","venue":"Cognition représentation langages","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Philosophy; Humanities","score_opus":0.029044000019413572,"score_gpt":0.3493958718145303,"score_spread":0.3203518717951167,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2071786876","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2292204,0.00181605,0.5153444,0.0037972804,0.0007803386,0.00012851204,0.00063352013,0.002457528,0.24582203],"genre_scores_gemma":[0.8934492,0.0006453922,0.070017874,0.00047885976,0.00013896369,0.000089540656,0.0003710161,0.0005745407,0.034234542],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99885535,0.0005229326,0.00008333149,0.00023503462,0.00023435017,0.000069025424],"domain_scores_gemma":[0.99818856,0.0007364951,0.00013159435,0.00049129484,0.000410829,0.000041178646],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011005417,0.0005856016,0.00037182702,0.0005846301,0.0012352914,0.0024851866,0.0004428441,0.0005996844,0.006521773],"category_scores_gemma":[0.0027452335,0.00036344954,0.00050477195,0.00052044285,0.0041620503,0.00336816,0.0010845931,0.0020088225,0.0018775109],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001986868,0.0000327341,0.002906764,0.00029893397,0.00004165822,0.00052485266,0.020552356,0.00094267563,0.03288955,0.8452078,0.00448187,0.091922075],"study_design_scores_gemma":[0.00005066574,0.0002190018,0.01422502,0.00032663895,0.00014174882,0.004109258,0.014360973,0.013848398,0.034213033,0.3424978,0.5757752,0.00023232256],"about_ca_topic_score_codex":0.004462233,"about_ca_topic_score_gemma":0.0042103906,"teacher_disagreement_score":0.006521773,"about_ca_system_score_codex":0.0009098614,"about_ca_system_score_gemma":0.0007952758,"threshold_uncertainty_score":0.021817446},"labels":[],"label_agreement":null},{"id":"W2071994981","doi":"10.7202/1011262ar","title":"Gazing and Typing Activities during Translation: A Comparative Study of Translation Units of Professional and Student Translators","year":2012,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":120,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Source text; Computer science; Focus (optics); Linguistics; Comprehension; Danish; Eye tracking; Translation (biology); Fixation (population genetics); Cognition; Reading (process); Reading comprehension; Gaze; Target text; Natural language processing; Psychology; Artificial intelligence; Sociology; Programming language","score_opus":0.0840224521660508,"score_gpt":0.34054217088840366,"score_spread":0.25651971872235285,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2071994981","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99856013,0.00012885469,0.00046725123,0.000020905618,0.0000029899443,0.000008327107,0.000030572064,0.000011078679,0.0007698886],"genre_scores_gemma":[0.9988544,0.0001213337,0.0002760301,0.000010985442,0.0000060030898,0.000015584777,0.00007327578,0.000011131281,0.0006312327],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9987496,0.00050888123,0.00011206977,0.00026185147,0.00022751026,0.00014007493],"domain_scores_gemma":[0.99174964,0.004550821,0.0015853749,0.00039752762,0.0010823894,0.0006342304],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010293003,0.00026554454,0.0003873644,0.0015666,0.0004906326,0.0012951967,0.00032704882,0.00058814895,0.0022898668],"category_scores_gemma":[0.01476858,0.00023711949,0.00022787724,0.0011269372,0.00070594193,0.00092261954,0.00085002644,0.00029508202,0.0007791801],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014962775,0.00026791036,0.4623652,0.00050528947,0.00014950956,0.0018251901,0.36731675,0.00021086923,0.03776385,0.0002585354,0.00060636905,0.12723427],"study_design_scores_gemma":[0.000017253427,0.000596494,0.93633807,0.00004407124,0.000033873897,0.0010814263,0.0572266,0.00047479608,0.0022020421,0.00016134308,0.0017867144,0.000037328333],"about_ca_topic_score_codex":0.0015412758,"about_ca_topic_score_gemma":0.002057917,"teacher_disagreement_score":0.0022898668,"about_ca_system_score_codex":0.00026950947,"about_ca_system_score_gemma":0.00024048965,"threshold_uncertainty_score":0.0076603293},"labels":[],"label_agreement":null},{"id":"W2073372957","doi":"10.1109/ichit.2006.48","title":"Action Representation for Natural Language Interfaces to Agent Systems","year":2006,"lang":"en","type":"article","venue":"International Conference on Hybrid Information Technology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba","funders":"","keywords":"Computer science; Natural language; Hierarchy; Natural language understanding; Artificial intelligence; Parsing; Natural language processing; Human–computer interaction; Representation (politics); Action (physics); Frame (networking); Natural language user interface; Embodied agent; Embodied cognition","score_opus":0.02590900249658061,"score_gpt":0.32962914083015105,"score_spread":0.3037201383335704,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2073372957","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0007145418,0.00024439508,0.9909039,0.00045889305,0.000086099666,0.00010181735,0.00014218426,0.002322886,0.0050252248],"genre_scores_gemma":[0.07136373,0.0006750193,0.91810995,0.0005122358,0.00017689772,0.0008201461,0.0011207626,0.00055946043,0.0066618016],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99593675,0.0017629103,0.00053047284,0.00060544856,0.0009194816,0.0002448706],"domain_scores_gemma":[0.99749076,0.0012643178,0.00018781183,0.0005125622,0.0004477222,0.00009696487],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0039196406,0.001272155,0.0010273487,0.001683155,0.0010603382,0.006071174,0.0031136717,0.002705313,0.011347085],"category_scores_gemma":[0.008577315,0.0007659989,0.002201336,0.0012908992,0.0032348498,0.0067814477,0.003451576,0.0033167421,0.0036316982],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000043185515,0.000031556596,0.00009129767,0.00026929245,0.000025305222,0.00017691376,0.0008910523,0.010647149,0.001845124,0.94092625,0.004704348,0.040348534],"study_design_scores_gemma":[0.00006247624,0.0000628151,0.000115465366,0.00020802645,0.00006800228,0.0002254869,0.00023032313,0.14481184,0.0038800833,0.7210585,0.12921865,0.000058295154],"about_ca_topic_score_codex":0.003692611,"about_ca_topic_score_gemma":0.002822135,"teacher_disagreement_score":0.011347085,"about_ca_system_score_codex":0.0021867289,"about_ca_system_score_gemma":0.001606996,"threshold_uncertainty_score":0.037959754},"labels":[],"label_agreement":null},{"id":"W2073965571","doi":"10.1109/tvcg.2012.226","title":"Facilitating Discourse Analysis with Interactive Visualization","year":2012,"lang":"en","type":"article","venue":"IEEE Transactions on Visualization and Computer Graphics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":56,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ontario Tech University; University of Toronto","funders":"","keywords":"Computer science; Automatic summarization; Parsing; Visualization; Computational linguistics; Natural language processing; Rhetorical question; Domain (mathematical analysis); Question answering; Process (computing); Interactive visualization; Data visualization; Formative assessment; Natural language; Artificial intelligence; Human–computer interaction; Programming language; Linguistics","score_opus":0.014116119413438113,"score_gpt":0.3144544023644846,"score_spread":0.3003382829510465,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2073965571","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02523814,0.00041017047,0.9042394,0.0010531592,0.00015193585,0.00062016485,0.0011574504,0.059094913,0.008034567],"genre_scores_gemma":[0.15794095,0.00041631167,0.8302083,0.00033750958,0.00015536415,0.001548739,0.0013969765,0.0037057607,0.004290239],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9957496,0.0025567363,0.00023067766,0.00053910876,0.0007008478,0.00022303972],"domain_scores_gemma":[0.9680434,0.026355784,0.0007273352,0.0024830105,0.0018310361,0.00055940694],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006537568,0.0021857582,0.0011439322,0.0030516493,0.0011596759,0.0050303848,0.0021644835,0.0015841986,0.01876975],"category_scores_gemma":[0.030375078,0.0007735308,0.0010156987,0.0017071709,0.0009364135,0.0047053434,0.0057203635,0.0020256524,0.0029819359],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001865381,0.00069082156,0.004791819,0.00287478,0.00025633976,0.0021801537,0.04836972,0.019512214,0.15864556,0.04654989,0.09495777,0.61930555],"study_design_scores_gemma":[0.00078142615,0.0006506602,0.0065402742,0.0011710371,0.00026637255,0.001466098,0.007046417,0.2879656,0.13008066,0.10571785,0.45771486,0.00059870584],"about_ca_topic_score_codex":0.00096897164,"about_ca_topic_score_gemma":0.0012475026,"teacher_disagreement_score":0.01876975,"about_ca_system_score_codex":0.0005664844,"about_ca_system_score_gemma":0.0010849197,"threshold_uncertainty_score":0.06279105},"labels":[],"label_agreement":null},{"id":"W2074118501","doi":"10.3115/1641976.1641987","title":"A measure of aggregate syntactic distance","year":2006,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":28,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Oulun Yliopisto; Rijksuniversiteit Groningen; Université de Moncton","keywords":"Trigram; Syntax; Computer science; Natural language processing; Measure (data warehouse); Aggregate (composite); Artificial intelligence; Permutation (music); Syntactic structure","score_opus":0.007379287816726015,"score_gpt":0.23481634911632882,"score_spread":0.2274370612996028,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2074118501","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.628944,0.00081578235,0.34627005,0.00032455742,0.00018234397,0.00022673537,0.0043856804,0.0016510419,0.017199954],"genre_scores_gemma":[0.9044977,0.00021248568,0.08841786,0.00008975391,0.00011910072,0.00026277668,0.0039495397,0.00026222895,0.002188601],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9954521,0.00076817075,0.0005836129,0.00090286025,0.0020545966,0.00023868358],"domain_scores_gemma":[0.9875244,0.0059974105,0.0015438664,0.0021892253,0.0022868416,0.0004582083],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025236702,0.0006158829,0.0009151332,0.00767521,0.0006786427,0.0024461974,0.0010305749,0.000999694,0.0032220418],"category_scores_gemma":[0.017052757,0.00021680084,0.0005738493,0.006942266,0.0012255897,0.0040431092,0.0019774418,0.0009748506,0.0014582416],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014848959,0.0005584698,0.19184324,0.000868179,0.0011905197,0.0006661398,0.0028042272,0.01842674,0.088639334,0.07286567,0.007092324,0.61356014],"study_design_scores_gemma":[0.00014093154,0.0029909206,0.45564497,0.00017160931,0.0006381534,0.0052953465,0.004339234,0.20733531,0.05520808,0.23283783,0.034817874,0.0005798178],"about_ca_topic_score_codex":0.00058814586,"about_ca_topic_score_gemma":0.00059511117,"teacher_disagreement_score":0.00767521,"about_ca_system_score_codex":0.0005199054,"about_ca_system_score_gemma":0.0004176871,"threshold_uncertainty_score":0.013346612},"labels":[],"label_agreement":null},{"id":"W2074781470","doi":"10.1075/ml.2.2.09wes","title":"LINGUA","year":2007,"lang":"en","type":"article","venue":"The Mental Lexicon","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Lingua franca; Computer science; Natural language processing; Artificial intelligence; Neighbourhood (mathematics); Generator (circuit theory); Java; Set (abstract data type); Linguistics; Programming language; Mathematics","score_opus":0.01237956198954438,"score_gpt":0.28987693490315514,"score_spread":0.27749737291361076,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2074781470","genre_codex":"software","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008793881,0.0012936564,0.18922578,0.0011776661,0.0012693034,0.0005815716,0.08632627,0.36905038,0.34228155],"genre_scores_gemma":[0.05111828,0.0013022097,0.32418805,0.0010586107,0.00028589548,0.0009012666,0.20073299,0.08583879,0.33457392],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990808,0.00009332628,0.00005019459,0.00024042862,0.00040590964,0.00012946731],"domain_scores_gemma":[0.99840015,0.0002247287,0.000058647136,0.00040185053,0.00071158184,0.000202937],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014075561,0.0010788004,0.0009820763,0.0023010361,0.0020106118,0.003481771,0.0020800696,0.0008204749,0.17604016],"category_scores_gemma":[0.003874316,0.0011821222,0.000946561,0.002162749,0.0005758256,0.0022796537,0.002721947,0.0017893123,0.15359387],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00038801512,0.00007191227,0.001725883,0.0005137944,0.00004018274,0.00018202406,0.0006141733,0.00066593435,0.0074703307,0.026993884,0.6464737,0.3148602],"study_design_scores_gemma":[0.000055427045,0.000023397015,0.0017162353,0.00006777735,0.000023397193,0.00021833752,0.00013428212,0.0031769166,0.004148114,0.008396256,0.9819917,0.00004820934],"about_ca_topic_score_codex":0.04654608,"about_ca_topic_score_gemma":0.071804136,"teacher_disagreement_score":0.17604016,"about_ca_system_score_codex":0.0021733285,"about_ca_system_score_gemma":0.0055002864,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W2075249800","doi":"10.3115/1621474.1621580","title":"UofL","year":2007,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Lethbridge","funders":"","keywords":"Computer science; SemEval; Automatic summarization; Natural language processing; Word (group theory); Context (archaeology); Artificial intelligence; Task (project management); Word-sense disambiguation; Question answering; Natural language; Information retrieval; Linguistics; WordNet","score_opus":0.009317795113138455,"score_gpt":0.2802391759585755,"score_spread":0.27092138084543704,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2075249800","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006666769,0.0014364414,0.018767135,0.0018502808,0.0021921599,0.00027252108,0.007895295,0.0052193375,0.9557],"genre_scores_gemma":[0.086854935,0.002065208,0.019912962,0.002896189,0.0005109969,0.00040526708,0.020926183,0.0022215003,0.8642068],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99875677,0.00016298944,0.00007832238,0.00041878957,0.00034798877,0.00023505781],"domain_scores_gemma":[0.998961,0.00009741127,0.000061416715,0.0003050736,0.0004678886,0.00010714362],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0007280161,0.0010124476,0.00064856047,0.0019021962,0.0028194634,0.004528307,0.0013252547,0.0018457542,0.4440846],"category_scores_gemma":[0.0027099254,0.00040640318,0.0005908227,0.0014936129,0.0007555655,0.0031553663,0.0035803656,0.0012744174,0.3017984],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005758854,0.00012924384,0.0029390387,0.00061418046,0.000043672055,0.000660271,0.00079917785,0.0008322329,0.009888009,0.1159271,0.40243265,0.46515855],"study_design_scores_gemma":[0.000013842057,0.000033458688,0.00066175085,0.00007878844,0.00001057197,0.00022957045,0.00020922812,0.0003752223,0.0023997447,0.0075802086,0.98839265,0.000014965534],"about_ca_topic_score_codex":0.005369697,"about_ca_topic_score_gemma":0.0045471713,"teacher_disagreement_score":0.55591536,"about_ca_system_score_codex":0.0012709952,"about_ca_system_score_gemma":0.0016496708,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W2075708439","doi":"10.1044/sbi4.1.52","title":"Language-Reading Resource Model","year":2003,"lang":"en","type":"article","venue":"Perspectives on School-Based Issues","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"CobB; Library science; Reading (process); Resource (disambiguation); Curriculum; Section (typography); Computer science; Sociology; Political science; Pedagogy; Law","score_opus":0.012932679198178245,"score_gpt":0.30428997974320776,"score_spread":0.2913573005450295,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2075708439","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011804633,0.0007944613,0.14653954,0.02616552,0.0003441476,0.00034517146,0.0021555212,0.0021609499,0.8096901],"genre_scores_gemma":[0.4800595,0.0012720763,0.08717739,0.004125429,0.00059701497,0.0010172103,0.004443289,0.0011073286,0.4202007],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99545693,0.0017526489,0.0002251183,0.0010097116,0.001055908,0.00049967977],"domain_scores_gemma":[0.99292856,0.0027290187,0.00027045578,0.0015438673,0.001601054,0.000927106],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033335704,0.0008518606,0.0009030625,0.0021045504,0.0029229717,0.009544611,0.0039730417,0.003428571,0.08350899],"category_scores_gemma":[0.011466441,0.0006675133,0.0012012831,0.0023505487,0.0041237003,0.026855022,0.0051793614,0.0034502281,0.02658886],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003264392,0.00005550286,0.00044334334,0.00004761326,0.0000066662687,0.0001219475,0.00049765944,0.002153491,0.00015221896,0.9555245,0.027208827,0.01375551],"study_design_scores_gemma":[0.000049315466,0.000037207094,0.0007380883,0.000086282584,0.000024587627,0.00049366493,0.001251095,0.03559667,0.0003982662,0.7468611,0.21442005,0.00004360391],"about_ca_topic_score_codex":0.01896278,"about_ca_topic_score_gemma":0.013784878,"teacher_disagreement_score":0.08350899,"about_ca_system_score_codex":0.005903315,"about_ca_system_score_gemma":0.0057334155,"threshold_uncertainty_score":0.27936542},"labels":[],"label_agreement":null},{"id":"W2076749833","doi":"10.3115/1614049.1614099","title":"Comparing the roles of textual, acoustic and spoken-language features on spontaneous-conversation summarization","year":2006,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Automatic summarization; Computer science; Conversation; Natural language processing; Feature (linguistics); Speech recognition; Artificial intelligence; Linguistics","score_opus":0.006667169071351452,"score_gpt":0.23136671073366635,"score_spread":0.2246995416623149,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2076749833","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.91649395,0.004397222,0.06926857,0.00054812554,0.00017761359,0.0003732185,0.00081050355,0.0034672113,0.0044634384],"genre_scores_gemma":[0.9513158,0.00075074006,0.043831922,0.000059493665,0.00018994776,0.00014300714,0.0023619435,0.0001804309,0.0011666078],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9968888,0.0020168533,0.00021317456,0.00035010173,0.00036988815,0.00016107227],"domain_scores_gemma":[0.95985216,0.03375625,0.0014162422,0.0010700976,0.003309188,0.00059604505],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004783762,0.0010958873,0.0009069053,0.0017228402,0.00044861806,0.0016307551,0.0005722989,0.0007589586,0.0011143442],"category_scores_gemma":[0.025574535,0.0002714922,0.00045408407,0.0010866392,0.00031821185,0.0033101407,0.0006339064,0.0007166632,0.0006333726],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.005970993,0.0011858674,0.018045843,0.0018977901,0.0007881812,0.0003119094,0.0015469502,0.02864901,0.104094245,0.0006566862,0.003228213,0.8336243],"study_design_scores_gemma":[0.0007020251,0.017269537,0.16428809,0.00028499163,0.0027646543,0.0006988575,0.0045263683,0.5584133,0.23626813,0.003537072,0.01072415,0.0005228204],"about_ca_topic_score_codex":0.0014275768,"about_ca_topic_score_gemma":0.002480655,"teacher_disagreement_score":0.004783762,"about_ca_system_score_codex":0.00034684598,"about_ca_system_score_gemma":0.00037923324,"threshold_uncertainty_score":0.025299251},"labels":[],"label_agreement":null},{"id":"W2077193853","doi":"10.1145/1562764.1562798","title":"Human interaction for high-quality machine translation","year":2009,"lang":"en","type":"article","venue":"Communications of the ACM","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":49,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Machine translation; Quality (philosophy); Computer science; Task (project management); Newspaper; Artificial intelligence; Field (mathematics); Point (geometry); Globalization; Natural language processing; Political science; Law; Management; Economics","score_opus":0.09096658317031812,"score_gpt":0.41161388785084935,"score_spread":0.3206473046805312,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2077193853","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012045428,0.014241934,0.6888607,0.011747862,0.0028253046,0.00066222314,0.0014686409,0.021092787,0.24705517],"genre_scores_gemma":[0.43783125,0.0071422467,0.4081193,0.007883385,0.0026737694,0.0012850442,0.0032941727,0.004146548,0.12762432],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99458176,0.0027133825,0.00020769186,0.0008286852,0.0014331696,0.00023523039],"domain_scores_gemma":[0.9922808,0.0045544617,0.00031163773,0.0012116879,0.0012602764,0.00038105846],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003624078,0.0012385348,0.0007291245,0.00087497785,0.0014164803,0.005206391,0.0017227256,0.0033925392,0.14171903],"category_scores_gemma":[0.013814568,0.00043056384,0.00060200086,0.0009464389,0.0017577517,0.004016381,0.005311971,0.0014184228,0.051840592],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00071472436,0.00014313107,0.0012906963,0.0022624067,0.00008822912,0.0007281756,0.0021878888,0.0037208423,0.03079377,0.039894305,0.18819268,0.7299832],"study_design_scores_gemma":[0.00010024248,0.00031368574,0.0033052557,0.00079364044,0.0000810897,0.0017666216,0.0012273572,0.025629597,0.013896132,0.086595304,0.86610806,0.00018305313],"about_ca_topic_score_codex":0.0010834076,"about_ca_topic_score_gemma":0.0010773609,"teacher_disagreement_score":0.14171903,"about_ca_system_score_codex":0.0008598882,"about_ca_system_score_gemma":0.0010005978,"threshold_uncertainty_score":0.47409737},"labels":[],"label_agreement":null},{"id":"W2077255111","doi":"10.3115/1118627.1118636","title":"Acquiring collocations for lexical choice between near-synonyms","year":2002,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":46,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada; University of Toronto; University of Pennsylvania","keywords":"Collocation (remote sensing); Computer science; Synonym (taxonomy); Natural language processing; Artificial intelligence; Lexical density; Word (group theory); Linguistics; Task (project management); Lexical item","score_opus":0.04991608310263382,"score_gpt":0.31198219993089615,"score_spread":0.2620661168282623,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2077255111","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5360507,0.000361845,0.4447891,0.00055149314,0.00008392101,0.00036660576,0.0038825967,0.0028279256,0.011085829],"genre_scores_gemma":[0.7819096,0.00012467238,0.21109655,0.00010120648,0.000024108302,0.00020617158,0.004864415,0.00026909364,0.0014041772],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99717903,0.00068994646,0.00031822504,0.0011530898,0.00051210745,0.00014761936],"domain_scores_gemma":[0.9846631,0.00983435,0.00094150717,0.002171378,0.0019667526,0.00042290377],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020692653,0.0007155539,0.00089296856,0.0034664795,0.0011060991,0.0016859745,0.0014935341,0.00120584,0.008213305],"category_scores_gemma":[0.01869494,0.0006535536,0.00067353674,0.0024616988,0.0011238465,0.0069955466,0.0027961337,0.00156021,0.0032974961],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00071341026,0.00071136124,0.083040915,0.00093440345,0.00024248312,0.0021087269,0.007861663,0.010789174,0.097447984,0.022163771,0.008323559,0.76566267],"study_design_scores_gemma":[0.000335045,0.0008201972,0.16941643,0.0005491984,0.00040446725,0.0053211185,0.014481788,0.41055885,0.11900252,0.20753653,0.07103167,0.0005422843],"about_ca_topic_score_codex":0.0027544925,"about_ca_topic_score_gemma":0.0057255947,"teacher_disagreement_score":0.008213305,"about_ca_system_score_codex":0.0006741909,"about_ca_system_score_gemma":0.0010526753,"threshold_uncertainty_score":0.027476251},"labels":[],"label_agreement":null},{"id":"W2077476855","doi":"10.7202/003356ar","title":"Quantum Mechanics and the Theory of Poetry Translation","year":2002,"lang":"fr","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Philosophy; Humanities; Relation (database); Poetry; Physics; Epistemology; Linguistics; Computer science","score_opus":0.058053620456790606,"score_gpt":0.269560381290502,"score_spread":0.21150676083371142,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2077476855","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08197352,0.02336793,0.15654181,0.037529033,0.0037437766,0.00009187675,0.00033606944,0.00031746321,0.6960985],"genre_scores_gemma":[0.90635693,0.0062779523,0.020846058,0.0019385574,0.0017523966,0.000117699135,0.00016127936,0.00007579613,0.062473346],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99928254,0.0003227129,0.00002981507,0.00010113547,0.00017409986,0.00008969459],"domain_scores_gemma":[0.9991635,0.00047218316,0.00008983002,0.00011892342,0.00010754733,0.000048004797],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00087060087,0.000415429,0.00049986655,0.0011371754,0.0022532428,0.0044797384,0.0006994365,0.001455453,0.011813939],"category_scores_gemma":[0.0027132586,0.0002768465,0.0006606461,0.0012917812,0.0067982366,0.0059218355,0.001716989,0.0022767629,0.0016948784],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000037583113,0.0000032625258,0.000024223875,0.0000118477765,0.0000015137497,0.000029031913,0.0001950426,0.00018685417,0.00004958467,0.9963676,0.00089501083,0.0022323956],"study_design_scores_gemma":[0.0000034789155,0.000004632933,0.00004568783,0.000010862569,9.268746e-7,0.00002746929,0.00008165814,0.00071521313,0.000039211773,0.993276,0.005791956,0.0000029288867],"about_ca_topic_score_codex":0.0014544077,"about_ca_topic_score_gemma":0.001039111,"teacher_disagreement_score":0.011813939,"about_ca_system_score_codex":0.0014295726,"about_ca_system_score_gemma":0.0010203803,"threshold_uncertainty_score":0.039521575},"labels":[],"label_agreement":null},{"id":"W2077811574","doi":"10.3115/1073336.1073350","title":"Identifying cognates by phonetic and semantic similarity","year":2001,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":88,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada; University of Pennsylvania","keywords":"WordNet; Computer science; Similarity (geometry); Artificial intelligence; Natural language processing; Semantic similarity; Longest common subsequence problem; Selection (genetic algorithm); Algorithm","score_opus":0.013766470490859325,"score_gpt":0.27195486169596567,"score_spread":0.25818839120510634,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2077811574","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.49270794,0.0018454075,0.46736938,0.000331895,0.00016516991,0.000311671,0.0028469781,0.003353148,0.031068308],"genre_scores_gemma":[0.78168106,0.00043233534,0.21077856,0.00008145212,0.00006784997,0.00018665179,0.0030877143,0.0002934723,0.0033909255],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987859,0.00022845867,0.00021364115,0.00036433127,0.00032104383,0.00008666188],"domain_scores_gemma":[0.99672467,0.001128795,0.00039094506,0.00065085094,0.0009801021,0.00012471598],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010807125,0.0005609548,0.0005713157,0.0074199573,0.0011932367,0.0019622983,0.00071065885,0.00057937426,0.0047336365],"category_scores_gemma":[0.009572839,0.0003350152,0.0008550427,0.0030838768,0.00086400297,0.0032349573,0.0019485924,0.00044728327,0.002237852],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007242381,0.0001971981,0.13316752,0.00069892796,0.0003069871,0.0009402709,0.002784086,0.0035654504,0.07454192,0.0314569,0.0046071787,0.7470093],"study_design_scores_gemma":[0.00015466071,0.00085122953,0.27697772,0.00059936196,0.0013748197,0.010467273,0.008664816,0.26214924,0.18387553,0.14141862,0.11288141,0.0005853108],"about_ca_topic_score_codex":0.0051180734,"about_ca_topic_score_gemma":0.0066458173,"teacher_disagreement_score":0.0074199573,"about_ca_system_score_codex":0.00050985464,"about_ca_system_score_gemma":0.0010145748,"threshold_uncertainty_score":0.015835524},"labels":[],"label_agreement":null},{"id":"W2077813743","doi":"10.3166/ria.16.339-366","title":"Raisonnement à base de cas textuels Etat de l'art et perspectives","year":2002,"lang":"fr","type":"article","venue":"Revue d intelligence artificielle","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Philosophy; Humanities","score_opus":0.058168671589977315,"score_gpt":0.31619782401449625,"score_spread":0.25802915242451896,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2077813743","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024923569,0.042366873,0.8153609,0.011774274,0.0009870274,0.00043507785,0.000311454,0.0016039207,0.10223698],"genre_scores_gemma":[0.23907197,0.039677408,0.65651333,0.0022005723,0.0009640355,0.00045117922,0.0007993061,0.0005807926,0.059741504],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99235654,0.0029322573,0.00052521785,0.0013081315,0.0026212353,0.00025655705],"domain_scores_gemma":[0.99021405,0.004872586,0.00044092117,0.001878396,0.002403266,0.00019086544],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0078472225,0.00091187056,0.00081282575,0.0044006603,0.0019163586,0.01131601,0.0032284877,0.002859638,0.011141045],"category_scores_gemma":[0.015117794,0.0007680785,0.001318595,0.004342479,0.00517008,0.01034794,0.002395437,0.0025539692,0.0026519329],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020116408,0.00020181558,0.0016408735,0.0023325228,0.00012332128,0.0011619512,0.009306857,0.014119654,0.012361047,0.48065785,0.014390444,0.4635025],"study_design_scores_gemma":[0.000057758625,0.00020338096,0.0017910023,0.0018704991,0.0001729339,0.0029599927,0.005337225,0.041662335,0.020659441,0.1786896,0.746386,0.00020978333],"about_ca_topic_score_codex":0.011576164,"about_ca_topic_score_gemma":0.0073451973,"teacher_disagreement_score":0.011576164,"about_ca_system_score_codex":0.003092669,"about_ca_system_score_gemma":0.0025613203,"threshold_uncertainty_score":0.04150057},"labels":[],"label_agreement":null},{"id":"W2077818552","doi":"10.1162/coli.2010.36.1.36104","title":"Automatically Identifying the Source Words of Lexical Blends in English","year":2010,"lang":"en","type":"article","venue":"Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":36,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canada Research Chairs; University of Toronto","funders":"University of Toronto","keywords":"Computer science; Natural language processing; Lexicon; Artificial intelligence; Task (project management); Set (abstract data type); Identification (biology); Word (group theory); Source text; Linguistics; Programming language","score_opus":0.013274499254340271,"score_gpt":0.29620196724369235,"score_spread":0.2829274679893521,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2077818552","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9248342,0.0010666247,0.062561646,0.00041955922,0.00008583207,0.00011023999,0.0030352783,0.001577151,0.00630935],"genre_scores_gemma":[0.9448154,0.00048359134,0.047611803,0.00007942822,0.000046219913,0.00006215484,0.0041070282,0.00033039539,0.0024639892],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990075,0.00024424188,0.00018821619,0.000358569,0.00014854336,0.000053047366],"domain_scores_gemma":[0.99576145,0.0027162265,0.00059595157,0.0003164616,0.0005270848,0.00008294561],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010545212,0.000705841,0.0005604802,0.0021538998,0.00084324635,0.0013727536,0.0005703391,0.0006234822,0.004629331],"category_scores_gemma":[0.0048730285,0.0005177396,0.00049559027,0.0016388639,0.0007904175,0.004932688,0.0013134448,0.00067825185,0.0017797464],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017308524,0.00039587406,0.1301558,0.003000718,0.00022539042,0.0045432593,0.012854994,0.006506252,0.1806267,0.015099924,0.015488085,0.6293721],"study_design_scores_gemma":[0.0002342961,0.0006191266,0.32825944,0.0006218338,0.00058108394,0.011042084,0.019117331,0.3299426,0.17919199,0.03495542,0.095079005,0.00035576362],"about_ca_topic_score_codex":0.003005474,"about_ca_topic_score_gemma":0.0047369264,"teacher_disagreement_score":0.004629331,"about_ca_system_score_codex":0.00051726453,"about_ca_system_score_gemma":0.0004126126,"threshold_uncertainty_score":0.015486717},"labels":[],"label_agreement":null},{"id":"W2079048175","doi":"10.3115/1626516.1626524","title":"Phonological reconstruction of a dead language using the gradual learning algorithm","year":2007,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Orthography; Phonology; Lexicon; Computer science; Phonological rule; Artificial intelligence; Natural language processing; Linguistics; Philosophy","score_opus":0.01978259389532651,"score_gpt":0.2946673617370643,"score_spread":0.2748847678417378,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2079048175","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04885682,0.00005343395,0.9468718,0.000089139896,0.00003560093,0.000021555217,0.000026483805,0.0006436099,0.0034015724],"genre_scores_gemma":[0.55759394,0.00011345239,0.43690485,0.000066280976,0.000022460394,0.00003519505,0.00009910065,0.00018015906,0.004984535],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997882,0.000055076016,0.000015531747,0.000061207094,0.000060729737,0.000019248377],"domain_scores_gemma":[0.99953604,0.00021370864,0.00003450504,0.000117078554,0.00007597239,0.00002266174],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00053019053,0.00028217823,0.00024414784,0.0004635367,0.000330165,0.00085032213,0.0010687262,0.00036107216,0.0025308102],"category_scores_gemma":[0.0022439312,0.00023390021,0.000535599,0.00021146718,0.0011413449,0.0010949309,0.0013413407,0.00088872446,0.0005363791],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036343955,0.00007025743,0.0022985404,0.0002832801,0.000088590394,0.0010028699,0.0011832833,0.17234045,0.06414232,0.35782796,0.0014063319,0.39899272],"study_design_scores_gemma":[0.00006391363,0.00026317663,0.0009242136,0.00004277259,0.000058820442,0.0007597626,0.00031013275,0.8211473,0.03535769,0.1302755,0.010725254,0.00007135849],"about_ca_topic_score_codex":0.0007404654,"about_ca_topic_score_gemma":0.0013459226,"teacher_disagreement_score":0.0025308102,"about_ca_system_score_codex":0.0002580025,"about_ca_system_score_gemma":0.0003908303,"threshold_uncertainty_score":0.008466423},"labels":[],"label_agreement":null},{"id":"W2079342277","doi":"10.1093/ijl/ecu023","title":"Dictionaries and the Digital Revolution: A Focus on Users and Lexical Databases","year":2014,"lang":"en","type":"article","venue":"International Journal of Lexicography","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":58,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Lexicography; Digital humanities; Focus (optics); Sociology; Library science; Linguistics; Humanities; Computer science; Art; Philosophy; Physics","score_opus":0.010882790041837044,"score_gpt":0.2667891316734581,"score_spread":0.255906341631621,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2079342277","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01585951,0.3243098,0.011854522,0.48168314,0.012118175,0.000040626892,0.00057482853,0.00074129424,0.15281804],"genre_scores_gemma":[0.34918308,0.3371187,0.020631814,0.07299243,0.05601182,0.00020556255,0.0011629504,0.0023974245,0.1602963],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.98975813,0.00549622,0.00082252,0.0012387576,0.0020407771,0.0006436373],"domain_scores_gemma":[0.93828785,0.047153927,0.002564753,0.0039465227,0.004119438,0.00392743],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01102518,0.0004871315,0.0013012493,0.010066201,0.004823565,0.04176407,0.0017718028,0.005042385,0.020267801],"category_scores_gemma":[0.022302344,0.00086020783,0.0005639771,0.017727273,0.01853153,0.05897289,0.010453305,0.006549911,0.0034252503],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009640944,0.00004147262,0.001737744,0.0008218348,0.000025358715,0.00036776182,0.018373586,0.00011878122,0.00054466224,0.5739987,0.2345637,0.16930994],"study_design_scores_gemma":[0.000008213358,0.00001861555,0.0010038794,0.0005730517,0.0000075004755,0.0005362629,0.010083032,0.00021158568,0.00019686173,0.04363242,0.94369686,0.0000316769],"about_ca_topic_score_codex":0.0042254576,"about_ca_topic_score_gemma":0.004664078,"teacher_disagreement_score":0.04176407,"about_ca_system_score_codex":0.008203872,"about_ca_system_score_gemma":0.003522307,"threshold_uncertainty_score":0.06780261},"labels":[],"label_agreement":null},{"id":"W2079754016","doi":"10.3115/1613186.1613187","title":"Statistical measures of the semi-productivity of light verb constructions","year":2004,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":61,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Measure (data warehouse); Productivity; Verb; Natural language processing; Computer science; Artificial intelligence; Statistical analysis; Linguistics; Mathematics; Statistics; Data mining; Philosophy","score_opus":0.014977395325542998,"score_gpt":0.2579192293006384,"score_spread":0.2429418339750954,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2079754016","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7260802,0.0005405645,0.2645734,0.00016446556,0.000040229326,0.00019067753,0.0015124043,0.0007609707,0.006137092],"genre_scores_gemma":[0.97203964,0.000105330495,0.025839752,0.000028852437,0.000058587357,0.00019815084,0.0012999767,0.000115035815,0.00031458956],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.98446,0.005560984,0.0024770238,0.0022141444,0.0048326827,0.00045523216],"domain_scores_gemma":[0.5947693,0.3168658,0.045274626,0.023594273,0.016297631,0.0031983536],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013611011,0.00065239484,0.00067666615,0.006146438,0.00078249985,0.0026500216,0.0012566467,0.0011171509,0.0021531067],"category_scores_gemma":[0.13019462,0.0005181153,0.0008027167,0.0033407074,0.0035980532,0.003489574,0.002551194,0.0015850812,0.00063189084],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021732794,0.00050309865,0.59287745,0.0013643783,0.0009149669,0.0009266249,0.006772845,0.03016241,0.08577474,0.032578513,0.0017529575,0.24419865],"study_design_scores_gemma":[0.0001001704,0.0015030499,0.67267853,0.00014414774,0.00029887224,0.003259223,0.0025942696,0.1678163,0.045192495,0.09954751,0.006441155,0.00042432477],"about_ca_topic_score_codex":0.00047616716,"about_ca_topic_score_gemma":0.0005300779,"teacher_disagreement_score":0.013611011,"about_ca_system_score_codex":0.000540482,"about_ca_system_score_gemma":0.0005099922,"threshold_uncertainty_score":0.07198274},"labels":[],"label_agreement":null},{"id":"W2080021477","doi":"10.1016/j.neucom.2008.12.025","title":"Improving a statistical language model through non-linear prediction","year":2009,"lang":"en","type":"article","venue":"Neurocomputing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Perplexity; Computer science; Feature (linguistics); Context (archaeology); Artificial intelligence; Language model; Word (group theory); Feature vector; Linear model; Term (time); Pattern recognition (psychology); Machine learning; Mathematics","score_opus":0.012650860748946606,"score_gpt":0.28618398190358435,"score_spread":0.27353312115463774,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2080021477","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.041214116,0.00053277554,0.95067835,0.00085484865,0.00024536476,0.000048938255,0.00036948142,0.004950363,0.0011058222],"genre_scores_gemma":[0.6208225,0.0007055994,0.36815932,0.0007382094,0.00035271203,0.0002038728,0.0018477602,0.0005697071,0.0066002174],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99899966,0.0003634437,0.00007176143,0.00028913346,0.00019015186,0.000085745596],"domain_scores_gemma":[0.9955349,0.0031523744,0.00016471831,0.00036681996,0.0006940379,0.00008717742],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015449558,0.0010419213,0.0010414574,0.0009710112,0.00062385935,0.0012320101,0.0013723687,0.001070307,0.0025594048],"category_scores_gemma":[0.0063201506,0.0006108979,0.001035394,0.0010586068,0.00044333297,0.0031078365,0.0011561458,0.0028906749,0.0029842588],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00051083614,0.0005441851,0.0029911834,0.00021304823,0.0003066881,0.00032008888,0.00017367292,0.35629624,0.022219064,0.007187629,0.010492849,0.59874445],"study_design_scores_gemma":[0.000006280524,0.000025400657,0.00012704825,0.0000029950518,0.000023035858,0.000021186966,0.0000069793077,0.99521905,0.001802307,0.0025037748,0.00025566123,0.000006274609],"about_ca_topic_score_codex":0.0064610806,"about_ca_topic_score_gemma":0.008669,"teacher_disagreement_score":0.0064610806,"about_ca_system_score_codex":0.0005678347,"about_ca_system_score_gemma":0.0013006817,"threshold_uncertainty_score":0.012846947},"labels":[],"label_agreement":null},{"id":"W2080390169","doi":"10.1111/0824-7935.00125","title":"Realizing Presuppositions in a Montague Grammar‐Like Fragment of English","year":2000,"lang":"en","type":"article","venue":"Computational Intelligence","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Presupposition; Linguistics; Sentence; Focus (optics); Projection (relational algebra); Semantics (computer science); Interpretation (philosophy); Computer science; Mathematics; Natural language processing; Artificial intelligence; Philosophy; Algorithm","score_opus":0.015172452451037542,"score_gpt":0.2850867162630434,"score_spread":0.26991426381200584,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2080390169","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.40058586,0.00014000923,0.5757954,0.0007573412,0.000041295058,0.00008609003,0.00027021323,0.0029686093,0.019355075],"genre_scores_gemma":[0.8988248,0.0000735558,0.09776496,0.0001713211,0.000023965444,0.000033979548,0.00045619922,0.00018332123,0.0024678826],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9995326,0.00016660152,0.000018154778,0.00014645101,0.000079478144,0.000056761317],"domain_scores_gemma":[0.9993222,0.00039014206,0.000055052387,0.000108873144,0.000087981265,0.0000357426],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006417772,0.00046179202,0.00036549574,0.00058969675,0.0010102316,0.0016785869,0.0009064599,0.00061174884,0.0026821499],"category_scores_gemma":[0.0021037923,0.00043683883,0.0009902404,0.0003662647,0.0021621357,0.0030122115,0.0012019118,0.0009421836,0.00046053852],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00063167687,0.00024116349,0.007697452,0.00020569336,0.00012813,0.002100304,0.009351718,0.041596003,0.081767105,0.71780866,0.004713186,0.13375898],"study_design_scores_gemma":[0.000083228006,0.00021513508,0.004111513,0.00004923677,0.00011234833,0.0005525366,0.0011354195,0.31811547,0.031795725,0.6311043,0.0126450425,0.00008000929],"about_ca_topic_score_codex":0.010186221,"about_ca_topic_score_gemma":0.012777261,"teacher_disagreement_score":0.010186221,"about_ca_system_score_codex":0.0010995362,"about_ca_system_score_gemma":0.0007861377,"threshold_uncertainty_score":0.020253897},"labels":[],"label_agreement":null},{"id":"W2081743236","doi":"10.1007/s10590-014-9166-8","title":"Introduction to special issue on post-editing","year":2014,"lang":"en","type":"article","venue":"Machine Translation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":31,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Computer science; Computational linguistics; Natural language processing; Artificial intelligence","score_opus":0.0073987869319434,"score_gpt":0.25830550127037893,"score_spread":0.25090671433843553,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2081743236","genre_codex":"editorial","genre_gemma":"editorial","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"editorial","genre_consensus":"editorial","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0012656979,0.023127476,0.07954898,0.03370966,0.7935155,0.00028003755,0.0020380346,0.00557165,0.060942892],"genre_scores_gemma":[0.009118924,0.020651951,0.03709964,0.013047982,0.60727155,0.0002521099,0.0058705476,0.007831653,0.2988557],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99701995,0.00054700184,0.0004219009,0.0006626668,0.0011397431,0.00020872668],"domain_scores_gemma":[0.98650926,0.0039019443,0.00066615245,0.0020892166,0.0056729573,0.0011604336],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030202845,0.002346087,0.0023513932,0.0046962546,0.0018936023,0.0052045193,0.0023303297,0.0029319238,0.16901264],"category_scores_gemma":[0.0125939585,0.0008125741,0.0015810453,0.0035486408,0.0010223653,0.0063362066,0.0029737188,0.0045594214,0.0948416],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005760478,0.000042055504,0.00008546298,0.00043197648,0.000021508813,0.00013634418,0.000036534206,0.00016057567,0.0012602708,0.0025230716,0.8803322,0.1149125],"study_design_scores_gemma":[0.000009940013,0.000056904453,0.00030020895,0.00015096346,0.000022184911,0.0004288148,0.000032487467,0.00071889465,0.001227556,0.005117513,0.9919045,0.0000300246],"about_ca_topic_score_codex":0.00089773274,"about_ca_topic_score_gemma":0.0019543003,"teacher_disagreement_score":0.16901264,"about_ca_system_score_codex":0.0009567514,"about_ca_system_score_gemma":0.0017862006,"threshold_uncertainty_score":0.5654036},"labels":[],"label_agreement":null},{"id":"W2082130990","doi":"10.1145/1631127.1631130","title":"Exploring fusion in a spontaneous speech retrieval task","year":2009,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Fuse (electrical); Computer science; Task (project management); Artificial intelligence; Fusion; Speech recognition; Natural language processing; Sensor fusion; Information retrieval; Linguistics; Engineering","score_opus":0.036481555911210994,"score_gpt":0.26759054652779535,"score_spread":0.23110899061658435,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2082130990","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.27067342,0.00093094114,0.7222887,0.00042993604,0.00007166102,0.00024238994,0.0003804145,0.0032084605,0.0017740749],"genre_scores_gemma":[0.8193762,0.0002771152,0.17732988,0.00014265566,0.0000771222,0.00019648386,0.0013364713,0.00029646556,0.00096760585],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9896722,0.0056617884,0.000569619,0.0017908566,0.0017960946,0.0005094434],"domain_scores_gemma":[0.98480475,0.010919891,0.0008942115,0.0016657612,0.0014570196,0.00025834344],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012839628,0.002006437,0.0015697627,0.0017100403,0.00077699125,0.0024051927,0.0016219197,0.0022233373,0.0013366867],"category_scores_gemma":[0.02806437,0.0006771932,0.001503338,0.0013959969,0.0010913304,0.005538213,0.002716327,0.0013560969,0.0009497576],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002755127,0.001326633,0.015553326,0.0011474211,0.001491095,0.0009635826,0.0035104456,0.15940824,0.18206243,0.0049127694,0.0030502477,0.62381864],"study_design_scores_gemma":[0.00014946345,0.001923364,0.013810112,0.00007188605,0.00049896707,0.0011255146,0.0012414618,0.8590434,0.09332236,0.025104243,0.0034506302,0.00025856364],"about_ca_topic_score_codex":0.0015108981,"about_ca_topic_score_gemma":0.0010378174,"teacher_disagreement_score":0.012839628,"about_ca_system_score_codex":0.00069735525,"about_ca_system_score_gemma":0.00086412684,"threshold_uncertainty_score":0.06790328},"labels":[],"label_agreement":null},{"id":"W2082229537","doi":"10.7202/019924ar","title":"A simple and robust method for extracting terminology","year":2009,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Simple (philosophy); Terminology; Process (computing); Lexicon; Natural language processing; Set (abstract data type); Porting; Artificial intelligence; Word (group theory); Domain (mathematical analysis); Context (archaeology); Information retrieval; Software; Linguistics; Programming language; Mathematics","score_opus":0.05540557279115226,"score_gpt":0.3390912007655686,"score_spread":0.2836856279744163,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2082229537","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.002488029,0.0005651328,0.98717934,0.0002121518,0.00025324477,0.0003333503,0.0011308662,0.0042829304,0.0035548992],"genre_scores_gemma":[0.010469849,0.0003411712,0.98176104,0.00010673695,0.000107858046,0.00029014223,0.0027557951,0.0006970739,0.0034703116],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9957765,0.0007612986,0.00072165934,0.001005113,0.0015555675,0.00018001041],"domain_scores_gemma":[0.9947431,0.0014556367,0.0004285137,0.0013054824,0.0019516832,0.00011554317],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023985615,0.00166768,0.0013548783,0.010546229,0.0017748128,0.0033445007,0.0018032403,0.0016382696,0.010483436],"category_scores_gemma":[0.010535377,0.000832112,0.0018977101,0.0059692906,0.0009951849,0.0042886725,0.0029978373,0.0018323994,0.01628519],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006622618,0.000060424187,0.001076677,0.0010858125,0.00011689464,0.00043810622,0.00082139793,0.0012991162,0.10239777,0.02781371,0.023867933,0.840956],"study_design_scores_gemma":[0.0001314553,0.0002681266,0.0060361084,0.0005592042,0.0003809294,0.004997126,0.0014831503,0.07255895,0.14687818,0.08851153,0.6776802,0.0005151269],"about_ca_topic_score_codex":0.0013352609,"about_ca_topic_score_gemma":0.0021500587,"teacher_disagreement_score":0.010546229,"about_ca_system_score_codex":0.0007154949,"about_ca_system_score_gemma":0.0030209236,"threshold_uncertainty_score":0.035070598},"labels":[],"label_agreement":null},{"id":"W2082475284","doi":"10.7202/003997ar","title":"La traduction automatique : l’ordinateur au service des traducteurs","year":2002,"lang":"fr","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Philosophy","score_opus":0.05776129454658851,"score_gpt":0.2869033927731876,"score_spread":0.2291420982265991,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2082475284","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.054713327,0.0060759163,0.81673515,0.010692038,0.0009520675,0.0009973524,0.00091909466,0.010679916,0.09823513],"genre_scores_gemma":[0.32571074,0.0039755325,0.5743919,0.0020193954,0.0003272257,0.0006680995,0.0017257064,0.0029511012,0.0882302],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.97013086,0.013736678,0.0024091578,0.0044094073,0.008026518,0.0012874096],"domain_scores_gemma":[0.95767164,0.015039702,0.0026462036,0.0133116525,0.00997616,0.0013546372],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.016541421,0.0015844731,0.0011246315,0.004794027,0.004628314,0.015632454,0.003448617,0.0036876723,0.017157031],"category_scores_gemma":[0.036361877,0.0011667336,0.0018556179,0.0066023446,0.00659498,0.015858559,0.0069025997,0.0044719493,0.012023539],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005750938,0.00031104757,0.009848218,0.00198396,0.00015590298,0.00078132446,0.035130784,0.003976762,0.015995527,0.1618439,0.020610103,0.74878734],"study_design_scores_gemma":[0.00011174298,0.00041776831,0.007415233,0.0018098635,0.0002096498,0.0019363079,0.019552166,0.019006554,0.03413183,0.07544715,0.839669,0.00029274428],"about_ca_topic_score_codex":0.017495437,"about_ca_topic_score_gemma":0.010195076,"teacher_disagreement_score":0.017495437,"about_ca_system_score_codex":0.004818427,"about_ca_system_score_gemma":0.010218315,"threshold_uncertainty_score":0.087480426},"labels":[],"label_agreement":null},{"id":"W2082597929","doi":"10.3115/1654449.1654479","title":"RALI","year":2005,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Machine translation; Phrase; Task (project management); Natural language processing; Translation (biology); Artificial intelligence; Quality (philosophy); Programming language; Engineering","score_opus":0.009836619167801024,"score_gpt":0.2696904957148943,"score_spread":0.25985387654709324,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2082597929","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0065215817,0.0025996533,0.03681002,0.00563405,0.0031722358,0.00030313927,0.007129622,0.0090898825,0.9287398],"genre_scores_gemma":[0.043450885,0.0022514928,0.026986782,0.0022388878,0.0006328322,0.00023896301,0.011521872,0.0017855946,0.91089267],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9986772,0.00021735074,0.00008373211,0.00031291822,0.0005331972,0.00017563166],"domain_scores_gemma":[0.9981804,0.00019555053,0.00010364021,0.0005055358,0.00077950826,0.00023537112],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012489496,0.001006008,0.0006971097,0.002476988,0.0019532747,0.0040123602,0.001485467,0.0016201802,0.32058606],"category_scores_gemma":[0.0042523737,0.00041622578,0.00063015276,0.0016170756,0.000704501,0.0025842525,0.0032894723,0.0019716225,0.3125958],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00026995013,0.00013805283,0.0021856986,0.00048470387,0.000035859586,0.0005398372,0.00064990856,0.00072896195,0.0095216455,0.084428795,0.38922817,0.5117883],"study_design_scores_gemma":[0.000012681268,0.000027897155,0.0005747128,0.000071521674,0.000009863982,0.00036497574,0.0001379091,0.00077404024,0.0026886677,0.0052732886,0.9900462,0.000018185601],"about_ca_topic_score_codex":0.003954049,"about_ca_topic_score_gemma":0.0058067176,"teacher_disagreement_score":0.32058606,"about_ca_system_score_codex":0.0016158666,"about_ca_system_score_gemma":0.0018425076,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W2082718666","doi":"10.1145/1871437.1871582","title":"Clickthrough-based translation models for web search","year":2010,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":82,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Phrase; Machine translation; Natural language processing; Artificial intelligence; Information retrieval; Language model; Word (group theory); Cross-language information retrieval; Translation (biology); Set (abstract data type); Linguistics","score_opus":0.04394242775007334,"score_gpt":0.32880174591792816,"score_spread":0.2848593181678548,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2082718666","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09609138,0.0023655717,0.8896952,0.0014765752,0.00015144877,0.00015194196,0.0013936497,0.002719598,0.005954586],"genre_scores_gemma":[0.8810494,0.0016014022,0.10030675,0.0002840378,0.00025381127,0.00042600755,0.0027576082,0.00043168533,0.012889225],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986721,0.0007185222,0.00009839299,0.00022445316,0.00019224147,0.00009432413],"domain_scores_gemma":[0.9936971,0.004741621,0.00048234157,0.00043339963,0.0005632651,0.00008230263],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0042322366,0.0009978799,0.0010709047,0.0017354599,0.0006583347,0.0015913267,0.0012888329,0.0016259174,0.0036190243],"category_scores_gemma":[0.013724489,0.0007116548,0.0012039292,0.0023561453,0.0009360237,0.004207593,0.0007216044,0.0015764319,0.0019921341],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00082924543,0.00040649323,0.006897635,0.0003526525,0.00028253006,0.00039218547,0.0006861713,0.67469317,0.004609782,0.112113394,0.008717823,0.19001895],"study_design_scores_gemma":[0.000026907903,0.00004838146,0.0006386213,0.000010446823,0.000028352406,0.00007867035,0.000017167862,0.9647127,0.0005238837,0.032805495,0.001089005,0.0000203702],"about_ca_topic_score_codex":0.0119889565,"about_ca_topic_score_gemma":0.011515154,"teacher_disagreement_score":0.0119889565,"about_ca_system_score_codex":0.001785632,"about_ca_system_score_gemma":0.0010242108,"threshold_uncertainty_score":0.023838341},"labels":[],"label_agreement":null},{"id":"W2083136324","doi":"10.7202/012243ar","title":"Apport du Web dans la reconnaissance des entités nommées","year":2006,"lang":"fr","type":"article","venue":"Revue québécoise de linguistique","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Philosophy; Political science","score_opus":0.014804635224592253,"score_gpt":0.267685861946196,"score_spread":0.25288122672160374,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2083136324","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4766944,0.004436687,0.47610205,0.0011553442,0.00042225144,0.00059095764,0.0019418602,0.020409359,0.018247047],"genre_scores_gemma":[0.6024743,0.0014777718,0.36624357,0.00034223325,0.000110647954,0.00021927443,0.0027742384,0.0012926075,0.025065428],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.997656,0.0006410797,0.00018610033,0.00047356682,0.0009137773,0.00012952344],"domain_scores_gemma":[0.9917361,0.0048209243,0.0004900335,0.0013858847,0.001371217,0.0001958072],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030033018,0.0009567902,0.0008938054,0.0020794254,0.0012969015,0.0024846403,0.0009241874,0.0013018791,0.0045560487],"category_scores_gemma":[0.010174905,0.0005231171,0.00097519293,0.001457,0.0006401235,0.0039507095,0.0014540185,0.00096322113,0.0020747345],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013947978,0.000281839,0.035878196,0.0013478711,0.00040440122,0.0028668344,0.0074167605,0.012202698,0.13522385,0.008013605,0.012121398,0.7828478],"study_design_scores_gemma":[0.00012533614,0.0007988636,0.062057376,0.0006386437,0.0006648974,0.0056690634,0.005863876,0.23241359,0.34168836,0.012397361,0.33726928,0.0004133851],"about_ca_topic_score_codex":0.010125091,"about_ca_topic_score_gemma":0.013564613,"teacher_disagreement_score":0.010125091,"about_ca_system_score_codex":0.0006315398,"about_ca_system_score_gemma":0.00080418744,"threshold_uncertainty_score":0.020132303},"labels":[],"label_agreement":null},{"id":"W2083443881","doi":"10.7202/012249ar","title":"Procédures de désambiguïsation pour les systèmes de recherche d’information","year":2006,"lang":"fr","type":"article","venue":"Revue québécoise de linguistique","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Philosophy; Political science","score_opus":0.04943564259378552,"score_gpt":0.33324422165344875,"score_spread":0.2838085790596632,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2083443881","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03476814,0.00064833934,0.9353449,0.0011327168,0.00028679115,0.00047407838,0.0004668631,0.019652715,0.0072254576],"genre_scores_gemma":[0.11432909,0.0004959759,0.8694822,0.00055646827,0.00009618215,0.00042108164,0.0007417096,0.0029557059,0.010921579],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9915937,0.0029535934,0.0008911168,0.0018486507,0.00246362,0.00024925236],"domain_scores_gemma":[0.96441144,0.021361612,0.0015571801,0.006246787,0.0060729138,0.0003500414],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008658913,0.0016749677,0.0015594913,0.0021552057,0.003168674,0.0051709167,0.0025868274,0.0023648299,0.015166405],"category_scores_gemma":[0.040028527,0.0015554286,0.0013597641,0.0017473001,0.0029190665,0.00635129,0.0029929,0.0028616046,0.0075509534],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015059749,0.00024700622,0.008502343,0.0019110177,0.00019371811,0.0013119935,0.018128075,0.0042238426,0.13121171,0.05466825,0.014738702,0.7633574],"study_design_scores_gemma":[0.00031538657,0.00096111407,0.010897279,0.0005901847,0.00033291642,0.005034147,0.007914758,0.11090084,0.47473252,0.051556237,0.33620912,0.0005554953],"about_ca_topic_score_codex":0.0067902794,"about_ca_topic_score_gemma":0.007786778,"teacher_disagreement_score":0.015166405,"about_ca_system_score_codex":0.0015835017,"about_ca_system_score_gemma":0.0029061907,"threshold_uncertainty_score":0.050736666},"labels":[],"label_agreement":null},{"id":"W2084011220","doi":"10.7202/003026ar","title":"Terminological Difficulties in Dene Language Interpretation and Translation","year":2002,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Terminology; Interpreter; Computer science; Interpretation (philosophy); Field (mathematics); Linguistics; Natural language processing; Artificial intelligence; Programming language; Mathematics","score_opus":0.037954634858631785,"score_gpt":0.27410615745252714,"score_spread":0.23615152259389535,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2084011220","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.028547838,0.012880259,0.8285691,0.020251669,0.0015824906,0.00029237673,0.0003235229,0.0008212476,0.10673155],"genre_scores_gemma":[0.3910016,0.009329058,0.57143575,0.0033262577,0.0009651165,0.0005860012,0.00075200776,0.0011327323,0.02147152],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"qualitative","domain_scores_codex":[0.9570626,0.027025228,0.0046144244,0.0022883986,0.008061834,0.0009474651],"domain_scores_gemma":[0.95043784,0.032104712,0.00317798,0.0057414547,0.008197189,0.00034083446],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.030939225,0.0006118463,0.0011881075,0.0047464026,0.0056843003,0.0140534425,0.0027605088,0.0033759882,0.00272276],"category_scores_gemma":[0.057680298,0.001066418,0.00076764455,0.006235373,0.017133672,0.019577624,0.008768451,0.007469232,0.00147574],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00004945322,0.000015909853,0.00061971793,0.00027473658,0.000015793152,0.00043706663,0.018968154,0.00089387724,0.00085380586,0.9248336,0.0037565988,0.049281277],"study_design_scores_gemma":[0.000018957664,0.000019492112,0.00052561116,0.000598019,0.000020696558,0.00095471746,0.011823094,0.004966994,0.0031678358,0.78870004,0.1891347,0.0000699332],"about_ca_topic_score_codex":0.005638491,"about_ca_topic_score_gemma":0.0071256207,"teacher_disagreement_score":0.030939225,"about_ca_system_score_codex":0.0045913416,"about_ca_system_score_gemma":0.005592605,"threshold_uncertainty_score":0.16362423},"labels":[],"label_agreement":null},{"id":"W2084189120","doi":"10.1016/j.mcm.2006.06.001","title":"A formal approach to subgrammar extraction for NLP","year":2006,"lang":"en","type":"article","venue":"Mathematical and Computer Modelling","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Computer science; Parsing; Set (abstract data type); Context (archaeology); Artificial intelligence; Rule-based machine translation; Context-free grammar; Grammar; Algorithm; Natural language processing; Theoretical computer science; Programming language; Linguistics","score_opus":0.022930478442257204,"score_gpt":0.2505988035083456,"score_spread":0.22766832506608842,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2084189120","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00039624362,0.00009951834,0.9972453,0.00029041574,0.000038858787,0.0000537477,0.00010937407,0.0005701602,0.0011964448],"genre_scores_gemma":[0.028117975,0.0003177628,0.96829426,0.0002193426,0.00014821348,0.00021337431,0.00051508687,0.00032300182,0.0018510167],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99495476,0.0015398164,0.00083732576,0.00079586916,0.0015974144,0.00027478443],"domain_scores_gemma":[0.9925222,0.0034410502,0.00042034884,0.001971402,0.0014387424,0.00020622725],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0055178027,0.0009224944,0.00111664,0.00364812,0.0021950337,0.007351079,0.0044667386,0.0021274965,0.009497858],"category_scores_gemma":[0.011673266,0.0021551454,0.0041624317,0.0026934533,0.0060544447,0.010688871,0.005151621,0.005309102,0.0036653362],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000026641916,0.00005535099,0.00017498618,0.00030020616,0.000052225354,0.00023238189,0.00046538384,0.0077147977,0.002535016,0.93607825,0.0034867343,0.048878092],"study_design_scores_gemma":[0.000023925737,0.0000288085,0.00013641312,0.0001420905,0.000069500195,0.000355333,0.00014743982,0.07483923,0.0049575213,0.87380797,0.045429844,0.00006189522],"about_ca_topic_score_codex":0.0035548867,"about_ca_topic_score_gemma":0.0061366116,"teacher_disagreement_score":0.009497858,"about_ca_system_score_codex":0.0021972624,"about_ca_system_score_gemma":0.0028705904,"threshold_uncertainty_score":0.031773448},"labels":[],"label_agreement":null},{"id":"W2084423341","doi":"10.1121/1.3588069","title":"Word duration and segment deletion as measures of reduction in a corpus of spontaneous speech.","year":2011,"lang":"en","type":"article","venue":"The Journal of the Acoustical Society of America","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Duration (music); Word (group theory); Syllable; Context (archaeology); Predictability; Word lists by frequency; Speech recognition; Computer science; Linguistics; Mathematics; Statistics; Natural language processing; History; Sentence","score_opus":0.017942154977162835,"score_gpt":0.24650688706483836,"score_spread":0.22856473208767553,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2084423341","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98909223,0.0005601526,0.0064878087,0.000026618158,0.0000278251,0.000102496015,0.0017045144,0.000074709445,0.0019236488],"genre_scores_gemma":[0.97913134,0.00036459992,0.012993877,0.000030831477,0.00003836015,0.00050347473,0.0058284635,0.00008781932,0.0010212224],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9973289,0.0009811594,0.00030023145,0.00060380384,0.00071604975,0.00006985548],"domain_scores_gemma":[0.9823386,0.011838883,0.0024575118,0.0017800655,0.0012325428,0.00035237047],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027514936,0.00031307872,0.00037709452,0.0017500235,0.0003480978,0.00088870584,0.00041195806,0.0003900736,0.0014918668],"category_scores_gemma":[0.020854633,0.00024638948,0.0002427806,0.0019698734,0.00061337865,0.00077697565,0.000749834,0.0005423139,0.00046597066],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.006110377,0.0008729179,0.45928755,0.0018972703,0.00089129544,0.0005831861,0.019141236,0.0041449266,0.28392303,0.0019751466,0.0023467017,0.2188264],"study_design_scores_gemma":[0.00002580636,0.00068717764,0.9808101,0.000029098193,0.00010314858,0.0007581393,0.0010497773,0.0036086368,0.009580475,0.00040355886,0.0029009073,0.000043219825],"about_ca_topic_score_codex":0.0013748497,"about_ca_topic_score_gemma":0.0023330338,"teacher_disagreement_score":0.0027514936,"about_ca_system_score_codex":0.00033615422,"about_ca_system_score_gemma":0.000217442,"threshold_uncertainty_score":0.01455152},"labels":[],"label_agreement":null},{"id":"W2084815925","doi":"10.1080/0950236042000183250","title":"The Oulipo factor: the procedural poetics of Christian Bök and Caroline Bergvall","year":2004,"lang":"en","type":"article","venue":"Textual Practice","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Poetics; Poetry; Theme (computing); Literature; Repetition (rhetorical device); Philosophy; Art; Linguistics; Computer science","score_opus":0.013701449567984983,"score_gpt":0.29413315666632384,"score_spread":0.28043170709833887,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2084815925","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16872647,0.020945586,0.0057318187,0.051050846,0.0017223464,0.0000747126,0.00013251635,0.00011325226,0.7515024],"genre_scores_gemma":[0.9308377,0.0030098953,0.0012986433,0.0024759339,0.00017364365,0.00003134886,0.000031370513,0.00013901271,0.062002447],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99903536,0.00033719334,0.000017626231,0.00013711565,0.00023256478,0.0002401876],"domain_scores_gemma":[0.9993623,0.00029962728,0.00004993645,0.000034535744,0.00012287582,0.00013082607],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012141983,0.00035300726,0.00024948217,0.0010207843,0.008769952,0.005595667,0.00055173377,0.0012115395,0.0032565284],"category_scores_gemma":[0.0027936406,0.00015850103,0.0001052324,0.0008489985,0.016271377,0.0026590468,0.0018800327,0.0029784106,0.0005131119],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000102148355,0.000028760023,0.0007821819,0.000108273496,0.0000059843333,0.00031156026,0.14680773,0.00019661631,0.0006779723,0.8086915,0.024681976,0.017605357],"study_design_scores_gemma":[0.0000121338235,0.000023551414,0.0026968182,0.00029175356,0.000006651387,0.00035457494,0.06895872,0.0002759576,0.00040938088,0.022813091,0.9041123,0.000045091696],"about_ca_topic_score_codex":0.11651082,"about_ca_topic_score_gemma":0.19213074,"teacher_disagreement_score":0.11651082,"about_ca_system_score_codex":0.010044903,"about_ca_system_score_gemma":0.004987726,"threshold_uncertainty_score":0.23166525},"labels":[],"label_agreement":null},{"id":"W2084942640","doi":"10.1145/2072221.2072253","title":"A phonetic approach to handling spelling variations in medieval documents","year":2011,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Athabasca University","funders":"","keywords":"Spelling; Orthography; German; Computer science; Natural language processing; Linguistics; Artificial intelligence; Process (computing); Comprehension; Software; Programming language","score_opus":0.030248376610475376,"score_gpt":0.26760593803288046,"score_spread":0.23735756142240508,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2084942640","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018621866,0.0005256422,0.9663216,0.00029228756,0.000155091,0.000112995534,0.001064894,0.0072166193,0.00568903],"genre_scores_gemma":[0.10118981,0.00059999304,0.8921897,0.000084993626,0.000098139724,0.0000845231,0.0010874334,0.00061989727,0.0040455298],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987538,0.00020954358,0.00016759489,0.00048139968,0.00031902862,0.00006857927],"domain_scores_gemma":[0.9983594,0.0005926427,0.00018916371,0.00039190694,0.00038525835,0.00008163958],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006464287,0.000988547,0.0004880797,0.0029575173,0.001320462,0.0034905442,0.0009536916,0.00074901985,0.007936522],"category_scores_gemma":[0.0042178463,0.00051312946,0.00080755743,0.0026010107,0.0012089176,0.0020566385,0.0018254885,0.0014237956,0.003755451],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015964093,0.000046350935,0.0033428702,0.00043859016,0.00007927725,0.00045141726,0.002497743,0.0033142013,0.05901225,0.013065686,0.004201986,0.9133899],"study_design_scores_gemma":[0.00009724518,0.0004521547,0.04489838,0.0004700761,0.0004154023,0.0075274147,0.007937724,0.16959837,0.24848111,0.08671073,0.43297234,0.00043906618],"about_ca_topic_score_codex":0.003875958,"about_ca_topic_score_gemma":0.0057682125,"teacher_disagreement_score":0.007936522,"about_ca_system_score_codex":0.0006479807,"about_ca_system_score_gemma":0.0015901529,"threshold_uncertainty_score":0.026550233},"labels":[],"label_agreement":null},{"id":"W2085851735","doi":"10.1145/2641483.2641527","title":"Classification and Generation of Grammatical Errors","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Syntax; Grammar induction; Sentence; Grammar; Natural language; Feature (linguistics); Rule-based machine translation; Linguistics","score_opus":0.06160801234603632,"score_gpt":0.2905635601397072,"score_spread":0.22895554779367086,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2085851735","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8581963,0.0004920701,0.12763682,0.00053908286,0.00018235171,0.0003415077,0.0033638834,0.0030318438,0.0062161745],"genre_scores_gemma":[0.88748825,0.0002605648,0.10536416,0.00010168495,0.000063718864,0.00015576732,0.004055756,0.00028601591,0.0022241108],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99636394,0.0012079636,0.00034966136,0.00081104366,0.0011166079,0.00015084253],"domain_scores_gemma":[0.97625095,0.012945711,0.0040447167,0.002204147,0.004281825,0.000272681],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002046837,0.000472231,0.00054832216,0.0034871653,0.0005480053,0.0012258027,0.00075473427,0.0007099092,0.0014377396],"category_scores_gemma":[0.021968039,0.00017828422,0.00043022458,0.0013635299,0.0005021239,0.0012018237,0.0006824567,0.0006397311,0.000826014],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00059268955,0.00029899733,0.24331601,0.00048325935,0.00014746831,0.0014169621,0.0019644438,0.014128634,0.028353196,0.00675677,0.008292902,0.6942487],"study_design_scores_gemma":[0.00007166624,0.0005813517,0.28407153,0.00026089148,0.00025130349,0.004372614,0.0018107732,0.5506809,0.11556039,0.01955185,0.022615634,0.00017107445],"about_ca_topic_score_codex":0.0012421085,"about_ca_topic_score_gemma":0.0014737605,"teacher_disagreement_score":0.0034871653,"about_ca_system_score_codex":0.0006538324,"about_ca_system_score_gemma":0.0008653281,"threshold_uncertainty_score":0.010824859},"labels":[],"label_agreement":null},{"id":"W2086747158","doi":"10.1111/1467-9612.00057","title":"Anaphoric R–Expressions as Bound Variables","year":2003,"lang":"en","type":"article","venue":"Syntax","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":100,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Upper and lower bounds; Linguistics; Mathematics; Crossover; Combinatorics; Computer science; Philosophy; Artificial intelligence","score_opus":0.00929937327015266,"score_gpt":0.2624024794438426,"score_spread":0.2531031061736899,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2086747158","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6925365,0.0011646176,0.17684509,0.0032012847,0.00013466906,0.00009791656,0.0005856582,0.0015891239,0.1238451],"genre_scores_gemma":[0.9826135,0.00024152249,0.012880205,0.00013834555,0.000028224764,0.000026111456,0.00014310362,0.00017819133,0.0037508241],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99888426,0.00043734568,0.00008756352,0.0002626208,0.00018256751,0.00014565689],"domain_scores_gemma":[0.99825245,0.0005864928,0.0004687622,0.00045793215,0.00019188452,0.000042477175],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012173506,0.0003479236,0.00042488115,0.001076005,0.002012696,0.004222706,0.0007639149,0.00095929025,0.004130338],"category_scores_gemma":[0.0027311423,0.0007333705,0.00042439034,0.0016188987,0.004473382,0.006661247,0.0033597082,0.0018306874,0.00048669335],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014744578,0.00001879127,0.003575045,0.00018532068,0.000028291397,0.0007707863,0.010078975,0.00058353046,0.016853176,0.9464182,0.0015264532,0.01981392],"study_design_scores_gemma":[0.00017721734,0.0001078537,0.018115869,0.00022701811,0.00036316627,0.003669251,0.016643623,0.014665817,0.045288503,0.6849,0.21559386,0.00024773207],"about_ca_topic_score_codex":0.0054451595,"about_ca_topic_score_gemma":0.0049369778,"teacher_disagreement_score":0.0054451595,"about_ca_system_score_codex":0.0021027285,"about_ca_system_score_gemma":0.0010832946,"threshold_uncertainty_score":0.0152564645},"labels":[],"label_agreement":null},{"id":"W2086873147","doi":"10.7202/003073ar","title":"Outils linguistiques et système terminologique multilingue","year":2002,"lang":"fr","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Political science; Philosophy","score_opus":0.08085489410066951,"score_gpt":0.3211662302287214,"score_spread":0.24031133612805192,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2086873147","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.039919198,0.0026674238,0.8976931,0.0007128392,0.00034379907,0.00032263133,0.0050880252,0.03851589,0.014737069],"genre_scores_gemma":[0.13852492,0.0019962904,0.82086885,0.00034860976,0.0001946443,0.0003911685,0.015587152,0.0038787078,0.018209612],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99630475,0.001056776,0.00042179012,0.0008118498,0.0012395871,0.00016532745],"domain_scores_gemma":[0.9924124,0.0044317967,0.00026230648,0.0010858374,0.0016579481,0.00014980347],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0044842726,0.0013518436,0.0014139754,0.0036668396,0.0013887698,0.0050484454,0.0012707436,0.0013477042,0.010381267],"category_scores_gemma":[0.012096077,0.0008507655,0.0014118657,0.0024337233,0.0011940941,0.004365807,0.0019972678,0.0019031888,0.007691879],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001238866,0.00024856156,0.007996764,0.002381401,0.0003743691,0.0013158151,0.0051594735,0.00800951,0.15743895,0.03312229,0.022871448,0.75984263],"study_design_scores_gemma":[0.00018421345,0.00032718494,0.006945493,0.00061564817,0.0005747652,0.0035875835,0.0022225925,0.08347506,0.30713475,0.021438519,0.57316697,0.00032720336],"about_ca_topic_score_codex":0.004763444,"about_ca_topic_score_gemma":0.0041543553,"teacher_disagreement_score":0.010381267,"about_ca_system_score_codex":0.0010142371,"about_ca_system_score_gemma":0.0018647929,"threshold_uncertainty_score":0.034728825},"labels":[],"label_agreement":null},{"id":"W2087551996","doi":"10.3115/1220355.1220387","title":"Symmetric word alignments for statistical machine translation","year":2004,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":72,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Machine translation; Computer science; Word (group theory); Sentence; Task (project management); Word error rate; Artificial intelligence; Natural language processing; Translation (biology); Speech recognition; Graph; Baseline (sea); IBM; Theoretical computer science; Mathematics","score_opus":0.019927652461188383,"score_gpt":0.2999664075396394,"score_spread":0.280038755078451,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2087551996","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0011300306,0.0007485387,0.99491286,0.00017662135,0.00011271537,0.000048245685,0.00017475855,0.0014025039,0.0012938073],"genre_scores_gemma":[0.075020134,0.0020652406,0.9155738,0.00027973272,0.000509864,0.00042768382,0.0019927525,0.0014565997,0.002674202],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99402905,0.0033524646,0.0004141301,0.00096782425,0.0011106652,0.00012589019],"domain_scores_gemma":[0.990737,0.0048034345,0.0009444859,0.002635069,0.0007680425,0.000111980335],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0040013306,0.0015399915,0.0014778691,0.0022902633,0.001107361,0.0022045486,0.001925061,0.0016743236,0.009126577],"category_scores_gemma":[0.019662905,0.0009864409,0.0010613716,0.0037175333,0.0013436582,0.004557375,0.0027630765,0.0025224881,0.008663965],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002081455,0.00008627362,0.0005577742,0.0006155874,0.00019276158,0.0003083781,0.00018936867,0.09412962,0.00904669,0.33875552,0.017960418,0.53794956],"study_design_scores_gemma":[0.000040428702,0.00006120869,0.0002272585,0.000084513915,0.000036108613,0.000248233,0.00004410378,0.32890016,0.0059808725,0.6420945,0.022240108,0.000042511867],"about_ca_topic_score_codex":0.00089129223,"about_ca_topic_score_gemma":0.0012685801,"teacher_disagreement_score":0.009126577,"about_ca_system_score_codex":0.0008960245,"about_ca_system_score_gemma":0.0013400976,"threshold_uncertainty_score":0.030531406},"labels":[],"label_agreement":null},{"id":"W2089015446","doi":"10.7202/039601ar","title":"Propriétés transformationnelles unaires en lexicographie informatique","year":2010,"lang":"fr","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Philosophy","score_opus":0.02491599160449917,"score_gpt":0.28392985366952656,"score_spread":0.2590138620650274,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2089015446","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.054826777,0.008291493,0.80733174,0.0030999065,0.00065320777,0.00022746644,0.0026587679,0.0025057523,0.12040494],"genre_scores_gemma":[0.49107137,0.010345736,0.44863153,0.0007663178,0.0004212689,0.00031500202,0.004179888,0.0012412509,0.043027636],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9959197,0.0015199065,0.00064981636,0.00070946134,0.0009886579,0.0002124699],"domain_scores_gemma":[0.9943289,0.0029985395,0.0003247676,0.0011824388,0.0010639005,0.00010144663],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033349681,0.0010730583,0.0007053683,0.0045810016,0.001699621,0.010087691,0.00071561424,0.0011016064,0.012542734],"category_scores_gemma":[0.0077926354,0.0010916241,0.0012451806,0.005629926,0.0063234456,0.012950293,0.0023285223,0.003260439,0.004344003],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000113202135,0.000017378163,0.0025256786,0.00044551212,0.00004160647,0.00035761716,0.0041793254,0.0014080711,0.0040179673,0.8643182,0.002501301,0.12007418],"study_design_scores_gemma":[0.00004498419,0.0001281842,0.0052633444,0.0007912613,0.00010258239,0.002669579,0.0058156024,0.012434038,0.014981499,0.47895524,0.47865912,0.0001545184],"about_ca_topic_score_codex":0.0068052653,"about_ca_topic_score_gemma":0.0056174393,"teacher_disagreement_score":0.012542734,"about_ca_system_score_codex":0.0025582497,"about_ca_system_score_gemma":0.0021907266,"threshold_uncertainty_score":0.041959584},"labels":[],"label_agreement":null},{"id":"W2089601028","doi":"10.2316/journal.206.2005.3.206-2840","title":"Introduction of Logic in Language Modelling: The Minimum Perplexity Criterion","year":2005,"lang":"en","type":"article","venue":"International Journal of Robotics and Automation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Perplexity; Computer science; Programming language; Natural language processing; Language model","score_opus":0.015418889910876346,"score_gpt":0.28991528264987904,"score_spread":0.27449639273900267,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2089601028","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.002968041,0.0016255893,0.9897001,0.0017797702,0.00012764282,0.000050071052,0.00015824502,0.00033760158,0.0032530257],"genre_scores_gemma":[0.24631956,0.0042632036,0.7377862,0.001970182,0.0020614383,0.00056596345,0.00069810124,0.00079039246,0.005545041],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99227,0.0036996116,0.0006254864,0.001686323,0.0013886621,0.0003297975],"domain_scores_gemma":[0.97506136,0.020117104,0.00072964065,0.0018975483,0.0017399817,0.0004543821],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008962338,0.001969668,0.0025321457,0.0031691738,0.0012793173,0.0052737664,0.003900506,0.0033645544,0.004721008],"category_scores_gemma":[0.031973597,0.0017574464,0.0029750848,0.0031534906,0.0065718875,0.016202651,0.00560759,0.007157643,0.001697252],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023115893,0.000052232626,0.00048352015,0.0005610061,0.00013103496,0.00021847026,0.0005198646,0.04061532,0.0031064614,0.85233986,0.0038469767,0.0978941],"study_design_scores_gemma":[0.000015826201,0.000036123347,0.00006973397,0.00006401627,0.000031049705,0.0000911244,0.000023312845,0.08588939,0.0011768192,0.908206,0.0043637785,0.000032774733],"about_ca_topic_score_codex":0.0013973755,"about_ca_topic_score_gemma":0.001019341,"teacher_disagreement_score":0.008962338,"about_ca_system_score_codex":0.0028565964,"about_ca_system_score_gemma":0.001688613,"threshold_uncertainty_score":0.04739791},"labels":[],"label_agreement":null},{"id":"W2089679262","doi":"10.1073/pnas.1204678110","title":"Automated reconstruction of ancient languages using probabilistic models of sound change","year":2013,"lang":"en","type":"article","venue":"Proceedings of the National Academy of Sciences","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":134,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Sound change; Computer science; Probabilistic logic; Austronesian languages; Comparative method; Set (abstract data type); Process (computing); Inference; Function (biology); Artificial intelligence; Natural language processing; Linguistics; Programming language","score_opus":0.07284500180176527,"score_gpt":0.32732780739977807,"score_spread":0.2544828055980128,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2089679262","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.26339558,0.0002764619,0.72586966,0.00034305637,0.000043124423,0.00006872519,0.00052798213,0.0069045816,0.0025708345],"genre_scores_gemma":[0.7195113,0.00013142024,0.27653468,0.000088194014,0.00003050169,0.00006355395,0.0012879042,0.0006767838,0.001675679],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986663,0.00047599184,0.000072022856,0.00045551333,0.00022595201,0.0001041978],"domain_scores_gemma":[0.99556375,0.002802499,0.0003470331,0.0007571363,0.00042526601,0.00010440603],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002301541,0.00093292975,0.00087408075,0.0024237828,0.0010520151,0.001981238,0.0014587229,0.0010959274,0.0021954023],"category_scores_gemma":[0.009662799,0.0011165742,0.0015682956,0.0013897894,0.0012148955,0.002709598,0.0018330128,0.0015471508,0.0008097856],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006767451,0.0001387998,0.030451052,0.00023453317,0.00031744136,0.0009936845,0.003088018,0.48663053,0.045074582,0.02050344,0.004454528,0.40743664],"study_design_scores_gemma":[0.000022831298,0.000027404145,0.002310867,0.000012096806,0.000026042477,0.00014765441,0.00022643224,0.9774648,0.0045719184,0.013511414,0.0016471199,0.000031443644],"about_ca_topic_score_codex":0.008975032,"about_ca_topic_score_gemma":0.011844473,"teacher_disagreement_score":0.008975032,"about_ca_system_score_codex":0.0012047258,"about_ca_system_score_gemma":0.0012323967,"threshold_uncertainty_score":0.017845571},"labels":[],"label_agreement":null},{"id":"W2090445977","doi":"10.1093/jos/fft012","title":"Beyond Demonstratives: Direct Reference in Perceptually Grounded Descriptions","year":2013,"lang":"en","type":"article","venue":"Journal of Semantics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Linguistics; Philosophy","score_opus":0.021156855896456005,"score_gpt":0.26698930372535107,"score_spread":0.24583244782889507,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2090445977","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.26588994,0.0031965387,0.5353313,0.004376281,0.00022305574,0.00014239535,0.0003133455,0.00066964945,0.18985756],"genre_scores_gemma":[0.97749114,0.00044251728,0.017922617,0.00022528833,0.00005631822,0.00003575941,0.00010319932,0.00012582703,0.00359735],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99313873,0.003949217,0.00040121624,0.0010608885,0.0011454443,0.00030452295],"domain_scores_gemma":[0.9905004,0.0052057765,0.00075573934,0.002191367,0.0011263174,0.00022043045],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005366105,0.0004855655,0.0005482553,0.0017340735,0.0020529325,0.005066157,0.0014604811,0.0018235329,0.00447706],"category_scores_gemma":[0.01528038,0.00055760867,0.00056964887,0.001579536,0.009633061,0.025329633,0.0073865536,0.0023313344,0.00064786675],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000397985,0.000010602873,0.0003357267,0.00010499918,0.0000085997435,0.00025953143,0.018689588,0.00023214181,0.0022665013,0.96631324,0.00026925583,0.011469924],"study_design_scores_gemma":[0.00003199101,0.00006747484,0.00085239066,0.00010501971,0.000029717521,0.00057031156,0.010468588,0.0022619017,0.004900781,0.95300823,0.027650207,0.000053335414],"about_ca_topic_score_codex":0.0019906436,"about_ca_topic_score_gemma":0.0013329065,"teacher_disagreement_score":0.005366105,"about_ca_system_score_codex":0.0014545047,"about_ca_system_score_gemma":0.00090131315,"threshold_uncertainty_score":0.028379023},"labels":[],"label_agreement":null},{"id":"W2091014931","doi":"10.1145/1177055.1177057","title":"Confidence estimation for NLP applications","year":2006,"lang":"en","type":"article","venue":"ACM Transactions on Speech and Language Processing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":30,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada; Université de Montréal","funders":"","keywords":"Computer science; Machine translation; Artificial intelligence; Natural language processing; Estimation; Confidence interval; Natural language; Machine learning; Statistics; Mathematics","score_opus":0.00987436474930131,"score_gpt":0.28429790564261975,"score_spread":0.27442354089331844,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2091014931","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0023728015,0.0010895975,0.99248123,0.0004443128,0.000075426826,0.000057901063,0.0002662093,0.0011505857,0.002061976],"genre_scores_gemma":[0.32846558,0.002281546,0.6604008,0.0007382846,0.0010192234,0.0006410613,0.0024488322,0.0011918071,0.002812807],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.97368944,0.011618336,0.0022348706,0.0032226054,0.008601945,0.0006328522],"domain_scores_gemma":[0.83602744,0.13370478,0.0056683305,0.010809118,0.012843124,0.0009473805],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.016697397,0.0013999122,0.0018444652,0.0052940473,0.0012853,0.0054213847,0.00334534,0.003020306,0.0074038934],"category_scores_gemma":[0.22055812,0.00100757,0.0012833207,0.0048166686,0.002120306,0.007606319,0.004191077,0.0049284445,0.003392886],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00049091625,0.00014402978,0.006720331,0.0012487725,0.00024209241,0.00048770805,0.00056192116,0.12106341,0.005415148,0.25226086,0.020061241,0.59130347],"study_design_scores_gemma":[0.00004395898,0.00008241483,0.0013549677,0.00024058777,0.000055538098,0.00042134442,0.000104675375,0.65002495,0.0066668633,0.32409078,0.016835846,0.00007799829],"about_ca_topic_score_codex":0.0023666092,"about_ca_topic_score_gemma":0.0010898344,"teacher_disagreement_score":0.016697397,"about_ca_system_score_codex":0.0014133352,"about_ca_system_score_gemma":0.0017412083,"threshold_uncertainty_score":0.088305295},"labels":[],"label_agreement":null},{"id":"W2091480242","doi":"10.1111/0824-7935.00122","title":"Introduction To the Special Issue on the 1999 Pacific Association for Computational Linguistics Conference","year":2000,"lang":"en","type":"article","venue":"Computational Intelligence","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Library science; Citation; Computational linguistics; Association (psychology); Computer science; Linguistics; Information retrieval; Artificial intelligence; Philosophy; Epistemology","score_opus":0.02423773006766811,"score_gpt":0.30050979031448505,"score_spread":0.27627206024681694,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2091480242","genre_codex":"editorial","genre_gemma":"editorial","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"editorial","genre_consensus":"editorial","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00050775165,0.007678829,0.0031070563,0.053502377,0.8848911,0.00021381435,0.0014423855,0.00079189823,0.047864784],"genre_scores_gemma":[0.0020977529,0.010965606,0.0017620808,0.02022635,0.47141853,0.00035109563,0.0036317508,0.0017008781,0.48784596],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9980015,0.00027391498,0.00021282086,0.0003158758,0.00095827726,0.0002375699],"domain_scores_gemma":[0.98658943,0.0026049383,0.00076200604,0.0010084654,0.005746542,0.0032885696],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004155569,0.0024450622,0.0048289425,0.0064820424,0.003743128,0.011258708,0.002722523,0.004678031,0.27611476],"category_scores_gemma":[0.010006194,0.0009426191,0.0019228981,0.003742541,0.0011638969,0.009881618,0.0049785646,0.006341415,0.19553678],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000019620502,0.000019330892,0.000054692013,0.000054109456,0.000004372727,0.000022261924,0.000009942447,0.000015387579,0.000074537566,0.0002652216,0.9901596,0.009300995],"study_design_scores_gemma":[0.000015103962,0.000024601217,0.00048702446,0.00007807297,0.000014203116,0.000035758083,0.000043192955,0.00015774665,0.00006809717,0.00078981376,0.9982736,0.0000127470075],"about_ca_topic_score_codex":0.002351646,"about_ca_topic_score_gemma":0.00814525,"teacher_disagreement_score":0.27611476,"about_ca_system_score_codex":0.0017150041,"about_ca_system_score_gemma":0.0034155338,"threshold_uncertainty_score":0.9236959},"labels":[],"label_agreement":null},{"id":"W209234326","doi":"","title":"Classification et catégorisation automatiques : application à l'analyse thématique des données textuelles","year":2004,"lang":"fr","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Categorization; Computer science; Thematic map; Natural language processing; Comprehension; Artificial intelligence; Information retrieval; Data science; Cartography","score_opus":0.045208941293670674,"score_gpt":0.31432363515997247,"score_spread":0.2691146938663018,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W209234326","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023249859,0.0017582313,0.9616043,0.0009598367,0.00018647352,0.00087382586,0.0012469,0.006980411,0.0031402656],"genre_scores_gemma":[0.05993065,0.0006717671,0.9343742,0.00011576898,0.00008780093,0.0007836052,0.0016958204,0.00032976762,0.0020106153],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9864579,0.006260215,0.0013778674,0.0025356787,0.002943928,0.00042448],"domain_scores_gemma":[0.9592187,0.030696975,0.001491761,0.0022724199,0.005950365,0.00036986932],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010234275,0.0019439129,0.0016911437,0.01972773,0.0018913796,0.0064828526,0.002633953,0.0031538922,0.0060867057],"category_scores_gemma":[0.03553476,0.0008727795,0.0024089757,0.011134555,0.0021269226,0.005004587,0.0022778518,0.0031729662,0.0036932584],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021928328,0.00033317597,0.008531188,0.001043403,0.00023284108,0.00012623714,0.0025485677,0.005604769,0.01024283,0.011086437,0.006627872,0.9534035],"study_design_scores_gemma":[0.00030835796,0.00040372476,0.040120456,0.0010888574,0.00047268093,0.0018029802,0.00789131,0.68281496,0.050435953,0.10173681,0.112511635,0.00041227043],"about_ca_topic_score_codex":0.011892946,"about_ca_topic_score_gemma":0.011308773,"teacher_disagreement_score":0.01972773,"about_ca_system_score_codex":0.0019341524,"about_ca_system_score_gemma":0.0030978613,"threshold_uncertainty_score":0.054124713},"labels":[],"label_agreement":null},{"id":"W2092527610","doi":"10.1162/089120102760173625","title":"Near-Synonymy and Lexical Choice","year":2002,"lang":"en","type":"article","venue":"Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":258,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Natural language processing; Denotation (semiotics); Artificial intelligence; Context (archaeology); Lexicon; Lexical choice; Set (abstract data type); Selection (genetic algorithm); Lexical semantics; Ontology; Lexical item; Linguistics","score_opus":0.02487434097093336,"score_gpt":0.28440130432046784,"score_spread":0.25952696334953446,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2092527610","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021573396,0.00020851874,0.9677023,0.00071761594,0.0000503013,0.00010364211,0.0002081746,0.00019474399,0.009241168],"genre_scores_gemma":[0.46403012,0.0002978935,0.5263891,0.00029721714,0.00007738081,0.0004240198,0.00062551425,0.00013733086,0.0077213557],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9967014,0.0010054677,0.00030510023,0.0010240866,0.00076087145,0.00020300734],"domain_scores_gemma":[0.9966118,0.0017314921,0.00035998473,0.0007326696,0.00040062953,0.00016340181],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020688106,0.00067599,0.001127268,0.0025086123,0.0016534659,0.0041449247,0.0031549344,0.0017857943,0.0071521318],"category_scores_gemma":[0.007947455,0.00075842027,0.0029765733,0.0028011203,0.004996182,0.013060553,0.0038125303,0.0022133759,0.0015309411],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000036624366,0.000023986113,0.00056430197,0.00007986676,0.00003870005,0.00015193298,0.0005940836,0.02199537,0.0011624619,0.9567356,0.000641158,0.017975945],"study_design_scores_gemma":[0.000014692784,0.000017661667,0.00016301683,0.000022771263,0.00001717957,0.000097507735,0.00012797352,0.09206878,0.0006908448,0.90154225,0.005215084,0.000022352855],"about_ca_topic_score_codex":0.003954945,"about_ca_topic_score_gemma":0.0033679921,"teacher_disagreement_score":0.0071521318,"about_ca_system_score_codex":0.002397544,"about_ca_system_score_gemma":0.0016680879,"threshold_uncertainty_score":0.023926318},"labels":[],"label_agreement":null},{"id":"W2093592543","doi":"10.7202/004200ar","title":"La génération de textes multilingues par un utilisateur monolingue","year":2002,"lang":"fr","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Philosophy; Art","score_opus":0.05287125969336089,"score_gpt":0.28799329601942664,"score_spread":0.23512203632606574,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2093592543","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.049880836,0.0005315401,0.9337165,0.00037421775,0.00019218783,0.00014580277,0.00026215182,0.009193316,0.0057034763],"genre_scores_gemma":[0.2899396,0.00048699733,0.6856113,0.00029428842,0.00011583958,0.0002357835,0.000923822,0.002038322,0.020354062],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9984004,0.00059883273,0.000095280004,0.0004936499,0.00033951818,0.00007218649],"domain_scores_gemma":[0.996689,0.0018789448,0.0001398889,0.0006077502,0.0006093261,0.00007516431],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014423288,0.0011550538,0.00068453694,0.00075701583,0.000852124,0.0020131492,0.0009750568,0.0012098968,0.0064924783],"category_scores_gemma":[0.005796566,0.0004913466,0.00095398567,0.00050925335,0.00079369615,0.0023196673,0.0013581158,0.0009681922,0.0046300786],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00080531026,0.00018702574,0.0021763865,0.0009963815,0.00016649858,0.0013843075,0.0053036986,0.009692711,0.35708895,0.01288517,0.005623565,0.60369],"study_design_scores_gemma":[0.00016584831,0.00090778753,0.0035570955,0.00025155483,0.00053735875,0.0026536994,0.0020843144,0.18578902,0.57224333,0.02586605,0.20569313,0.0002507834],"about_ca_topic_score_codex":0.0015975229,"about_ca_topic_score_gemma":0.0018913021,"teacher_disagreement_score":0.0064924783,"about_ca_system_score_codex":0.00037784787,"about_ca_system_score_gemma":0.00049675157,"threshold_uncertainty_score":0.021719456},"labels":[],"label_agreement":null},{"id":"W2094821119","doi":"10.7202/002000ar","title":"Studies of Translation Models 3 : An Interaction Model of the Translation Process","year":2002,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Notation; Translation (biology); Computer science; Process (computing); Task (project management); Dynamic and formal equivalence; Natural language processing; Artificial intelligence; Programming language; Machine translation; Linguistics; Engineering; Systems engineering","score_opus":0.19074537898347274,"score_gpt":0.34777991798258096,"score_spread":0.15703453899910821,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2094821119","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0031086807,0.0014602074,0.9824358,0.000975278,0.000106306674,0.00008044079,0.00013658668,0.00045573508,0.011240937],"genre_scores_gemma":[0.27678657,0.0039845677,0.69818336,0.0008493859,0.00061678153,0.0013545947,0.0009760037,0.0012574869,0.015991312],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9906204,0.0056573767,0.0006029968,0.0013772885,0.0013783396,0.0003636355],"domain_scores_gemma":[0.979779,0.014645168,0.0015958918,0.0026704096,0.0010729823,0.00023649438],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0077297795,0.0026285907,0.0015877602,0.0027827928,0.002386402,0.008064146,0.004174502,0.0046906387,0.012367984],"category_scores_gemma":[0.024059176,0.002062875,0.004627214,0.0033458346,0.005682269,0.013838443,0.0037571178,0.0045738337,0.0044259275],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006157376,0.00003310025,0.00024118807,0.00038188073,0.000095332936,0.00021767158,0.0013070032,0.029405667,0.0009707526,0.9469649,0.0021277943,0.01819311],"study_design_scores_gemma":[0.00004579353,0.00009318326,0.00019782361,0.00021608063,0.00013974452,0.0003839031,0.00017801365,0.1517315,0.0019273289,0.81284356,0.03215963,0.000083487925],"about_ca_topic_score_codex":0.0032741963,"about_ca_topic_score_gemma":0.001547089,"teacher_disagreement_score":0.012367984,"about_ca_system_score_codex":0.0035563833,"about_ca_system_score_gemma":0.0019945446,"threshold_uncertainty_score":0.04137504},"labels":[],"label_agreement":null},{"id":"W2094874816","doi":"10.1145/860435.860534","title":"Passage retrieval vs. document retrieval for factoid question answering","year":2003,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":40,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Question answering; Information retrieval; Computer science; Document retrieval; Natural language processing; Artificial intelligence","score_opus":0.012281876279744681,"score_gpt":0.2841401935360828,"score_spread":0.2718583172563381,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2094874816","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.18051536,0.07357943,0.5940068,0.0074502714,0.005223033,0.0035997762,0.010353729,0.01561575,0.109655894],"genre_scores_gemma":[0.6740776,0.014251397,0.266142,0.0009884103,0.0019426895,0.0008992367,0.01157137,0.0009216969,0.029205633],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9976484,0.0012307304,0.00016119546,0.00031055094,0.0003920262,0.00025712576],"domain_scores_gemma":[0.98910344,0.00892237,0.00022008976,0.0007930209,0.00065099366,0.00031010175],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036741127,0.00072680856,0.001771398,0.0030021756,0.0010039274,0.0023885146,0.0017055706,0.0017852308,0.043129656],"category_scores_gemma":[0.017422156,0.00026560304,0.0012065799,0.0021714424,0.0008268132,0.0046341876,0.0012327941,0.001089622,0.0095650125],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.009163329,0.00062873116,0.0028518697,0.0052704024,0.00043452115,0.00043591906,0.00058545923,0.0059158676,0.061726876,0.027387362,0.061385307,0.82421434],"study_design_scores_gemma":[0.0026699412,0.009485411,0.023767011,0.002125639,0.0047816024,0.005309526,0.0026669777,0.4083425,0.19012807,0.1465807,0.20343073,0.0007118505],"about_ca_topic_score_codex":0.0046933223,"about_ca_topic_score_gemma":0.0034959621,"teacher_disagreement_score":0.043129656,"about_ca_system_score_codex":0.00072949537,"about_ca_system_score_gemma":0.00073666614,"threshold_uncertainty_score":0.14428306},"labels":[],"label_agreement":null},{"id":"W2095083114","doi":"10.7202/012241ar","title":"Présentation : TALN, Web et corpus","year":2003,"lang":"fr","type":"article","venue":"Revue québécoise de linguistique","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Computer science; Natural language processing; Linguistics; World Wide Web; Philosophy","score_opus":0.017651259057589002,"score_gpt":0.3071661175675773,"score_spread":0.28951485850998826,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2095083114","genre_codex":"dataset","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0055664834,0.0065007512,0.049625963,0.1172377,0.17934693,0.0008798699,0.32482734,0.026673973,0.28934103],"genre_scores_gemma":[0.02134693,0.003449192,0.014914567,0.007016032,0.025119254,0.00078095903,0.19691636,0.009553934,0.72090286],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9984871,0.00029866194,0.00010198391,0.0002223045,0.00079789455,0.000092090166],"domain_scores_gemma":[0.99393964,0.0016299719,0.00016266138,0.00067917054,0.0029158397,0.0006727541],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0028747073,0.0011341986,0.0014106238,0.0055139684,0.0026703493,0.007594299,0.0017436227,0.0017349742,0.36854324],"category_scores_gemma":[0.015050933,0.00053515605,0.00057607313,0.005659664,0.0008942824,0.0052856496,0.0029097935,0.0029958966,0.14340045],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007557537,0.000013682472,0.000055265446,0.000080862075,0.00000460778,0.000038636954,0.000025973213,0.00007662281,0.00030198914,0.0031778254,0.98545295,0.010696091],"study_design_scores_gemma":[0.000035105844,0.000010270443,0.00029789325,0.000056527268,0.000007678962,0.00007947582,0.00007730121,0.001114388,0.0007463589,0.0040645716,0.9934935,0.000016948625],"about_ca_topic_score_codex":0.020184683,"about_ca_topic_score_gemma":0.01952007,"teacher_disagreement_score":0.36854324,"about_ca_system_score_codex":0.0027079554,"about_ca_system_score_gemma":0.004121065,"threshold_uncertainty_score":0.90069646},"labels":[],"label_agreement":null},{"id":"W2095324656","doi":"10.1177/0267658308100294","title":"Some questions about feature re-assembly","year":2009,"lang":"en","type":"article","venue":"Second language Research","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Feature (linguistics); Linguistics; Computer science; Feature selection; Natural language processing; Artificial intelligence; Psychology; Second-language acquisition; Selection (genetic algorithm); Philosophy","score_opus":0.026190026469349086,"score_gpt":0.37926040457413995,"score_spread":0.35307037810479086,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2095324656","genre_codex":"commentary","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00081853033,0.0069591696,0.009186091,0.9582481,0.011974417,0.000012186647,0.00004110257,0.00007637173,0.01268403],"genre_scores_gemma":[0.090153374,0.013411799,0.01381651,0.76033956,0.08952967,0.0001690844,0.00012317141,0.0003902943,0.03206657],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99215007,0.0027181616,0.00044026697,0.0017926432,0.0022491603,0.0006497466],"domain_scores_gemma":[0.9789392,0.015413657,0.00054924504,0.00088017335,0.0037705791,0.0004472264],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.014744026,0.0006667801,0.00083024206,0.00091082376,0.0042130486,0.0040302514,0.0048648496,0.015894752,0.0037727065],"category_scores_gemma":[0.031222733,0.00038857418,0.0010423616,0.0008725508,0.015381907,0.011736501,0.0020612068,0.019241262,0.0015716681],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005151243,0.000014140612,0.00018858572,0.00017054808,0.000014471018,0.0002681911,0.0030496074,0.0002866974,0.0004893725,0.73018146,0.23489717,0.03038826],"study_design_scores_gemma":[0.000023330642,0.00003461036,0.00044137545,0.0002552844,0.000012202957,0.0002985599,0.0015935566,0.0009245792,0.0005941359,0.3353088,0.66046125,0.000052255647],"about_ca_topic_score_codex":0.010493615,"about_ca_topic_score_gemma":0.010561416,"teacher_disagreement_score":0.015894752,"about_ca_system_score_codex":0.0034707482,"about_ca_system_score_gemma":0.0036191414,"threshold_uncertainty_score":0.0779748},"labels":[],"label_agreement":null},{"id":"W2095705961","doi":"10.3115/1654650.1654668","title":"Mood at work","year":2006,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Phrase; Task (project management); Computer science; Machine translation; Translation (biology); Baseline (sea); Natural language processing; Open source; Artificial intelligence; Programming language; Engineering; Software; Systems engineering","score_opus":0.006141032563635111,"score_gpt":0.22970271501538003,"score_spread":0.22356168245174493,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2095705961","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008398448,0.0011007593,0.0031140503,0.008273416,0.0029619937,0.00024975187,0.0084333895,0.004519754,0.96294844],"genre_scores_gemma":[0.09910578,0.0011821612,0.0060931593,0.005548134,0.001244573,0.00056187843,0.013263032,0.0023088043,0.87069243],"study_design_codex":"not_applicable","study_design_gemma":"observational","domain_scores_codex":[0.99801385,0.00045171715,0.00010484849,0.00044877458,0.00058836356,0.0003925058],"domain_scores_gemma":[0.9958533,0.00025816943,0.00021251533,0.0005532409,0.0011042112,0.002018644],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0016076396,0.00081480877,0.00075788784,0.0016942193,0.0046471013,0.0061652623,0.001496548,0.0015410974,0.40710205],"category_scores_gemma":[0.00787027,0.0004079251,0.00058197486,0.0016209447,0.0008587488,0.0033408264,0.005256031,0.0017068922,0.35380563],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017301041,0.00010517948,0.002068515,0.00016473838,0.000011713198,0.0002300828,0.002676694,0.000050680872,0.00052181876,0.011722854,0.87824124,0.10403338],"study_design_scores_gemma":[0.000025018817,0.00003576777,0.0019105084,0.00006508638,0.000009194513,0.0001737109,0.0011135064,0.0000768721,0.00016178748,0.0033941958,0.9930161,0.000018232433],"about_ca_topic_score_codex":0.0037242165,"about_ca_topic_score_gemma":0.0077049127,"teacher_disagreement_score":0.40710205,"about_ca_system_score_codex":0.0015449764,"about_ca_system_score_gemma":0.0015452884,"threshold_uncertainty_score":0.845697},"labels":[],"label_agreement":null},{"id":"W2096160878","doi":"10.7202/012245ar","title":"Le modèle Lstat : ou comment se constituer une base de données morphologique à partir du Web","year":2006,"lang":"fr","type":"article","venue":"Revue québécoise de linguistique","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Philosophy; Humanities","score_opus":0.024786078235831892,"score_gpt":0.2675359519191309,"score_spread":0.24274987368329898,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2096160878","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0152610885,0.0004474085,0.95938456,0.0013197972,0.00018463224,0.00024292235,0.0033991956,0.012380802,0.0073794746],"genre_scores_gemma":[0.11938135,0.000687703,0.8441196,0.0006643536,0.00011217736,0.0008240753,0.008277279,0.003257651,0.022675887],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9985977,0.00034911532,0.00012144907,0.00042713922,0.00044549932,0.00005922721],"domain_scores_gemma":[0.9974136,0.0010581157,0.00012407092,0.00063663983,0.00070304103,0.000064605396],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018846446,0.00083819707,0.00065248,0.0028313824,0.0010305267,0.004522024,0.0014218383,0.0012314445,0.009527686],"category_scores_gemma":[0.008304799,0.0008860463,0.0015414482,0.0026236374,0.0011429704,0.005565582,0.0019232424,0.0016712977,0.0061451704],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010776446,0.00021993856,0.018360948,0.0019597549,0.0005491859,0.0011671234,0.007355766,0.038306836,0.06942601,0.11166182,0.04651708,0.7033979],"study_design_scores_gemma":[0.000115886156,0.0001625501,0.0054331254,0.00047047227,0.00037117748,0.0011466041,0.0022442238,0.41327265,0.07282492,0.07730361,0.4264719,0.00018293767],"about_ca_topic_score_codex":0.016176859,"about_ca_topic_score_gemma":0.018207494,"teacher_disagreement_score":0.016176859,"about_ca_system_score_codex":0.001464619,"about_ca_system_score_gemma":0.003214705,"threshold_uncertainty_score":0.032165408},"labels":[],"label_agreement":null},{"id":"W2096223775","doi":"10.3115/1613715.1613823","title":"Complexity of finding the BLEU-optimal hypothesis in a confusion network","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"National Research Council Canada; Defense Advanced Research Projects Agency","keywords":"BLEU; Confusion; Computer science; Metric (unit); Translation (biology); Artificial intelligence; Representation (politics); Word (group theory); Machine translation; Simple (philosophy); Time complexity; Natural language processing; Theoretical computer science; Algorithm; Mathematics","score_opus":0.06938032207941744,"score_gpt":0.27235590840822904,"score_spread":0.2029755863288116,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2096223775","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.33221686,0.0019287292,0.6285031,0.009586108,0.00020509453,0.00063158385,0.0070006936,0.003475216,0.016452659],"genre_scores_gemma":[0.7967283,0.0008585091,0.1862164,0.00085052033,0.0002699832,0.00061116344,0.009462709,0.0007446407,0.0042577884],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9924959,0.0035300837,0.00044421392,0.001857125,0.0011049643,0.00056770985],"domain_scores_gemma":[0.9488642,0.04612985,0.0011223655,0.0019702073,0.001329091,0.0005843159],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0056011667,0.0011376337,0.002681849,0.0019240133,0.0019768726,0.005500004,0.0027325465,0.003464887,0.009444548],"category_scores_gemma":[0.04991028,0.0010484367,0.001514278,0.0026696026,0.0021897461,0.009507718,0.0034435682,0.0029158124,0.0016791271],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0022196653,0.00047018574,0.010447656,0.0014170791,0.00040795494,0.00067454914,0.0010982594,0.5796991,0.006142127,0.07739761,0.03131216,0.28871363],"study_design_scores_gemma":[0.00016294181,0.00012738252,0.0019912794,0.000055574892,0.000074785545,0.00024381535,0.00030825648,0.7628509,0.0024476473,0.2302357,0.0014419699,0.00005973653],"about_ca_topic_score_codex":0.0054122005,"about_ca_topic_score_gemma":0.0045489417,"teacher_disagreement_score":0.009444548,"about_ca_system_score_codex":0.0039033827,"about_ca_system_score_gemma":0.0031593328,"threshold_uncertainty_score":0.03159511},"labels":[],"label_agreement":null},{"id":"W2096446497","doi":"10.3115/v1/w14-2405","title":"A Deep Architecture for Semantic Parsing","year":2014,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":54,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Engineering and Physical Sciences Research Council; Canadian Institute for Advanced Research; Xerox Foundation","keywords":"Computer science; Natural language processing; Parsing; Artificial intelligence; Ontology; Semantics (computer science); Bottom-up parsing; Architecture; Natural language; Top-down parsing; Programming language","score_opus":0.008633571998138169,"score_gpt":0.2575819397505558,"score_spread":0.24894836775241766,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2096446497","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00776703,0.0003154473,0.9795776,0.00081028754,0.000098668614,0.000047771533,0.0006283454,0.0052900813,0.00546472],"genre_scores_gemma":[0.37755704,0.0008209156,0.6035182,0.001032477,0.00013298212,0.00020905837,0.0030362774,0.0005420622,0.013150932],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99958056,0.0000853144,0.000030959884,0.00014718671,0.00009995917,0.00005594489],"domain_scores_gemma":[0.9995534,0.00016728323,0.000026356967,0.00010571975,0.00011926928,0.000028033655],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007623596,0.0006996358,0.00045610085,0.0006659638,0.0006406151,0.0015227397,0.0012868918,0.0015479788,0.004874155],"category_scores_gemma":[0.001996917,0.00051215314,0.0010582892,0.0008551409,0.0010280493,0.0036887254,0.0014262492,0.0024616416,0.0018125095],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002494131,0.00018900014,0.0015849966,0.00033667253,0.00016599846,0.0003208183,0.000609326,0.14247735,0.029063636,0.30814812,0.029216329,0.48763826],"study_design_scores_gemma":[0.000012545523,0.000037110764,0.00029609227,0.000040308754,0.000053923846,0.000098350385,0.00004984583,0.76169235,0.0075946543,0.21448302,0.01561572,0.000026016674],"about_ca_topic_score_codex":0.0068069613,"about_ca_topic_score_gemma":0.00923187,"teacher_disagreement_score":0.0068069613,"about_ca_system_score_codex":0.0012667074,"about_ca_system_score_gemma":0.0016686224,"threshold_uncertainty_score":0.016305625},"labels":[],"label_agreement":null},{"id":"W2096565906","doi":"","title":"Alignment-Based Discriminative String Similarity","year":2007,"lang":"en","type":"article","venue":"Meeting of the Association for Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":54,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Discriminative model; Coreference; Artificial intelligence; Computer science; Substring; String metric; Similarity (geometry); Character (mathematics); Natural language processing; Longest common subsequence problem; String (physics); Word (group theory); Pattern recognition (psychology); Transliteration; Heuristic; String searching algorithm; Mathematics; Pattern matching; Resolution (logic); Algorithm; Set (abstract data type)","score_opus":0.016287383324153314,"score_gpt":0.2953723765151392,"score_spread":0.2790849931909859,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2096565906","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.071590476,0.0004947179,0.92107713,0.00008972681,0.00006303943,0.00014085282,0.00049438543,0.0024366227,0.0036129039],"genre_scores_gemma":[0.65871793,0.00021564485,0.33593118,0.00013686359,0.000090831214,0.00016617337,0.0022112723,0.00026294275,0.002267076],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9973628,0.0006219286,0.00016836106,0.00069978915,0.0009777171,0.00016941133],"domain_scores_gemma":[0.9960316,0.0011260952,0.00054293894,0.0012828233,0.000843252,0.00017325483],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015423504,0.0006003175,0.0015307086,0.0038469615,0.00068933656,0.0011024094,0.0015710277,0.0007734903,0.0025075197],"category_scores_gemma":[0.007312349,0.0002540779,0.0005424664,0.004968217,0.00085128366,0.002222912,0.0016423991,0.000862533,0.0017948373],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004615589,0.00038852202,0.013246618,0.00041860636,0.00015520724,0.00018255835,0.00024567105,0.034781102,0.08358547,0.031604696,0.007315861,0.8276141],"study_design_scores_gemma":[0.00007156301,0.0008021127,0.01997499,0.00003487004,0.00010344212,0.0017160926,0.00019462398,0.8387704,0.073135614,0.052172024,0.012901578,0.0001227065],"about_ca_topic_score_codex":0.0008711013,"about_ca_topic_score_gemma":0.0015194186,"teacher_disagreement_score":0.0038469615,"about_ca_system_score_codex":0.00055124256,"about_ca_system_score_gemma":0.000798322,"threshold_uncertainty_score":0.00838846},"labels":[],"label_agreement":null},{"id":"W2096701874","doi":"10.1109/imcsit.2010.5679866","title":"SyMGiza++: A tool for parallel computation of symmetrized word alignment models","year":2010,"lang":"en","type":"article","venue":"Proceedings of the International Multiconference on Computer Science and Information Technology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computation; Word (group theory); Computer science; IBM; Task (project management); Series (stratigraphy); Simple (philosophy); Parallel computing; Algorithm; Theoretical computer science; Mathematics; Engineering","score_opus":0.012491314247379632,"score_gpt":0.25899301551830556,"score_spread":0.24650170127092594,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2096701874","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0035364938,0.00017636105,0.85638404,0.00010134955,0.00011315182,0.000081000886,0.0014384242,0.13585515,0.0023140002],"genre_scores_gemma":[0.04441922,0.00025234942,0.9281772,0.00018249727,0.00007266665,0.0004740081,0.00604828,0.015700556,0.004673251],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9990797,0.00023459985,0.000086574844,0.00027293462,0.00024782223,0.00007829319],"domain_scores_gemma":[0.9988586,0.00046909737,0.00009502493,0.00033541888,0.00018973232,0.00005217017],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010985625,0.003159309,0.0015805343,0.0015034783,0.00090954977,0.0022917686,0.0036307059,0.0011218671,0.033657834],"category_scores_gemma":[0.0043744836,0.0015154688,0.001784941,0.0018980335,0.00068801374,0.0029198201,0.0023894247,0.0024876099,0.020996952],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009842243,0.00014774667,0.0018210555,0.001390516,0.00045069042,0.0006643096,0.0005766302,0.059968848,0.03246264,0.04718464,0.1469501,0.70739853],"study_design_scores_gemma":[0.0004157598,0.00017073995,0.00062348123,0.00010175466,0.00013060386,0.00049747573,0.00019206681,0.7050498,0.04947333,0.08473341,0.15845011,0.0001614894],"about_ca_topic_score_codex":0.003478125,"about_ca_topic_score_gemma":0.006316558,"teacher_disagreement_score":0.033657834,"about_ca_system_score_codex":0.0008415952,"about_ca_system_score_gemma":0.001682112,"threshold_uncertainty_score":0.11259663},"labels":[],"label_agreement":null},{"id":"W2098303862","doi":"","title":"A semantic MediaWiki-empowered terminology registry","year":2009,"lang":"en","type":"article","venue":"International Conference on Dublin Core and Metadata Applications","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Terminology; Computer science; World Wide Web; Semantic Web; Leverage (statistics); Metadata; Semantic interoperability; Semantic grid; Semantic analytics; Semantic Web Stack; Semantic computing; Semantic technology; Information retrieval; Knowledge management; Data science; Interoperability; Artificial intelligence","score_opus":0.06696194600916354,"score_gpt":0.3588482288106158,"score_spread":0.29188628280145223,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2098303862","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010952746,0.00017868598,0.9434644,0.0013338996,0.00061503233,0.0010524718,0.001460786,0.019612698,0.021329248],"genre_scores_gemma":[0.056182854,0.00042286466,0.9177523,0.00036718417,0.00022081523,0.0011875696,0.0082252305,0.0020998733,0.0135412235],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.992927,0.0020969315,0.0015029834,0.0011698316,0.001924124,0.00037908126],"domain_scores_gemma":[0.9891376,0.0015417652,0.00089549826,0.004448706,0.0026800334,0.0012964306],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.015852561,0.00065280223,0.0012735581,0.0052466076,0.0041229757,0.009928415,0.004047922,0.0018867168,0.0061769867],"category_scores_gemma":[0.016197575,0.0012583925,0.0017660336,0.0051685427,0.0024078945,0.024163557,0.011664181,0.0036136312,0.007514951],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006788137,0.0008605239,0.0047127595,0.00064981706,0.00013591425,0.0019004361,0.0038246561,0.0066533014,0.03041642,0.64711314,0.058277816,0.24477626],"study_design_scores_gemma":[0.00025671098,0.00030292865,0.0010587098,0.00029787686,0.00018414353,0.0020178466,0.0016572695,0.05193161,0.045945548,0.13332734,0.7627209,0.00029915723],"about_ca_topic_score_codex":0.002433605,"about_ca_topic_score_gemma":0.002667223,"teacher_disagreement_score":0.015852561,"about_ca_system_score_codex":0.0015233373,"about_ca_system_score_gemma":0.01142633,"threshold_uncertainty_score":0.08383733},"labels":[],"label_agreement":null},{"id":"W2098320952","doi":"10.1109/nlpke.2010.5587782","title":"An unsupervised approach to preposition error correction","year":2010,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Natural language processing; Set (abstract data type); n-gram; Artificial intelligence; Error detection and correction; State (computer science); Data set; Information retrieval; Language model; Algorithm; Programming language","score_opus":0.011187066518193598,"score_gpt":0.27468304864849274,"score_spread":0.2634959821302991,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2098320952","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024436858,0.0003778758,0.954735,0.00023972611,0.00029523566,0.00028757643,0.0015292767,0.014828303,0.0032702133],"genre_scores_gemma":[0.14556351,0.00027793273,0.8405151,0.00026592633,0.00029084797,0.0003443742,0.004923977,0.0018570219,0.005961209],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.993771,0.0015136392,0.00049576035,0.0017920078,0.0021835265,0.00024399442],"domain_scores_gemma":[0.98140115,0.006173192,0.0019772362,0.0035564196,0.006662852,0.00022915364],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002499174,0.0016972182,0.001317094,0.0060556703,0.0016292033,0.0016608277,0.0027301072,0.0014481505,0.0036187838],"category_scores_gemma":[0.014415041,0.000631263,0.0012768168,0.0042192247,0.0011581982,0.002554205,0.0019017027,0.0024553628,0.0056115505],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035086452,0.00029199128,0.0077742985,0.00056752487,0.00029963802,0.0004671203,0.0004558932,0.010730185,0.062133353,0.0055454113,0.01822718,0.89315647],"study_design_scores_gemma":[0.000081269536,0.0003222953,0.019034822,0.00016041174,0.00024842643,0.003161579,0.0007746527,0.66076976,0.22140856,0.027461478,0.06617481,0.00040188833],"about_ca_topic_score_codex":0.0037062066,"about_ca_topic_score_gemma":0.008757082,"teacher_disagreement_score":0.0060556703,"about_ca_system_score_codex":0.0006066588,"about_ca_system_score_gemma":0.0033797608,"threshold_uncertainty_score":0.013217092},"labels":[],"label_agreement":null},{"id":"W2098551021","doi":"10.1007/3-540-45820-4_11","title":"Merging Example-Based and Statistical Machine Translation: An Experiment","year":2002,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Machine translation; Computer science; Translation (biology); Natural language processing; Task (project management); Artificial intelligence; Field (mathematics); Point (geometry); Example-based machine translation; Machine translation software usability; Computer-assisted translation","score_opus":0.0299062382411443,"score_gpt":0.2857405022603446,"score_spread":0.2558342640192003,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2098551021","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9528243,0.0022649975,0.025014237,0.0006773228,0.00053466717,0.000671102,0.0036333583,0.0063206716,0.008059305],"genre_scores_gemma":[0.894458,0.0007820752,0.08611007,0.00055382197,0.00021913789,0.00041089652,0.012173405,0.0014175402,0.003875221],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99174374,0.005182057,0.0008800538,0.0010108552,0.00091279956,0.00027045913],"domain_scores_gemma":[0.94883454,0.040231995,0.00080423494,0.0054871105,0.0040176846,0.0006245113],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0076264697,0.0013533372,0.0017530274,0.0012866877,0.0012676091,0.0018590525,0.0019553702,0.0031476049,0.010356485],"category_scores_gemma":[0.038007237,0.0010305543,0.0011948796,0.002397163,0.00084874773,0.004450623,0.0019672287,0.0015616042,0.005314922],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.053194035,0.008181333,0.01860489,0.0070176427,0.0030618454,0.0032492003,0.0071152844,0.043656748,0.14750692,0.0038858843,0.02759749,0.67692876],"study_design_scores_gemma":[0.015739966,0.035479087,0.06561679,0.00056993833,0.007541659,0.00914232,0.008599535,0.43425608,0.33559972,0.018677475,0.067753315,0.0010241471],"about_ca_topic_score_codex":0.0026622266,"about_ca_topic_score_gemma":0.0023679312,"teacher_disagreement_score":0.010356485,"about_ca_system_score_codex":0.00040703377,"about_ca_system_score_gemma":0.0014683568,"threshold_uncertainty_score":0.04033315},"labels":[],"label_agreement":null},{"id":"W2098699913","doi":"","title":"Scalable Variational Inference for Extracting Hierarchical Phrase-based Translation Rules","year":2013,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Phrase; Inference; Artificial intelligence; Scalability; Translation (biology); Heuristic; Machine translation; Natural language processing; Set (abstract data type); Bayes' theorem; Bayesian probability","score_opus":0.026481362579417758,"score_gpt":0.2966106733359311,"score_spread":0.27012931075651336,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2098699913","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.003559125,0.00011310195,0.9952604,0.00009034136,0.000013499165,0.000028441409,0.00008617861,0.00032936607,0.0005195007],"genre_scores_gemma":[0.2818476,0.0003280074,0.7124754,0.00025898186,0.00011292315,0.00031311385,0.0011720128,0.00043962593,0.0030523045],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99881697,0.00042859485,0.00007259508,0.0002927723,0.00031007998,0.000079028585],"domain_scores_gemma":[0.9967925,0.0024217735,0.00015241797,0.00030175247,0.00025183122,0.000079719124],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025841887,0.00087382086,0.0016998686,0.00082719116,0.00052089075,0.0012181788,0.002778395,0.0014641546,0.0030676078],"category_scores_gemma":[0.00854383,0.0012659613,0.0011759934,0.0013402246,0.0010944726,0.002358746,0.0016368958,0.002431867,0.00091278367],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007986199,0.00006866129,0.0004961419,0.00008866276,0.00010567501,0.00009624462,0.00007856254,0.8740089,0.0038791448,0.03585692,0.0022632387,0.08297811],"study_design_scores_gemma":[0.000005136189,0.0000034958855,0.000019431525,0.0000018371127,0.0000027548097,0.0000042532815,0.0000015000455,0.99017054,0.00018745226,0.009478749,0.00012281442,0.0000020735597],"about_ca_topic_score_codex":0.010201656,"about_ca_topic_score_gemma":0.018328004,"teacher_disagreement_score":0.010201656,"about_ca_system_score_codex":0.0013565116,"about_ca_system_score_gemma":0.0018602551,"threshold_uncertainty_score":0.020284593},"labels":[],"label_agreement":null},{"id":"W2099032682","doi":"","title":"Neutralizing Linguistically Problematic Annotations in Unsupervised Dependency Parsing Evaluation","year":2011,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":42,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Parsing; Computer science; Dependency (UML); Annotation; Dependency grammar; Artificial intelligence; Natural language processing; Set (abstract data type); Task (project management); Measure (data warehouse); Quality (philosophy); Data mining; Programming language","score_opus":0.07206090392163111,"score_gpt":0.3153775708490825,"score_spread":0.2433166669274514,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2099032682","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2719042,0.0012316927,0.71137446,0.00047758708,0.00021243426,0.0004671464,0.0005600523,0.005660979,0.0081114415],"genre_scores_gemma":[0.7166696,0.0002780091,0.27729607,0.00036502577,0.000060718507,0.0003862768,0.0014757764,0.0015169659,0.001951532],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9565113,0.027810398,0.002854489,0.003378671,0.008543336,0.0009017856],"domain_scores_gemma":[0.8748586,0.091855876,0.0051338,0.012530454,0.014564166,0.0010570424],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.027169207,0.0022849012,0.0017745045,0.0027806317,0.001641232,0.0034454963,0.0022770478,0.002856299,0.0012641437],"category_scores_gemma":[0.09880848,0.0006794934,0.00069107214,0.002368794,0.0020336835,0.0050783698,0.0044432827,0.0024474883,0.0007731362],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021464445,0.00078455,0.04040648,0.0015491758,0.00072080124,0.0008998192,0.0027896652,0.07411489,0.11501708,0.025145087,0.012538132,0.7238879],"study_design_scores_gemma":[0.00020527735,0.0012396632,0.028494034,0.00039818956,0.0005361846,0.0019280714,0.0016234657,0.6504807,0.22742775,0.065597355,0.021585356,0.0004839473],"about_ca_topic_score_codex":0.0015783919,"about_ca_topic_score_gemma":0.0040929085,"teacher_disagreement_score":0.027169207,"about_ca_system_score_codex":0.0012803936,"about_ca_system_score_gemma":0.0019674534,"threshold_uncertainty_score":0.14368623},"labels":[],"label_agreement":null},{"id":"W2099050228","doi":"10.3115/v1/w15-0909","title":"The Impact of Multiword Expression Compositionality on Machine Translation Evaluation","year":2015,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of New Brunswick","funders":"","keywords":"Principle of compositionality; Computer science; Machine translation; Natural language processing; Metric (unit); Artificial intelligence; Translation (biology); Expression (computer science); Programming language","score_opus":0.07300703251397235,"score_gpt":0.39631094599685945,"score_spread":0.3233039134828871,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2099050228","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.60610074,0.004717864,0.36144957,0.0018435537,0.00050686876,0.00060086773,0.0007761455,0.0066984985,0.017305939],"genre_scores_gemma":[0.85723966,0.00041557278,0.13840348,0.000305325,0.000117707226,0.0001724453,0.0009260297,0.000599102,0.0018206935],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9575892,0.02978311,0.002905677,0.0022861839,0.0065853028,0.0008505439],"domain_scores_gemma":[0.91315323,0.063200966,0.0026729314,0.005856004,0.014037774,0.0010790961],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.030645758,0.0023703012,0.0016087536,0.0027372504,0.0017802981,0.004590875,0.0014434556,0.0014360682,0.0023235604],"category_scores_gemma":[0.10421971,0.0005360837,0.0010283584,0.002472469,0.0012072122,0.006054872,0.0030251935,0.0021486487,0.0015175664],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0037408753,0.0008022553,0.0799354,0.0009884136,0.0013083026,0.00040316046,0.0006571326,0.052673426,0.07788344,0.0076937485,0.0049474966,0.7689664],"study_design_scores_gemma":[0.00016317019,0.0032754785,0.030946817,0.00024310597,0.00072997896,0.00068733987,0.0005537308,0.80741316,0.13543047,0.014183999,0.0061544306,0.00021835565],"about_ca_topic_score_codex":0.0046640756,"about_ca_topic_score_gemma":0.009695405,"teacher_disagreement_score":0.030645758,"about_ca_system_score_codex":0.0010495785,"about_ca_system_score_gemma":0.0020901673,"threshold_uncertainty_score":0.16207218},"labels":[],"label_agreement":null},{"id":"W2099083511","doi":"10.1145/1449715.1449736","title":"Is the sky pure today? AwkChecker","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Collocation (remote sensing); Computer science; Set (abstract data type); Natural language processing; Interface (matter); Word (group theory); Artificial intelligence; Linguistics; Programming language; Machine learning","score_opus":0.02042356425322616,"score_gpt":0.2679772891031445,"score_spread":0.24755372484991833,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2099083511","genre_codex":"other","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.087096915,0.0077372375,0.36842677,0.057862543,0.0045758453,0.0002443068,0.002867836,0.09995525,0.37123322],"genre_scores_gemma":[0.32813364,0.0040661595,0.40071002,0.011663563,0.0006732764,0.0001635776,0.0033152483,0.014957948,0.23631644],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99806815,0.00054051506,0.00016393643,0.00034124727,0.00075066485,0.00013552496],"domain_scores_gemma":[0.9930987,0.003056219,0.00028157936,0.001646488,0.0014191231,0.0004980284],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025901676,0.00053022034,0.00063888234,0.0010077159,0.0018566409,0.004539317,0.0011356117,0.0012962495,0.035719637],"category_scores_gemma":[0.015069343,0.0003527623,0.00022223893,0.0013888619,0.0017683787,0.009497773,0.0023564827,0.0016607582,0.022210505],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029969722,0.000050037976,0.002460493,0.00030908594,0.000020232506,0.00042491802,0.0030491874,0.00040728727,0.006379727,0.028735494,0.33035102,0.6275129],"study_design_scores_gemma":[0.000032523953,0.000042615728,0.0016854036,0.00023593554,0.000023948316,0.0013322135,0.0030131263,0.006161289,0.008369162,0.049020443,0.9299748,0.000108573244],"about_ca_topic_score_codex":0.0029988321,"about_ca_topic_score_gemma":0.006677756,"teacher_disagreement_score":0.035719637,"about_ca_system_score_codex":0.00064648327,"about_ca_system_score_gemma":0.001373276,"threshold_uncertainty_score":0.11949408},"labels":[],"label_agreement":null},{"id":"W2099195973","doi":"10.7202/012244ar","title":"Webaffix : une boîte à outils d’acquisition lexicale à partir du Web","year":2006,"lang":"fr","type":"article","venue":"Revue québécoise de linguistique","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Philosophy; Art; Political science","score_opus":0.011030609101904898,"score_gpt":0.2653177897882622,"score_spread":0.2542871806863573,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2099195973","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.048479185,0.0015741056,0.84204406,0.0009713746,0.00032712857,0.00084949186,0.020248659,0.06990929,0.015596774],"genre_scores_gemma":[0.07063365,0.00136244,0.8500809,0.0003758276,0.00012713036,0.0009580269,0.045056053,0.01088216,0.020523788],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99712485,0.0006470454,0.00032424746,0.0006413305,0.0011199033,0.00014252917],"domain_scores_gemma":[0.9903044,0.004904723,0.0003179323,0.002290603,0.0018932597,0.0002891206],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030874999,0.0012488135,0.0012921318,0.005050283,0.0016294935,0.0037515033,0.001717217,0.0012227881,0.016927073],"category_scores_gemma":[0.011454782,0.0014162746,0.0010585337,0.0039394074,0.0014078283,0.007254064,0.0042535183,0.0020458894,0.009417556],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010524458,0.00027869703,0.008613019,0.0027195679,0.0002758314,0.0013989125,0.004936046,0.0022498872,0.1021124,0.014094346,0.055854004,0.80641484],"study_design_scores_gemma":[0.000171499,0.00032866644,0.022531606,0.0005987279,0.0002072395,0.004225624,0.002799914,0.04259018,0.16573375,0.016368467,0.74408275,0.00036154105],"about_ca_topic_score_codex":0.0096138595,"about_ca_topic_score_gemma":0.017499086,"teacher_disagreement_score":0.016927073,"about_ca_system_score_codex":0.0009005711,"about_ca_system_score_gemma":0.0024789176,"threshold_uncertainty_score":0.056626678},"labels":[],"label_agreement":null},{"id":"W2099242427","doi":"","title":"Entity-Based Local Coherence Modelling Using Topological Fields","year":2010,"lang":"en","type":"article","venue":"Meeting of the Association for Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Coherence (philosophical gambling strategy); Computer science; Sentence; Component (thermodynamics); Topology (electrical circuits); Grid; Natural language; Natural language processing; Field (mathematics); Artificial intelligence; Language model; Theoretical computer science; Mathematics; Physics; Pure mathematics","score_opus":0.022983696173067523,"score_gpt":0.29087224003836043,"score_spread":0.2678885438652929,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2099242427","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05698764,0.0001418149,0.9385676,0.0002056856,0.000022116114,0.00009298949,0.00037537076,0.0016642377,0.001942639],"genre_scores_gemma":[0.7269127,0.00013371906,0.27028677,0.00004423334,0.000019514942,0.00013022593,0.00083434774,0.00021498858,0.0014235975],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99920887,0.00037340782,0.00005538167,0.00016704004,0.00015136918,0.000043908465],"domain_scores_gemma":[0.9961398,0.0026373726,0.00040349178,0.00040049403,0.000333342,0.00008550796],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017639977,0.0004992763,0.0004229429,0.0015549066,0.00051336654,0.0014534115,0.0009400075,0.00063291605,0.002955499],"category_scores_gemma":[0.006124229,0.00039937923,0.00082030916,0.001048249,0.0007518596,0.00425284,0.0010516379,0.00075954566,0.00046960925],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00046246158,0.00014950981,0.00595338,0.00033328493,0.00009735518,0.00044769954,0.0012973804,0.61812544,0.012993755,0.23496929,0.0028459348,0.1223245],"study_design_scores_gemma":[0.000030966305,0.00005028101,0.00045027208,0.000012149997,0.000024899786,0.00004371049,0.0000745316,0.95421165,0.0035443283,0.038835883,0.0027046474,0.000016789829],"about_ca_topic_score_codex":0.0041893753,"about_ca_topic_score_gemma":0.0061545493,"teacher_disagreement_score":0.0041893753,"about_ca_system_score_codex":0.00084597623,"about_ca_system_score_gemma":0.0007361128,"threshold_uncertainty_score":0.009887099},"labels":[],"label_agreement":null},{"id":"W2099779943","doi":"10.3115/1621969.1621986","title":"SemEval-2010 task 8","year":2009,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":497,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"SemEval; Computer science; Task (project management); Testbed; Natural language processing; Artificial intelligence; Information extraction; Natural language; Semantics (computer science); Information retrieval; World Wide Web; Programming language","score_opus":0.009039731999430325,"score_gpt":0.26383693222791427,"score_spread":0.25479720022848396,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2099779943","genre_codex":"dataset","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12008913,0.007617997,0.07466604,0.008921039,0.007140889,0.004141478,0.5696771,0.07509587,0.13265045],"genre_scores_gemma":[0.09670084,0.0006602401,0.10121535,0.0020683245,0.00047211567,0.0024322965,0.763127,0.0026739447,0.030649956],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99415755,0.0018450158,0.0005736018,0.0017578941,0.0011214572,0.0005444985],"domain_scores_gemma":[0.9934604,0.0026620626,0.0003310729,0.0017993418,0.0011639881,0.00058293476],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0051989458,0.0040772744,0.0026944038,0.0036024197,0.0026340843,0.004150004,0.0042881873,0.0057903505,0.03894611],"category_scores_gemma":[0.013134541,0.0007238045,0.0024977126,0.0024744668,0.0012818621,0.006731471,0.0065026926,0.0038650963,0.03491705],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009239333,0.0007528941,0.0026625118,0.0022360138,0.0001892467,0.0007040164,0.00037377767,0.0020356574,0.004587538,0.006513066,0.84715515,0.13186625],"study_design_scores_gemma":[0.0010463604,0.0007268325,0.012028669,0.0007276273,0.00019573516,0.0039272564,0.0024262108,0.054223124,0.026054094,0.030529348,0.86782867,0.0002860241],"about_ca_topic_score_codex":0.008651507,"about_ca_topic_score_gemma":0.018330203,"teacher_disagreement_score":0.03894611,"about_ca_system_score_codex":0.0022367449,"about_ca_system_score_gemma":0.0041481047,"threshold_uncertainty_score":0.13028777},"labels":[],"label_agreement":null},{"id":"W2099853762","doi":"","title":"ERSS at TAC 2008","year":2008,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Readability; Computer science; Ranking (information retrieval); Fell; Artificial intelligence; Geography; Cartography; Programming language","score_opus":0.008361780675118412,"score_gpt":0.24803010676012227,"score_spread":0.23966832608500385,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2099853762","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06276719,0.0018200489,0.08664542,0.022372931,0.0153187495,0.0023513425,0.13751663,0.22436506,0.44684267],"genre_scores_gemma":[0.11343153,0.00046564228,0.08508947,0.004865189,0.0021815673,0.0008286426,0.36082226,0.024278088,0.40803757],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9920083,0.0018804901,0.00037944582,0.0009171829,0.003560232,0.0012542531],"domain_scores_gemma":[0.97679126,0.0015111805,0.0004404084,0.00405897,0.0146674905,0.002530749],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.015421237,0.0015177369,0.001350392,0.0047242283,0.002455656,0.007544163,0.0030268189,0.0028366297,0.09156742],"category_scores_gemma":[0.016143622,0.0008566811,0.0012103722,0.0029411833,0.0009595522,0.0060463296,0.0027243788,0.0045288983,0.12395875],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005569605,0.00023577716,0.0012850764,0.00009076696,0.000032930788,0.00013676299,0.00019926149,0.0006424757,0.0032058782,0.0027691445,0.9452361,0.04560891],"study_design_scores_gemma":[0.00038007618,0.00042905213,0.004387847,0.000083109655,0.00004651708,0.00012773406,0.00031731016,0.007171949,0.005411271,0.0044505005,0.9771024,0.000092258895],"about_ca_topic_score_codex":0.039580368,"about_ca_topic_score_gemma":0.052584156,"teacher_disagreement_score":0.09156742,"about_ca_system_score_codex":0.0039993473,"about_ca_system_score_gemma":0.00380371,"threshold_uncertainty_score":0.3063236},"labels":[],"label_agreement":null},{"id":"W2099942723","doi":"10.3115/1118108.1118111","title":"A web-based instructional platform for constraint-based grammar formalisms and parsing","year":2002,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":49,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Bundesministerium für Bildung und Forschung","keywords":"Computer science; Parsing; Rotation formalisms in three dimensions; Grammar; Natural language processing; Programming language; Artificial intelligence; Set (abstract data type); Constraint (computer-aided design); Feature (linguistics); Core (optical fiber); Rule-based machine translation; Linguistics","score_opus":0.02328502649735071,"score_gpt":0.24817853717122465,"score_spread":0.22489351067387395,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2099942723","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0038549106,0.00005295174,0.94131684,0.0007274608,0.000072012714,0.00022105685,0.00055545237,0.042386346,0.010813018],"genre_scores_gemma":[0.061983746,0.00017020904,0.917809,0.00049156876,0.00008526394,0.00063366623,0.0027855376,0.0027092474,0.0133318035],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99921095,0.00028679203,0.000061151746,0.00014451909,0.00023302266,0.00006358061],"domain_scores_gemma":[0.99588937,0.0019759967,0.00013632196,0.0010365653,0.00048468806,0.00047710817],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002144154,0.0005957695,0.00071796885,0.0013226495,0.00090050756,0.002856807,0.003571471,0.001882253,0.027421115],"category_scores_gemma":[0.007600045,0.0008931486,0.0007695628,0.0010243293,0.0010286847,0.007004718,0.0035773013,0.0037102588,0.01291221],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037695482,0.0024734447,0.0020202606,0.00038946627,0.00009415709,0.001008844,0.0015809536,0.035297524,0.018612972,0.2774162,0.09572216,0.56500703],"study_design_scores_gemma":[0.0001546542,0.00023368896,0.00084543077,0.0002013648,0.00005528298,0.0006396672,0.00027363133,0.43887368,0.028017163,0.24604152,0.28452906,0.00013481078],"about_ca_topic_score_codex":0.0018466815,"about_ca_topic_score_gemma":0.0025312172,"teacher_disagreement_score":0.027421115,"about_ca_system_score_codex":0.0006153675,"about_ca_system_score_gemma":0.0023292545,"threshold_uncertainty_score":0.0917328},"labels":[],"label_agreement":null},{"id":"W2100664567","doi":"10.3115/v1/p15-1001","title":"On Using Very Large Target Vocabulary for Neural Machine Translation","year":2015,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":872,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"Samsung; Compute Canada; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs; Canadian Institute for Advanced Research","keywords":"Machine translation; Computer science; Vocabulary; Artificial intelligence; Natural language processing; Joint (building); Computational linguistics; Artificial neural network; Translation (biology); Volume (thermodynamics); Association (psychology); Speech recognition; Linguistics; Engineering; Philosophy","score_opus":0.044902221950209355,"score_gpt":0.312351224200445,"score_spread":0.2674490022502356,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2100664567","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.039008822,0.005613179,0.91690993,0.0014864878,0.001407467,0.00041648195,0.0018594996,0.015000158,0.018297946],"genre_scores_gemma":[0.3509439,0.0027087368,0.6162224,0.0014186704,0.000823477,0.00070909405,0.0114203915,0.0017627724,0.0139905615],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9962769,0.0018496317,0.00038283193,0.00069214334,0.0005176926,0.0002808275],"domain_scores_gemma":[0.99320143,0.0031574527,0.00016957909,0.0019220073,0.0014253997,0.00012411973],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033577892,0.0014233569,0.0019279445,0.0030786626,0.001615439,0.0024426116,0.0024580539,0.002208272,0.0116190715],"category_scores_gemma":[0.012189166,0.00069962814,0.0010820631,0.0037721328,0.0010677471,0.008043237,0.004609047,0.0022335781,0.012357539],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00079844065,0.00027906577,0.0006808732,0.00046991295,0.00018910268,0.00032662455,0.0002991986,0.019703645,0.020482913,0.017298743,0.041657053,0.89781433],"study_design_scores_gemma":[0.00037769266,0.00044482152,0.00091171573,0.00015533876,0.00022997768,0.0005572947,0.00066310266,0.7870588,0.027439816,0.14315197,0.038892083,0.000117441734],"about_ca_topic_score_codex":0.006006271,"about_ca_topic_score_gemma":0.011580807,"teacher_disagreement_score":0.0116190715,"about_ca_system_score_codex":0.00061613275,"about_ca_system_score_gemma":0.0014079825,"threshold_uncertainty_score":0.03886962},"labels":[],"label_agreement":null},{"id":"W2101227875","doi":"10.7202/001924ar","title":"TRADEX, un système de traduction de télex","year":2002,"lang":"fr","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Philosophy; Political science","score_opus":0.05028938805609144,"score_gpt":0.2718734971493086,"score_spread":0.22158410909321713,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2101227875","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06978381,0.002940258,0.7418412,0.0024649454,0.0007753363,0.00084428163,0.006401141,0.113600716,0.061348304],"genre_scores_gemma":[0.29098433,0.0023437676,0.5607205,0.0013464729,0.00037798128,0.0007908674,0.013389905,0.012632276,0.11741394],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99604416,0.00070437544,0.00048398686,0.0010720001,0.0014956591,0.00019989094],"domain_scores_gemma":[0.9912589,0.0035476561,0.00059563876,0.0028327547,0.001571613,0.00019340482],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0046003377,0.0010680865,0.0010046809,0.0025914724,0.001968461,0.0054400386,0.0027134495,0.002085848,0.023545789],"category_scores_gemma":[0.013155511,0.001379969,0.0014962286,0.0028082256,0.0031189453,0.01006555,0.0033763412,0.0021613366,0.0070878295],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0029298468,0.00033484254,0.013700037,0.0029464911,0.0003475989,0.0018414252,0.009831374,0.0151309315,0.061307833,0.1644066,0.08732607,0.6398969],"study_design_scores_gemma":[0.00046262614,0.00076251896,0.007384483,0.00053282484,0.00034407398,0.0025652612,0.0016978058,0.105692066,0.1097319,0.06209383,0.7084245,0.00030825334],"about_ca_topic_score_codex":0.006842563,"about_ca_topic_score_gemma":0.005638814,"teacher_disagreement_score":0.023545789,"about_ca_system_score_codex":0.0021938267,"about_ca_system_score_gemma":0.0029195996,"threshold_uncertainty_score":0.07876855},"labels":[],"label_agreement":null},{"id":"W2101481293","doi":"10.1017/s1351324905003694","title":"Segmenting documents by stylistic character","year":2005,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":104,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"IBM (Canada); University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Character (mathematics); Bigram; Natural language processing; Artificial intelligence; Vocabulary; Baseline (sea); Artificial neural network; Market segmentation; Information retrieval; Linguistics; Trigram","score_opus":0.002330387928883056,"score_gpt":0.22306770749537055,"score_spread":0.2207373195664875,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2101481293","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.80492085,0.0014039766,0.17436254,0.000438691,0.00022833256,0.0006644359,0.004186722,0.005508958,0.008285484],"genre_scores_gemma":[0.75067335,0.00046296327,0.2357334,0.00006779058,0.00007567023,0.00012488423,0.006895785,0.00030616412,0.005659995],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99942136,0.00011499055,0.000074055235,0.00022136407,0.00010091105,0.00006726697],"domain_scores_gemma":[0.996298,0.0017672394,0.00041560738,0.000437721,0.00094255776,0.000138972],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00065757526,0.0007604807,0.00043019973,0.0025816236,0.00059237814,0.0014711111,0.0004751205,0.00069822604,0.0025702042],"category_scores_gemma":[0.0049259616,0.00024102426,0.00033544863,0.0018962693,0.00026854122,0.0015613508,0.00040190015,0.0007431085,0.0018414679],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014663542,0.00024349644,0.019756313,0.0005982485,0.00007541082,0.00054131116,0.0015391299,0.01635694,0.15219797,0.002612825,0.007200539,0.7974115],"study_design_scores_gemma":[0.00015091925,0.0005785497,0.03348251,0.00013420911,0.00015461729,0.0011797794,0.0026011274,0.6907223,0.2141464,0.013176785,0.043532558,0.00014032517],"about_ca_topic_score_codex":0.0047638323,"about_ca_topic_score_gemma":0.008074566,"teacher_disagreement_score":0.0047638323,"about_ca_system_score_codex":0.000708891,"about_ca_system_score_gemma":0.000600246,"threshold_uncertainty_score":0.009472251},"labels":[],"label_agreement":null},{"id":"W2101566153","doi":"","title":"The Trouble with SMT Consistency","year":2012,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":44,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Computer science; Consistency (knowledge bases); Natural language processing; Terminology; Phrase; Translation (biology); Context (archaeology); Sentence; Artificial intelligence; Machine translation; Textual entailment; Information retrieval; Linguistics; Logical consequence","score_opus":0.012572579888486259,"score_gpt":0.2487285728406088,"score_spread":0.23615599295212256,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2101566153","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.031003518,0.0028895573,0.9196067,0.014044328,0.0018182632,0.00030984374,0.0017608487,0.011891018,0.016676],"genre_scores_gemma":[0.57227784,0.0014176834,0.39718848,0.006104962,0.0013725755,0.0007505276,0.003846938,0.007735282,0.009305672],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.90642697,0.056069296,0.006836427,0.013002452,0.016418906,0.0012459821],"domain_scores_gemma":[0.75667936,0.16429374,0.008317096,0.050334405,0.019475756,0.0008996133],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.041957002,0.001683425,0.0023982762,0.001840603,0.0022453868,0.0053780545,0.0038660401,0.003433888,0.007840539],"category_scores_gemma":[0.2534095,0.0016592955,0.0014200981,0.0037221594,0.003576557,0.010247606,0.0054873694,0.007694049,0.0064319135],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017874471,0.00028822894,0.019766934,0.0030888221,0.0010191184,0.001796837,0.007208863,0.05981677,0.03453051,0.15919195,0.09448508,0.6170194],"study_design_scores_gemma":[0.00031307337,0.00040320205,0.007837534,0.0010604694,0.00039185464,0.0036618377,0.0028257777,0.2880538,0.056789994,0.51374084,0.124544926,0.0003766791],"about_ca_topic_score_codex":0.002450178,"about_ca_topic_score_gemma":0.0026418325,"teacher_disagreement_score":0.041957002,"about_ca_system_score_codex":0.0019993968,"about_ca_system_score_gemma":0.0027668604,"threshold_uncertainty_score":0.22189254},"labels":[],"label_agreement":null},{"id":"W2102451297","doi":"10.1145/1068009.1068352","title":"Use of a genetic algorithm in brill's transformation-based part-of-speech tagger","year":2005,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Brill; Computer science; Transformation (genetics); Ranking (information retrieval); A priori and a posteriori; Part-of-speech tagging; Algorithm; Genetic algorithm; Point (geometry); Computation; Artificial intelligence; Part of speech; Natural language processing; Machine learning; Mathematics","score_opus":0.018982386055125856,"score_gpt":0.2619017559631206,"score_spread":0.24291936990799473,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2102451297","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017430067,0.000057472775,0.97699445,0.00016285032,0.000048723137,0.00010701037,0.000037282443,0.0015470985,0.0036151153],"genre_scores_gemma":[0.174778,0.00009315344,0.8200484,0.00030444885,0.000020380829,0.00023522339,0.00015136426,0.00030811093,0.0040609506],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988464,0.00039437826,0.000056923833,0.00030559764,0.0003204778,0.00007628598],"domain_scores_gemma":[0.9989348,0.00055553927,0.00005876249,0.00021156805,0.00020964752,0.000029639497],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001743456,0.0006967636,0.00072391564,0.0008539692,0.0007935465,0.0012261015,0.001358134,0.0015698681,0.0016779278],"category_scores_gemma":[0.004028265,0.0004966813,0.00074069126,0.00091055175,0.00110831,0.0013507606,0.0008805609,0.0015664726,0.00097223907],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025072903,0.0003288001,0.0039420035,0.000114654584,0.00017249848,0.0004192287,0.00052214693,0.37344012,0.03510439,0.043562643,0.0026668247,0.539476],"study_design_scores_gemma":[0.00006266738,0.0001671291,0.00066554634,0.000025714984,0.00008418611,0.00020094567,0.000049375147,0.95524466,0.021523083,0.014407389,0.007515292,0.0000539727],"about_ca_topic_score_codex":0.0051128995,"about_ca_topic_score_gemma":0.005732771,"teacher_disagreement_score":0.0051128995,"about_ca_system_score_codex":0.0010631492,"about_ca_system_score_gemma":0.0010656664,"threshold_uncertainty_score":0.010166228},"labels":[],"label_agreement":null},{"id":"W2102867064","doi":"10.7202/014498ar","title":"Actes de langage et relations rhétoriques en dialogue homme-machine","year":2007,"lang":"fr","type":"article","venue":"Revue de l’Université de Moncton","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Philosophy","score_opus":0.01145731348873009,"score_gpt":0.24561502518143277,"score_spread":0.2341577116927027,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2102867064","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011904432,0.0020068747,0.968366,0.0017649924,0.00021757666,0.000113544294,0.00021615787,0.0007001992,0.014710277],"genre_scores_gemma":[0.3155442,0.002315045,0.66196454,0.0006360081,0.0003222095,0.0004600567,0.00074938394,0.0005464981,0.017462121],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9900359,0.006221733,0.00049414526,0.001335762,0.0016090285,0.00030344506],"domain_scores_gemma":[0.98851365,0.008689615,0.00064113457,0.0011235503,0.0008270165,0.00020504344],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0064891577,0.0012689668,0.0008961632,0.0029275184,0.00224401,0.008309438,0.0014860831,0.0030320713,0.007768221],"category_scores_gemma":[0.015963983,0.000912412,0.0014577946,0.002052496,0.008683527,0.016091892,0.0035035652,0.0037170863,0.0019747391],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011474167,0.000026491094,0.00069023913,0.00032728782,0.00003329805,0.00027847532,0.007855241,0.003805501,0.0038314492,0.93959147,0.0014172625,0.042028576],"study_design_scores_gemma":[0.000056471094,0.000100207195,0.00092457974,0.00032052223,0.00006582939,0.00079369626,0.0036908984,0.05015764,0.009244792,0.8041731,0.13036549,0.00010679551],"about_ca_topic_score_codex":0.0034392476,"about_ca_topic_score_gemma":0.0025093607,"teacher_disagreement_score":0.008309438,"about_ca_system_score_codex":0.002270741,"about_ca_system_score_gemma":0.0017512594,"threshold_uncertainty_score":0.034318328},"labels":[],"label_agreement":null},{"id":"W2102930870","doi":"10.14705/rpnet.2014.000200","title":"Investigating an open methodology for designing domain-specific language collections","year":2014,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"University of Waikato; Concordia University; Queen Mary University of London","keywords":"Suite; Computer science; Metadata; World Wide Web; Open domain; Domain (mathematical analysis); Open educational resources; Digital library; Information retrieval; Question answering; Linguistics","score_opus":0.11399007552225345,"score_gpt":0.3774762540352669,"score_spread":0.2634861785130135,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2102930870","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007320558,0.000098580196,0.9837173,0.0005687075,0.000049538427,0.0010426174,0.00008205152,0.0005803644,0.0065402137],"genre_scores_gemma":[0.04033807,0.000090676476,0.95445764,0.0002039855,0.000024291186,0.0013510337,0.00021231294,0.00027782138,0.0030441137],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.95932233,0.026171269,0.0024559242,0.0051514204,0.006011735,0.0008873917],"domain_scores_gemma":[0.9436855,0.029123489,0.0025861342,0.017704664,0.0051451535,0.0017550435],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03889487,0.0010983967,0.0007065221,0.0030342643,0.0035964414,0.009269191,0.0046577,0.0024417727,0.009683322],"category_scores_gemma":[0.05480998,0.0016006054,0.002510576,0.0022400708,0.009743688,0.015877375,0.012393084,0.003240568,0.0023243022],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023661605,0.0008944562,0.0036551752,0.001607961,0.00015564226,0.0006738218,0.03202757,0.0098220315,0.017167697,0.6386822,0.0039073997,0.2911695],"study_design_scores_gemma":[0.0003331757,0.00096743857,0.0015848621,0.00093815173,0.00025652762,0.0013698584,0.01682154,0.04940572,0.036564726,0.5216007,0.36994576,0.00021153917],"about_ca_topic_score_codex":0.0015031062,"about_ca_topic_score_gemma":0.0027440314,"teacher_disagreement_score":0.03889487,"about_ca_system_score_codex":0.002685642,"about_ca_system_score_gemma":0.007420473,"threshold_uncertainty_score":0.20569825},"labels":[],"label_agreement":null},{"id":"W2103419120","doi":"10.1007/978-3-642-01818-3_8","title":"Statistical Parsing with Context-Free Filtering Grammar","year":2009,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Parsing; Grammar; Context-free grammar; Artificial intelligence; Natural language processing; Context (archaeology); Programming language; Linguistics","score_opus":0.0131304933314721,"score_gpt":0.251016197908207,"score_spread":0.2378857045767349,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2103419120","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0021467088,0.00020511296,0.9890469,0.00014553484,0.000062662795,0.00003425631,0.00038160267,0.005392115,0.0025850763],"genre_scores_gemma":[0.12287785,0.000567832,0.859538,0.00031275972,0.00021306657,0.00019934942,0.0044942684,0.00521015,0.006586636],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9977507,0.0006556429,0.00021731235,0.0006381457,0.00059276924,0.00014543861],"domain_scores_gemma":[0.9953517,0.002938786,0.00014944958,0.0009905848,0.00050823204,0.00006128651],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023939745,0.0012920371,0.0019080525,0.002479651,0.0012388243,0.0025428315,0.0026566477,0.001791148,0.0092525],"category_scores_gemma":[0.0066931234,0.0016159844,0.0024793255,0.0033380648,0.0019155444,0.0040559326,0.0019146403,0.0024472696,0.0054986333],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015834537,0.00010545129,0.0010171352,0.00051263673,0.0001653094,0.0005991848,0.00042782334,0.06599164,0.01233737,0.39275712,0.027585115,0.4983429],"study_design_scores_gemma":[0.000029715538,0.000034541514,0.00038908556,0.00005968944,0.00008690305,0.00032822118,0.000036890706,0.24294886,0.012460778,0.7226654,0.020885523,0.000074368916],"about_ca_topic_score_codex":0.0030957314,"about_ca_topic_score_gemma":0.0036650891,"teacher_disagreement_score":0.0092525,"about_ca_system_score_codex":0.0010029571,"about_ca_system_score_gemma":0.0022082124,"threshold_uncertainty_score":0.030952632},"labels":[],"label_agreement":null},{"id":"W2103436282","doi":"10.7202/003227ar","title":"Technological and Scientific Hebrew Terminology","year":2002,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Terminology; Hebrew; Transliteration; Scientific terminology; Computer science; Lexicography; Linguistics; Lexicographical order; Service (business); Artificial intelligence; Natural language processing; Business; Mathematics; Philosophy","score_opus":0.047892033904075516,"score_gpt":0.269169772279066,"score_spread":0.2212777383749905,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2103436282","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0069639855,0.03860175,0.051889636,0.0125263315,0.011858137,0.00023385041,0.0017857843,0.00037960242,0.8757609],"genre_scores_gemma":[0.24784128,0.07905039,0.14870429,0.014945102,0.017559465,0.0018432365,0.014163043,0.001704537,0.47418866],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9964748,0.0012666046,0.0005178257,0.0003978451,0.0010217708,0.0003210839],"domain_scores_gemma":[0.9965145,0.0012030331,0.0005309953,0.00065867865,0.0008209311,0.00027185943],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003132585,0.0010335285,0.00063986465,0.00914863,0.0033848223,0.006073309,0.0011888795,0.001265437,0.021125074],"category_scores_gemma":[0.0075388276,0.00026710998,0.0004384943,0.010634992,0.005629694,0.008161187,0.0025264577,0.0029285958,0.012177779],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000036624046,0.000020888061,0.00035309876,0.00020322869,0.0000042457723,0.00010732568,0.003124413,0.00015660116,0.0005695582,0.8626116,0.06294522,0.069867216],"study_design_scores_gemma":[0.000002863266,0.000014095089,0.00042093007,0.00020625594,0.0000026393018,0.0002107605,0.0007850791,0.0001309698,0.00019368896,0.048451703,0.94957274,0.000008284013],"about_ca_topic_score_codex":0.0025796203,"about_ca_topic_score_gemma":0.0029968312,"teacher_disagreement_score":0.021125074,"about_ca_system_score_codex":0.0032851384,"about_ca_system_score_gemma":0.00311367,"threshold_uncertainty_score":0.070670426},"labels":[],"label_agreement":null},{"id":"W2103706669","doi":"10.5430/air.v2n3p35","title":"The role of statistical and semantic features in single-document extractive summarization","year":2013,"lang":"en","type":"article","venue":"Artificial Intelligence Research","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Natural language processing; Automatic summarization; Anaphora (linguistics); Sentence; Context (archaeology); Artificial intelligence; Word (group theory); Feature (linguistics); Term (time); Representation (politics); Resolution (logic); Information retrieval; Linguistics","score_opus":0.04453799708544237,"score_gpt":0.37893299108909595,"score_spread":0.3343949940036536,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2103706669","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.23839656,0.008353319,0.7343176,0.0011340904,0.0002589994,0.00050504913,0.0027817125,0.008954512,0.005298132],"genre_scores_gemma":[0.65186524,0.0016569942,0.3402133,0.00011254425,0.0003458774,0.000274612,0.0037382077,0.0004298455,0.0013633559],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99764156,0.0009902577,0.00027139168,0.00034644076,0.00064819556,0.00010215442],"domain_scores_gemma":[0.9850487,0.010808391,0.001104668,0.00090179034,0.001986419,0.00015000498],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004319478,0.0010569682,0.0013494549,0.004125265,0.0005470625,0.002511965,0.0008260144,0.00058458914,0.0021623312],"category_scores_gemma":[0.014164694,0.00030671555,0.0008977648,0.0028001068,0.00047724388,0.0042386134,0.00054897816,0.00076605694,0.0012151558],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008125384,0.00026325995,0.0051322845,0.0011267543,0.00026959062,0.000110155146,0.00020983408,0.011677939,0.04098809,0.0019864398,0.002207711,0.9352154],"study_design_scores_gemma":[0.00022431124,0.0042265416,0.07057118,0.00048589538,0.0020290902,0.0011314248,0.0016276087,0.66812956,0.19559638,0.026722651,0.028808655,0.00044666792],"about_ca_topic_score_codex":0.0010908346,"about_ca_topic_score_gemma":0.0015853234,"teacher_disagreement_score":0.004319478,"about_ca_system_score_codex":0.00043648249,"about_ca_system_score_gemma":0.0006376034,"threshold_uncertainty_score":0.022843838},"labels":[],"label_agreement":null},{"id":"W2103759455","doi":"","title":"AUTOMATIC TERM EXTRACTION AND DOCUMENT SIMILARITY IN SPECIAL TEXT CORPORA","year":2003,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":55,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Computer science; Term (time); Natural language processing; Collocation (remote sensing); Artificial intelligence; Similarity (geometry); Representation (politics); Information retrieval; Word (group theory); Lexical analysis; Text corpus; Vocabulary; Vector space model; Terminology; Computational linguistics; Linguistics; Machine learning","score_opus":0.011401864290806104,"score_gpt":0.2793095136083764,"score_spread":0.26790764931757033,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2103759455","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15643544,0.0056734476,0.8043711,0.0005865732,0.0005908013,0.0011695813,0.0069058626,0.016416227,0.007850917],"genre_scores_gemma":[0.23369312,0.0013741457,0.73739874,0.000111967114,0.00037166334,0.0012018591,0.020694956,0.00094500143,0.0042086067],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9904675,0.0025983453,0.0015781448,0.0020048635,0.0029880656,0.00036300515],"domain_scores_gemma":[0.9760049,0.012807055,0.002427423,0.0032178161,0.0052776057,0.00026525423],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0051961937,0.0011721813,0.0023431936,0.019318987,0.0014731248,0.0034914166,0.0018550758,0.001436653,0.0033707079],"category_scores_gemma":[0.035160728,0.00069519604,0.0018985042,0.01658746,0.0008843758,0.005695658,0.0024560462,0.0013014947,0.0036516055],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006958409,0.0002739255,0.006403803,0.001526788,0.0003180153,0.0005658649,0.0008045807,0.011968352,0.051814515,0.009736365,0.01767194,0.89822],"study_design_scores_gemma":[0.000372011,0.0010360712,0.04916751,0.0003695052,0.00060826045,0.0036533638,0.0016115069,0.6489211,0.15535711,0.048887737,0.0895791,0.0004367455],"about_ca_topic_score_codex":0.0019136074,"about_ca_topic_score_gemma":0.00196226,"teacher_disagreement_score":0.019318987,"about_ca_system_score_codex":0.0013075162,"about_ca_system_score_gemma":0.0013528307,"threshold_uncertainty_score":0.027480483},"labels":[],"label_agreement":null},{"id":"W2104017484","doi":"10.1504/ijeb.2005.007276","title":"Natural language asymmetry and internet infrastructures","year":2005,"lang":"en","type":"article","venue":"International Journal of Electronic Business","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"Social Sciences and Humanities Research Council of Canada; Royal Society of Canada","keywords":"Computer science; The Internet; Information retrieval; Natural language; Identification (biology); Search engine; Natural (archaeology); Information extraction; Natural language processing; World Wide Web","score_opus":0.0024706993435872973,"score_gpt":0.2501279856897731,"score_spread":0.24765728634618583,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2104017484","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14199625,0.0019262813,0.8304274,0.0029760094,0.00009846775,0.00019356002,0.00058932544,0.0029501729,0.01884264],"genre_scores_gemma":[0.82758635,0.0009191983,0.16724294,0.0004887364,0.00020698704,0.00019981009,0.00076134625,0.00020477253,0.0023898133],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9955894,0.001642369,0.0004500623,0.0005284777,0.0015243327,0.00026541503],"domain_scores_gemma":[0.9762763,0.015850047,0.0031646446,0.0033030775,0.0012133458,0.00019256686],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025902758,0.0004263814,0.00056965376,0.0023800714,0.0008712939,0.002885833,0.0010946754,0.00094662554,0.0036730275],"category_scores_gemma":[0.022667853,0.00040882282,0.000508622,0.0021460464,0.0019726108,0.0077929995,0.0020032132,0.0015001448,0.00095343776],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00079246284,0.00018882449,0.007237275,0.0006155551,0.00009221547,0.0012814701,0.0021253263,0.02573856,0.038646463,0.586645,0.004642111,0.33199474],"study_design_scores_gemma":[0.0000706663,0.00013478046,0.0032213167,0.00012797161,0.00012530686,0.0022969584,0.0005637989,0.1863232,0.04474573,0.729392,0.032882296,0.000116047384],"about_ca_topic_score_codex":0.0010166836,"about_ca_topic_score_gemma":0.0006014678,"teacher_disagreement_score":0.0036730275,"about_ca_system_score_codex":0.0011000464,"about_ca_system_score_gemma":0.0010549854,"threshold_uncertainty_score":0.013698816},"labels":[],"label_agreement":null},{"id":"W2104123854","doi":"10.1109/tai.1996.560793","title":"Natural language edit controls: constrained natural language devices in user interfaces","year":2005,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Natural language user interface; Natural language programming; Natural language; Component (thermodynamics); Interface (matter); User interface; Graphical user interface; Natural user interface; Personalization; Natural (archaeology); Human–computer interaction; Programming language; Natural language processing; Artificial intelligence; User interface design; Universal Networking Language; World Wide Web; Operating system","score_opus":0.005380282281301441,"score_gpt":0.2711228891848221,"score_spread":0.26574260690352064,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2104123854","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0067806426,0.001439049,0.97332,0.00046295646,0.00018160958,0.00015394318,0.0001699771,0.002863171,0.014628634],"genre_scores_gemma":[0.2960264,0.0026912354,0.67468387,0.0011581506,0.0004410926,0.0012656867,0.0008283898,0.0015614112,0.021343801],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99585307,0.0012459165,0.00037251698,0.0008683701,0.0014365474,0.00022361732],"domain_scores_gemma":[0.9945425,0.0032591335,0.00036035274,0.0011984195,0.00044237156,0.00019733465],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002677664,0.00105387,0.0006862708,0.0008931324,0.0007503529,0.0050170613,0.002620688,0.0019947933,0.008913995],"category_scores_gemma":[0.012306465,0.0009006535,0.00067319215,0.0009854838,0.00444473,0.010379416,0.0027699724,0.0018800731,0.0016442764],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022116811,0.00008036239,0.00055895397,0.0005570754,0.000038006943,0.00062275113,0.0022880326,0.0052396893,0.019737894,0.8099331,0.00739942,0.15332347],"study_design_scores_gemma":[0.00011125128,0.0002415264,0.00093653623,0.00041277977,0.000096366944,0.0016848956,0.00056537346,0.072440326,0.029628843,0.5246246,0.36908144,0.00017598136],"about_ca_topic_score_codex":0.001698302,"about_ca_topic_score_gemma":0.001474741,"teacher_disagreement_score":0.008913995,"about_ca_system_score_codex":0.00075421913,"about_ca_system_score_gemma":0.0008566876,"threshold_uncertainty_score":0.029820263},"labels":[],"label_agreement":null},{"id":"W2104326771","doi":"10.1002/meet.14504701411","title":"Human abstracts, machine summaries, cyborg solutions?","year":2010,"lang":"en","type":"article","venue":"Proceedings of the American Society for Information Science and Technology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Automatic summarization; Representativeness heuristic; Computer science; Information retrieval; Complement (music); Natural language processing; Sample (material); Data science; Artificial intelligence; Psychology","score_opus":0.009314683643051465,"score_gpt":0.2658900981913442,"score_spread":0.2565754145482927,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2104326771","genre_codex":"empirical","genre_gemma":"commentary","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"commentary","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.44092834,0.100290254,0.21151161,0.050915554,0.004441107,0.0009044026,0.006098334,0.01253821,0.17237213],"genre_scores_gemma":[0.8497722,0.013783812,0.11755445,0.0017779301,0.0018204568,0.00040415823,0.0026898272,0.00058617274,0.0116109755],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.986198,0.009124363,0.0012345993,0.00080389704,0.0024018972,0.00023715528],"domain_scores_gemma":[0.9282859,0.049899958,0.00968489,0.0039957105,0.007120265,0.0010133133],"candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.013116394,0.0005525182,0.00055550534,0.007095246,0.0008837226,0.00802648,0.0006864053,0.0010323437,0.0097674485],"category_scores_gemma":[0.08794497,0.00019994445,0.00032147393,0.0060224654,0.0010630261,0.008976151,0.0017200223,0.0006600391,0.0027047032],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011811248,0.0002559968,0.0127549255,0.003991466,0.00020541456,0.00043260932,0.012733037,0.0020475048,0.004442311,0.04757689,0.04447335,0.8699055],"study_design_scores_gemma":[0.00047809732,0.0027247102,0.04766611,0.005621377,0.00067816547,0.0035907417,0.052212648,0.023725716,0.018768577,0.15908118,0.68515486,0.00029787008],"about_ca_topic_score_codex":0.0005607102,"about_ca_topic_score_gemma":0.00052635744,"teacher_disagreement_score":0.9919735,"about_ca_system_score_codex":0.00085035415,"about_ca_system_score_gemma":0.0010606308,"threshold_uncertainty_score":0.06936693},"labels":[],"label_agreement":null},{"id":"W2104395554","doi":"","title":"Accurate Context-Free Parsing with Combinatory Categorial Grammar","year":2010,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":40,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Combinatory categorial grammar; Computer science; Parsing; Categorial grammar; Natural language processing; Context (archaeology); Grammar; Programming language; Combinatory logic; Artificial intelligence; Dependency (UML); Operator-precedence grammar; Locative case; Parseval's theorem; Link grammar; Linguistics; Generative grammar; Mildly context-sensitive grammar formalism; Attribute grammar; Phrase structure rules; Mathematics","score_opus":0.008092646695743997,"score_gpt":0.23838560575317475,"score_spread":0.23029295905743075,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2104395554","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16995691,0.0011407266,0.78445697,0.00093474565,0.00026398897,0.00014200222,0.0031413557,0.03013407,0.009829178],"genre_scores_gemma":[0.63930994,0.00029896357,0.34563974,0.0003249375,0.00009880345,0.00012137153,0.007054252,0.0031324795,0.0040196273],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9946221,0.0021471963,0.0003718745,0.0012751732,0.0011520922,0.00043162747],"domain_scores_gemma":[0.97542065,0.014613765,0.0005401723,0.0068879877,0.0022810185,0.00025646706],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0049394732,0.0014583353,0.0016549707,0.0022524805,0.0010900098,0.0031223644,0.002705048,0.002471493,0.0054550045],"category_scores_gemma":[0.024428127,0.0010952245,0.0012599804,0.0030137328,0.0018426742,0.008019152,0.0034533578,0.0028770578,0.00316673],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007134895,0.00041697602,0.019602077,0.0008813536,0.00033048936,0.0011072081,0.0016196466,0.34478065,0.027733354,0.10627698,0.052864328,0.44367343],"study_design_scores_gemma":[0.00006435473,0.00007357648,0.0030628995,0.000059492013,0.00009137148,0.00031919166,0.00020111132,0.77856773,0.014090356,0.19467676,0.008700538,0.00009260146],"about_ca_topic_score_codex":0.0065191532,"about_ca_topic_score_gemma":0.011875566,"teacher_disagreement_score":0.0065191532,"about_ca_system_score_codex":0.001534088,"about_ca_system_score_gemma":0.0024057971,"threshold_uncertainty_score":0.02612269},"labels":[],"label_agreement":null},{"id":"W2104444209","doi":"10.5539/elt.v2n3p53","title":"Text Coherence in Translation","year":2009,"lang":"en","type":"article","venue":"English Language Teaching","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Foregrounding; Coherence (philosophical gambling strategy); Source text; Target text; Translation (biology); Natural language processing; Computer science; Linguistics; Artificial intelligence; Complement (music); Process (computing); Space (punctuation); Psychology; Mathematics; Philosophy","score_opus":0.009913382092200971,"score_gpt":0.2732470129008626,"score_spread":0.2633336308086616,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2104444209","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06610936,0.014591653,0.63926405,0.029839652,0.0026665723,0.00045014953,0.000603905,0.0010571289,0.24541761],"genre_scores_gemma":[0.8488727,0.0033476863,0.119799346,0.0016266488,0.0016766636,0.00044777576,0.00058542215,0.0005443875,0.02309931],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9915405,0.0051250006,0.0005690826,0.001272121,0.0011270454,0.00036620002],"domain_scores_gemma":[0.9825117,0.011881867,0.0013517776,0.0019162184,0.001928154,0.00041039163],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005315159,0.00065361895,0.0006375654,0.003296448,0.0034130323,0.0062583084,0.0010415342,0.0021158012,0.011534998],"category_scores_gemma":[0.029088426,0.00058707193,0.00080324465,0.0041506267,0.010020923,0.016212938,0.006639968,0.0025570686,0.0018939376],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014008804,0.000023782522,0.00067420385,0.00037955484,0.000030523268,0.00035664582,0.011412074,0.0020260818,0.001259271,0.9203783,0.0058225333,0.057496794],"study_design_scores_gemma":[0.000071898954,0.000102280654,0.0010845703,0.00022680337,0.000052107196,0.0004990105,0.004931962,0.008925461,0.0028045864,0.88950837,0.09174627,0.00004672481],"about_ca_topic_score_codex":0.0015113789,"about_ca_topic_score_gemma":0.00089573127,"teacher_disagreement_score":0.011534998,"about_ca_system_score_codex":0.0025733956,"about_ca_system_score_gemma":0.0017926422,"threshold_uncertainty_score":0.038588405},"labels":[],"label_agreement":null},{"id":"W2104548632","doi":"10.1109/taes.2010.5545188","title":"Fast Learning of Grammar Production Probabilities in Radar Electronic Support","year":2010,"lang":"en","type":"article","venue":"IEEE Transactions on Aerospace and Electronic Systems","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":26,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Defence Research and Development Canada; École de Technologie Supérieure; Université de Montréal","funders":"","keywords":"Computer science; Viterbi algorithm; Radar; Computational complexity theory; Algorithm; Context (archaeology); Artificial intelligence; Expectation–maximization algorithm; Machine learning; Maximum likelihood; Hidden Markov model; Mathematics","score_opus":0.004785640587931589,"score_gpt":0.22041558737155195,"score_spread":0.21562994678362035,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2104548632","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.115142174,0.0002211377,0.8807197,0.00014215006,0.00001847857,0.00008935435,0.00013123697,0.0024954053,0.0010404005],"genre_scores_gemma":[0.5579089,0.0001884374,0.43948013,0.00008978214,0.000014103344,0.00014566173,0.0005090889,0.000187908,0.0014759932],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999183,0.00030214517,0.000031327607,0.00021814856,0.00019364935,0.00007168903],"domain_scores_gemma":[0.99774796,0.0017702443,0.00010755875,0.00012577987,0.00020364422,0.000044819077],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015845214,0.0005292524,0.00085238187,0.00086828956,0.00040469534,0.0009062116,0.001243754,0.0010975668,0.0013717127],"category_scores_gemma":[0.0049008853,0.0006605476,0.00056339416,0.00062677334,0.0006830196,0.001384912,0.0012027416,0.0011837368,0.0006012847],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017658937,0.00010391703,0.0022754828,0.00013312935,0.000041987794,0.00020984969,0.00031319505,0.650774,0.02086142,0.012747174,0.0012790483,0.31108415],"study_design_scores_gemma":[0.000011222045,0.000029005521,0.00029603695,0.0000057029965,0.000004167665,0.00003016491,0.00002337417,0.98802966,0.0057547987,0.0053945966,0.00041091154,0.000010360223],"about_ca_topic_score_codex":0.0047545386,"about_ca_topic_score_gemma":0.004890276,"teacher_disagreement_score":0.0047545386,"about_ca_system_score_codex":0.00071505253,"about_ca_system_score_gemma":0.0014626852,"threshold_uncertainty_score":0.009453714},"labels":[],"label_agreement":null},{"id":"W2104554525","doi":"10.5539/cis.v8n1p119","title":"Augmenting Performance of SMT Models by Deploying Fine Tokenization of the Text and Part-of-Speech Tag","year":2015,"lang":"en","type":"article","venue":"Computer and Information Science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"National Natural Science Foundation of China","keywords":"Computer science; Machine translation; Natural language processing; Lexical analysis; Artificial intelligence; Phrase; Word (group theory); Language model; Set (abstract data type); Translation (biology); Speech recognition; Linguistics; Programming language","score_opus":0.01804727061539576,"score_gpt":0.23965310868320666,"score_spread":0.2216058380678109,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2104554525","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.22592105,0.004932093,0.71406347,0.0015726706,0.0010444503,0.00037950752,0.0018681699,0.036640443,0.013578101],"genre_scores_gemma":[0.69820493,0.0015187692,0.28137058,0.0006751649,0.00029375174,0.00020758904,0.0051884,0.0017504212,0.010790468],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99853265,0.000547965,0.00009819328,0.0004078889,0.0002955187,0.00011776431],"domain_scores_gemma":[0.9953721,0.002650311,0.00015839876,0.0009022566,0.00081579987,0.000101163605],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024374132,0.0019441738,0.0014029532,0.0011162333,0.00060282735,0.0016980408,0.0013934299,0.0010682575,0.0043818434],"category_scores_gemma":[0.008163886,0.000610441,0.0010321371,0.0012489941,0.00039788315,0.0031572687,0.0012224999,0.0018224439,0.008651255],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007090512,0.00035044618,0.0063529066,0.0006099947,0.00044099882,0.00027614075,0.0002663641,0.16121979,0.046373438,0.0013810034,0.010434128,0.7715857],"study_design_scores_gemma":[0.000032187818,0.00034121436,0.0019158837,0.000038693885,0.0001531662,0.00023369229,0.00011684552,0.9591455,0.026640909,0.0024956826,0.008824253,0.0000619572],"about_ca_topic_score_codex":0.008812862,"about_ca_topic_score_gemma":0.014562386,"teacher_disagreement_score":0.008812862,"about_ca_system_score_codex":0.0006761125,"about_ca_system_score_gemma":0.0015620837,"threshold_uncertainty_score":0.01752317},"labels":[],"label_agreement":null},{"id":"W2105268931","doi":"","title":"Information Processing and the Recovery of Argument Structure Asymmetries","year":2006,"lang":"en","type":"article","venue":"New Trends in Software Methodologies, Tools and Techniques","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Argument (complex analysis); Computer science; Parsing; Relation (database); Natural language processing; Information structure; Artificial intelligence; Question answering; Linguistics; Data mining; Philosophy","score_opus":0.036107993123982184,"score_gpt":0.3134885648478385,"score_spread":0.27738057172385633,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2105268931","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.044349696,0.0016378496,0.9394493,0.002848032,0.00017080085,0.000073826865,0.00017982005,0.0011339979,0.010156613],"genre_scores_gemma":[0.63154846,0.0014584855,0.3591997,0.00068311376,0.00035735557,0.00015912215,0.0005974441,0.000855467,0.005140892],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99513054,0.001920354,0.00042001824,0.0008890405,0.0011860906,0.00045402092],"domain_scores_gemma":[0.9752303,0.018197756,0.0015100172,0.0033457393,0.0015075036,0.00020862267],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0051063085,0.0007748488,0.0012660982,0.0024781607,0.0018433796,0.00477057,0.0020489146,0.002795683,0.0046697105],"category_scores_gemma":[0.037492827,0.0009774142,0.001674911,0.002353374,0.005031719,0.015170015,0.0042125345,0.0038667857,0.0016009681],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003795289,0.00007337071,0.0012489381,0.000401155,0.00006233683,0.0008719518,0.0018915393,0.015362776,0.023582812,0.72136503,0.003099325,0.23166125],"study_design_scores_gemma":[0.000042256135,0.000034995104,0.0005654509,0.000068630936,0.000063834384,0.00042768594,0.00018706796,0.057317402,0.016157633,0.91573983,0.009328437,0.0000667019],"about_ca_topic_score_codex":0.0009702131,"about_ca_topic_score_gemma":0.00056928734,"teacher_disagreement_score":0.0051063085,"about_ca_system_score_codex":0.0013371283,"about_ca_system_score_gemma":0.0015040017,"threshold_uncertainty_score":0.027005076},"labels":[],"label_agreement":null},{"id":"W2105410942","doi":"10.3115/1620754.1620815","title":"Active learning for statistical phrase-based machine translation","year":2009,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":98,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Machine translation; Artificial intelligence; Natural language processing; Phrase; Evaluation of machine translation; Example-based machine translation; Machine translation software usability; Sentence; Translation (biology); Selection (genetic algorithm); Active learning (machine learning); Machine learning","score_opus":0.016677019508315488,"score_gpt":0.3063805958636106,"score_spread":0.2897035763552951,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2105410942","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0017135267,0.00034637834,0.99662834,0.00013299196,0.000039004408,0.000026514155,0.000034943023,0.0005869557,0.000491421],"genre_scores_gemma":[0.29038736,0.0011848056,0.7020301,0.0003238556,0.00045276538,0.0007194489,0.000676265,0.00035540032,0.003869876],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9971397,0.001780347,0.00015687235,0.00033725027,0.0005035309,0.00008230156],"domain_scores_gemma":[0.9878522,0.009704027,0.00047013315,0.0010027258,0.0008536833,0.000117286494],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004810693,0.00095301727,0.0013611853,0.0012357273,0.0007126657,0.0017187275,0.0024646244,0.0018225367,0.0033663984],"category_scores_gemma":[0.014091967,0.0008066691,0.0009195384,0.0018563954,0.001551773,0.0030843827,0.0017981722,0.0028884576,0.0014415571],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035564066,0.00021513441,0.0005553273,0.00036291714,0.0001986532,0.00016307748,0.00018803892,0.424286,0.008145503,0.0655023,0.0051600877,0.49486727],"study_design_scores_gemma":[0.000018386309,0.00003621641,0.00005410957,0.000008255513,0.000008107234,0.000022073124,0.00000670034,0.9678087,0.0017120127,0.029097667,0.0012181129,0.000009670515],"about_ca_topic_score_codex":0.0012228732,"about_ca_topic_score_gemma":0.0013175945,"teacher_disagreement_score":0.004810693,"about_ca_system_score_codex":0.00084381865,"about_ca_system_score_gemma":0.0008350175,"threshold_uncertainty_score":0.025441647},"labels":[],"label_agreement":null},{"id":"W2105580583","doi":"10.3115/1220575.1220625","title":"Bootstrapping without the boot","year":2005,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Strapping; Bootstrapping (finance); Computer science; Artificial intelligence; Machine learning; Process (computing); Outcome (game theory); Word (group theory); Oversampling; Natural language processing; Mathematics; Econometrics; Engineering","score_opus":0.013940832587142883,"score_gpt":0.27490023410206094,"score_spread":0.26095940151491803,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2105580583","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019221013,0.000305264,0.9685348,0.00057514803,0.00028759585,0.00028314276,0.00035413404,0.004808281,0.0056306357],"genre_scores_gemma":[0.43232906,0.00023308728,0.551594,0.001127475,0.00044347797,0.0010766947,0.0018985242,0.001493317,0.009804277],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9953595,0.0023521874,0.00020291428,0.00091699633,0.0008646764,0.00030382245],"domain_scores_gemma":[0.9770957,0.008483209,0.00066950195,0.011738103,0.0015852787,0.000428239],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0052735563,0.001249297,0.0020067845,0.0011847939,0.0017183855,0.0017251273,0.0027485278,0.002241877,0.009975476],"category_scores_gemma":[0.044283796,0.0010823796,0.001241225,0.0013820108,0.0022308636,0.0038738356,0.0041719023,0.0034248321,0.007295868],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014948754,0.00043234538,0.0076005505,0.0004821662,0.0003033488,0.00062549673,0.0007006912,0.12012476,0.010375701,0.15395494,0.051510815,0.6523943],"study_design_scores_gemma":[0.00024299856,0.0002727151,0.0016194698,0.00011828959,0.00008855671,0.0004461181,0.00017037129,0.7210097,0.008402852,0.24158226,0.02596665,0.00008011778],"about_ca_topic_score_codex":0.0016852001,"about_ca_topic_score_gemma":0.0027181776,"teacher_disagreement_score":0.009975476,"about_ca_system_score_codex":0.0007146431,"about_ca_system_score_gemma":0.0020331184,"threshold_uncertainty_score":0.03337127},"labels":[],"label_agreement":null},{"id":"W2105938030","doi":"","title":"Unified Classification of Nominal Classifiers and Formalization of Classifier-noun Phrases*","year":2011,"lang":"en","type":"article","venue":"Studies in literature and language","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Natural language processing; Computer science; Artificial intelligence; Noun phrase; Classifier (UML); Noun; Machine translation; Phrase; Linguistics","score_opus":0.04115653392967056,"score_gpt":0.31068813791301253,"score_spread":0.26953160398334197,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2105938030","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009999159,0.000991425,0.9706989,0.0014685587,0.00021881168,0.00016665249,0.000383521,0.00032669574,0.015746366],"genre_scores_gemma":[0.23637918,0.0010022757,0.75493264,0.00043948524,0.0003577644,0.00058763905,0.0012269849,0.00014294712,0.004931001],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99458236,0.0019977307,0.00097059854,0.0009848477,0.0012017832,0.0002627579],"domain_scores_gemma":[0.99483395,0.0019669316,0.0007469501,0.00087867567,0.0014477583,0.00012572123],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006063657,0.0007033632,0.0006530041,0.004547564,0.001781489,0.0041417317,0.0015598788,0.0010483859,0.0044595627],"category_scores_gemma":[0.010721059,0.0004702631,0.0016452373,0.0045552724,0.006294717,0.012011853,0.0019052268,0.0020477928,0.0012577681],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000012335214,0.000010206025,0.00033933908,0.000101530924,0.00001126903,0.00007234329,0.0007195094,0.0010354939,0.00078602013,0.97773767,0.0014387766,0.017735396],"study_design_scores_gemma":[0.000013546623,0.000038962015,0.00075352204,0.00019672429,0.000041429466,0.00029298107,0.0007473141,0.0351269,0.0020359722,0.9158264,0.044881877,0.000044314344],"about_ca_topic_score_codex":0.003550658,"about_ca_topic_score_gemma":0.0027955435,"teacher_disagreement_score":0.006063657,"about_ca_system_score_codex":0.0026403693,"about_ca_system_score_gemma":0.0028832038,"threshold_uncertainty_score":0.032068074},"labels":[],"label_agreement":null},{"id":"W2106394101","doi":"10.7202/029796ar","title":"The Unit of Translation: Statistics Speak","year":2009,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":32,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"Macquarie University","keywords":"Sentence; Context (archaeology); Linguistics; Subject (documents); Identification (biology); Computer science; Natural language processing; Psychology; History; Philosophy; Library science","score_opus":0.04371525780886655,"score_gpt":0.3000459238541289,"score_spread":0.2563306660452624,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2106394101","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.49085197,0.011535191,0.3557313,0.013227048,0.0037180854,0.0010830187,0.021025078,0.0030055735,0.09982269],"genre_scores_gemma":[0.9587913,0.0008827602,0.02853737,0.0009846064,0.0011397338,0.0018112069,0.004772127,0.0007651483,0.0023155971],"study_design_codex":"observational","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9185343,0.05185796,0.0056857667,0.010704008,0.011464969,0.0017529962],"domain_scores_gemma":[0.4476983,0.48830622,0.025540447,0.028266964,0.008577362,0.0016106805],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04357779,0.0007637204,0.0020013484,0.011924726,0.0018891374,0.0076466333,0.0015370073,0.0015496742,0.0123581095],"category_scores_gemma":[0.28705797,0.00047322505,0.0013389495,0.021266008,0.009379711,0.011034589,0.004981655,0.004170817,0.0031393839],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017961591,0.00030538999,0.3456154,0.0021544022,0.0021521135,0.0012413813,0.025864134,0.0035708693,0.0013261038,0.27288598,0.07366768,0.26942042],"study_design_scores_gemma":[0.00022004257,0.0025411292,0.2587172,0.001494538,0.0011111851,0.0044111544,0.033972755,0.043850385,0.0043120617,0.4782696,0.17059587,0.0005040573],"about_ca_topic_score_codex":0.001330135,"about_ca_topic_score_gemma":0.0007357768,"teacher_disagreement_score":0.04357779,"about_ca_system_score_codex":0.0015333244,"about_ca_system_score_gemma":0.0019071992,"threshold_uncertainty_score":0.23046416},"labels":[],"label_agreement":null},{"id":"W2106595396","doi":"","title":"Adaptation of Reordering Models for Statistical Machine Translation","year":2013,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Machine translation; Smoothing; Domain adaptation; Adaptation (eye); Phrase; Weighting; Artificial intelligence; Translation (biology); Natural language processing; Statistical model; Domain (mathematical analysis); Language model; Machine learning; Speech recognition; Mathematics","score_opus":0.030927318270226857,"score_gpt":0.2782899549544697,"score_spread":0.24736263668424283,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2106595396","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018891077,0.0007440031,0.9726901,0.00016078692,0.00012411339,0.00009462963,0.00019877419,0.005323668,0.0017728271],"genre_scores_gemma":[0.41284993,0.001125959,0.57418364,0.000338984,0.00018881918,0.00038311532,0.0023209725,0.001579099,0.007029444],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9982711,0.0009856495,0.00010402494,0.0003370802,0.00022457262,0.00007762718],"domain_scores_gemma":[0.99721485,0.001425712,0.00013388348,0.0007022092,0.00046278365,0.000060573195],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002577033,0.0012121791,0.00081935606,0.0010426125,0.00040855943,0.00078904786,0.0010967159,0.0008073524,0.0031936623],"category_scores_gemma":[0.006489201,0.0006181785,0.0010387582,0.001458511,0.00041102344,0.001878683,0.0011402604,0.0025096806,0.004109313],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00040626506,0.00029068062,0.0020265172,0.00021010639,0.00032284064,0.00018357139,0.00028003135,0.32012174,0.041936513,0.007830677,0.007441349,0.6189498],"study_design_scores_gemma":[0.000014475223,0.000054985445,0.00056930317,0.000008982215,0.000027482512,0.00007540279,0.000027381815,0.9816596,0.007919267,0.0050660116,0.0045497627,0.000027398952],"about_ca_topic_score_codex":0.0037446576,"about_ca_topic_score_gemma":0.0062607047,"teacher_disagreement_score":0.0037446576,"about_ca_system_score_codex":0.0006227716,"about_ca_system_score_gemma":0.00070589484,"threshold_uncertainty_score":0.01362884},"labels":[],"label_agreement":null},{"id":"W2106629332","doi":"","title":"Saarland University Spoken Language Systems at the Slot Filling Task of TAC KBP 2010","year":2010,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Pipeline (software); Computer science; Task (project management); Generalization; Word (group theory); Relationship extraction; Natural language processing; Relation (database); Entity linking; Simple (philosophy); Artificial intelligence; Knowledge base; Information extraction; Data mining; Programming language; Engineering; Mathematics","score_opus":0.004276613195414662,"score_gpt":0.22003944380967175,"score_spread":0.21576283061425708,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2106629332","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.33755723,0.0029941096,0.2909903,0.0041371034,0.0017710311,0.0014278587,0.045759723,0.23137891,0.08398377],"genre_scores_gemma":[0.5436018,0.0005036355,0.30792904,0.00067243003,0.00025164383,0.00092514267,0.093323804,0.005268095,0.047524344],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9972268,0.0009007001,0.00022636904,0.00068653945,0.0006577893,0.00030182782],"domain_scores_gemma":[0.9951296,0.0018185518,0.0001523394,0.0009992421,0.001546594,0.00035373218],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003255009,0.0012815057,0.0014563669,0.0015110335,0.0011704062,0.0025258954,0.0015776054,0.001651805,0.020097198],"category_scores_gemma":[0.010110492,0.0007111226,0.0006514004,0.001385792,0.0005314438,0.0045240936,0.0020180466,0.001701239,0.019701863],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015280822,0.0008563234,0.0025302577,0.00095334876,0.00020916035,0.0007619211,0.0024360442,0.013163646,0.05870305,0.0064977626,0.3502738,0.5620866],"study_design_scores_gemma":[0.0014567473,0.0012554104,0.010689637,0.00026687467,0.00031968814,0.0011990027,0.003362532,0.48938516,0.16864559,0.019441709,0.30348596,0.0004916283],"about_ca_topic_score_codex":0.017449008,"about_ca_topic_score_gemma":0.01619838,"teacher_disagreement_score":0.020097198,"about_ca_system_score_codex":0.001333163,"about_ca_system_score_gemma":0.0024858082,"threshold_uncertainty_score":0.067231834},"labels":[],"label_agreement":null},{"id":"W2106854223","doi":"10.1111/j.1467-8640.2012.00436.x","title":"EXPLOITING SYNTACTIC, SEMANTIC, AND LEXICAL REGULARITIES IN LANGUAGE MODELING VIA DIRECTED MARKOV RANDOM FIELDS","year":2012,"lang":"en","type":"article","venue":"Computational Intelligence","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Language model; Perplexity; Computer science; Trigram; Probabilistic latent semantic analysis; Artificial intelligence; Natural language processing; Smoothing; Context (archaeology); Algorithm","score_opus":0.02393159905830056,"score_gpt":0.2965222874839078,"score_spread":0.27259068842560724,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2106854223","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019432753,0.00018169657,0.97899514,0.00019122509,0.00001912819,0.000026337384,0.00015746648,0.00043956842,0.0005567051],"genre_scores_gemma":[0.519005,0.0007374589,0.47522506,0.00024736713,0.00009389087,0.00032603374,0.0008265143,0.00032515702,0.003213425],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99919707,0.00043343363,0.000031957334,0.00016768927,0.000112285095,0.00005753818],"domain_scores_gemma":[0.99526155,0.004051643,0.00023390618,0.00024387104,0.00013697274,0.00007216477],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020178237,0.00071489956,0.00094177393,0.0014017259,0.00059948786,0.0009404114,0.0015703413,0.0012824013,0.0013116272],"category_scores_gemma":[0.0067169163,0.0008339678,0.0015353119,0.0013026772,0.0008948654,0.002228394,0.0009718247,0.0015443844,0.00056673255],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008818377,0.00006787772,0.0015572488,0.00008661887,0.00006837977,0.0002479438,0.00019203917,0.8746275,0.0026579502,0.06837356,0.0011497575,0.050883062],"study_design_scores_gemma":[0.000005034767,0.000007589457,0.000076723496,0.0000044182534,0.000006636038,0.000022135953,0.000004127406,0.9714516,0.00023531275,0.027894465,0.00028357375,0.000008286911],"about_ca_topic_score_codex":0.008739644,"about_ca_topic_score_gemma":0.013248854,"teacher_disagreement_score":0.008739644,"about_ca_system_score_codex":0.0009866303,"about_ca_system_score_gemma":0.0015331234,"threshold_uncertainty_score":0.017377555},"labels":[],"label_agreement":null},{"id":"W2107295905","doi":"10.7202/004014ar","title":"Lexical Cohesion and Translation Equivalence","year":2002,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":29,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Cohesion (chemistry); Linguistics; Equivalence (formal languages); Computer science; Lexical functional grammar; Sentence; Lexical choice; Natural language processing; Lexical item; Co-occurrence; Lexical semantics; Artificial intelligence; Philosophy","score_opus":0.07181585355392639,"score_gpt":0.28813426447344265,"score_spread":0.21631841091951626,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2107295905","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2901742,0.0041177454,0.29542172,0.0060421224,0.0008402827,0.00042474415,0.0006033571,0.0007382587,0.4016375],"genre_scores_gemma":[0.96292746,0.000750127,0.02400233,0.00029678585,0.00042986093,0.00023476392,0.000445149,0.0001454139,0.010768134],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9952873,0.0020961359,0.00049813854,0.0009856266,0.0007701071,0.00036270174],"domain_scores_gemma":[0.9937337,0.0033325928,0.0007770087,0.00093129923,0.0010114345,0.00021396442],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022675388,0.0004137673,0.0005986896,0.0028268704,0.0023513306,0.0044421903,0.00068132463,0.0011966291,0.008826919],"category_scores_gemma":[0.01582359,0.00027451225,0.0005511653,0.0018346153,0.0060080225,0.008199248,0.004452079,0.0012256037,0.0012228744],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008319204,0.000050346236,0.0016753111,0.00016679165,0.0000306221,0.00046404047,0.008119142,0.00066161685,0.002413967,0.9243416,0.0021167933,0.059876658],"study_design_scores_gemma":[0.000035949,0.00008390091,0.0032006868,0.00007849296,0.000028297889,0.00039762005,0.0024191926,0.0018824124,0.0013581556,0.9647235,0.02576894,0.00002276646],"about_ca_topic_score_codex":0.00093892217,"about_ca_topic_score_gemma":0.00050008437,"teacher_disagreement_score":0.008826919,"about_ca_system_score_codex":0.0010903063,"about_ca_system_score_gemma":0.00067112787,"threshold_uncertainty_score":0.029528975},"labels":[],"label_agreement":null},{"id":"W2108059661","doi":"10.3115/1614108.1614133","title":"Exploiting rich syntactic information for relation extraction from biomedical articles","year":2007,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":31,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Relationship extraction; Tree kernel; Parsing; Artificial intelligence; Parse tree; Natural language processing; Information extraction; Support vector machine; Relation (database); Kernel (algebra); Tree (set theory); Word (group theory); Semantic relation; Kernel method; Data mining; Polynomial kernel; Mathematics","score_opus":0.01710854009509264,"score_gpt":0.296085748317495,"score_spread":0.27897720822240235,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2108059661","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.068967156,0.0024177644,0.9102716,0.0013951632,0.00016580854,0.00028776916,0.0036362277,0.0067598484,0.0060986304],"genre_scores_gemma":[0.3606084,0.001812559,0.62254786,0.00030950652,0.0003003377,0.00025548346,0.011124,0.00068602397,0.0023556799],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9975274,0.00070395326,0.0004205177,0.000393871,0.0008299938,0.00012428855],"domain_scores_gemma":[0.986914,0.008011117,0.0016833106,0.0012380125,0.0019786356,0.00017488885],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027239616,0.00091269024,0.001240428,0.010007833,0.0009945835,0.0027427217,0.000997628,0.0010205193,0.0032885235],"category_scores_gemma":[0.014193515,0.0005122401,0.0013946533,0.0057386314,0.00074626895,0.006723747,0.0018734639,0.0012071743,0.0038978744],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003992664,0.0003249801,0.008252674,0.0024432621,0.0002700052,0.0013461806,0.0011836744,0.0059877597,0.093379736,0.035497192,0.009832327,0.84108305],"study_design_scores_gemma":[0.0001385314,0.00041556545,0.027133705,0.0007916606,0.0010415232,0.0047820094,0.0014648371,0.42200208,0.2466268,0.18416122,0.11097933,0.00046279767],"about_ca_topic_score_codex":0.00094641215,"about_ca_topic_score_gemma":0.0016274041,"teacher_disagreement_score":0.010007833,"about_ca_system_score_codex":0.00072808994,"about_ca_system_score_gemma":0.0020990106,"threshold_uncertainty_score":0.014405847},"labels":[],"label_agreement":null},{"id":"W2108198408","doi":"10.1109/ipcc.1998.722090","title":"Translation, globalization and localization","year":2002,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Douglas College","funders":"","keywords":"Standardization; Documentation; Terminology; Globalization; Computer science; Interpreter; Competence (human resources); Language industry; Machine translation; Data science; World Wide Web; Natural language processing; Artificial intelligence; Linguistics; Natural language; Political science; Programming language; Management","score_opus":0.015666984989742786,"score_gpt":0.24589250359607548,"score_spread":0.2302255186063327,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2108198408","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017778225,0.07658131,0.04572555,0.08803512,0.004355268,0.000089620415,0.00014519758,0.00047193072,0.76681787],"genre_scores_gemma":[0.63146895,0.07967042,0.04066573,0.021897608,0.0065008695,0.0002948074,0.00034757247,0.0009233423,0.2182307],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9948394,0.002957014,0.00029972114,0.000770175,0.00080709223,0.00032654754],"domain_scores_gemma":[0.99505264,0.0026222928,0.0006318014,0.0008596966,0.0006022227,0.0002314098],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036605366,0.0007690701,0.0008005495,0.0020177471,0.00396723,0.010548463,0.0008082318,0.0031986192,0.013898843],"category_scores_gemma":[0.009025754,0.00031931163,0.0004544684,0.0038420162,0.02033747,0.0134707345,0.0059876903,0.0030034718,0.0034195553],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00002410211,0.000017180082,0.0005145141,0.0002849661,0.000011008735,0.000474147,0.011498244,0.00027719265,0.00027027534,0.8756076,0.02605769,0.08496305],"study_design_scores_gemma":[0.0000107604465,0.000034229157,0.0013990063,0.0005635042,0.00001048385,0.0011934504,0.006670277,0.00030712082,0.00030254913,0.4388766,0.55060744,0.000024597597],"about_ca_topic_score_codex":0.0020734447,"about_ca_topic_score_gemma":0.0017067135,"teacher_disagreement_score":0.013898843,"about_ca_system_score_codex":0.002843936,"about_ca_system_score_gemma":0.002301509,"threshold_uncertainty_score":0.046496272},"labels":[],"label_agreement":null},{"id":"W2108475409","doi":"10.1007/978-3-540-70939-8_16","title":"A Generalized Approach to Word Segmentation Using Maximum Length Descending Frequency and Entropy Rate","year":2007,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Principle of maximum entropy; Segmentation; Word (group theory); Entropy (arrow of time); Matching (statistics); Text segmentation; Algorithm; Pattern recognition (psychology); Artificial intelligence; Speech recognition; Word lists by frequency; Word length; Natural language processing; Mathematics; Statistics; Sentence","score_opus":0.034352980930623424,"score_gpt":0.2905912693922288,"score_spread":0.25623828846160535,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2108475409","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0030780612,0.0003001033,0.9953252,0.00004826439,0.00004908259,0.00003672101,0.00007231323,0.00031115106,0.0007791417],"genre_scores_gemma":[0.07630874,0.0006341677,0.91629463,0.000103816295,0.00039659112,0.00019912921,0.0004758052,0.0005180281,0.0050690663],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9980508,0.0005514417,0.00017533013,0.00046848695,0.00062677334,0.00012714257],"domain_scores_gemma":[0.99672747,0.0017739013,0.00015728378,0.00052710273,0.0007193224,0.00009489311],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022319856,0.0010824637,0.0023218421,0.0040484774,0.001005929,0.0024713122,0.0030439978,0.001381758,0.0046005803],"category_scores_gemma":[0.006826205,0.00089261896,0.0018491459,0.004364425,0.0014483326,0.004752028,0.0019864787,0.0018464915,0.0020089603],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00027011673,0.00018102606,0.0014033567,0.00047475417,0.00019722323,0.00035242966,0.00065712375,0.13119107,0.023280375,0.27094823,0.0045684213,0.56647587],"study_design_scores_gemma":[0.000011726609,0.000051428724,0.0006119095,0.000028891356,0.00004296334,0.00011787217,0.000058987167,0.832821,0.0032469425,0.15862443,0.00434022,0.000043711807],"about_ca_topic_score_codex":0.004596812,"about_ca_topic_score_gemma":0.0050477223,"teacher_disagreement_score":0.0046005803,"about_ca_system_score_codex":0.0012090683,"about_ca_system_score_gemma":0.0014010553,"threshold_uncertainty_score":0.015390515},"labels":[],"label_agreement":null},{"id":"W2108638558","doi":"10.1109/grc.2006.1635878","title":"An effective extension to okapi for biomedical text mining","year":2006,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Extension (predicate logic); Computer science; Natural language processing; Artificial intelligence; Information retrieval; Programming language","score_opus":0.007491673153196014,"score_gpt":0.29527870678172924,"score_spread":0.28778703362853325,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2108638558","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020989513,0.0041543995,0.81003237,0.0021030153,0.0025783593,0.0010229683,0.020409286,0.12870099,0.010009033],"genre_scores_gemma":[0.07161881,0.0013131433,0.87410253,0.0013469071,0.00086672493,0.00061895506,0.030188136,0.0022934265,0.01765141],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9983259,0.00026977502,0.00026386668,0.00030010456,0.0007353316,0.000105017294],"domain_scores_gemma":[0.9966833,0.0011632986,0.0001088245,0.00087526755,0.0010023065,0.00016697892],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001669634,0.0010129695,0.0013315774,0.0025445567,0.0014114926,0.0016353646,0.0018809529,0.0010582922,0.010139252],"category_scores_gemma":[0.00549017,0.00068785413,0.0012048009,0.0024205707,0.00034465583,0.0024273482,0.002779176,0.0016849121,0.010570045],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000376522,0.0003056933,0.0015112272,0.00070986344,0.00020143528,0.0004484883,0.0001778362,0.001587231,0.01597864,0.0027462004,0.15353769,0.822419],"study_design_scores_gemma":[0.00054032047,0.0004506778,0.012831531,0.00049192045,0.0012237917,0.003485603,0.00052577734,0.40564469,0.07515986,0.08078912,0.4185466,0.00031000312],"about_ca_topic_score_codex":0.0042718677,"about_ca_topic_score_gemma":0.010851017,"teacher_disagreement_score":0.010139252,"about_ca_system_score_codex":0.00033214892,"about_ca_system_score_gemma":0.00180592,"threshold_uncertainty_score":0.033919215},"labels":[],"label_agreement":null},{"id":"W2108664951","doi":"","title":"LetSum, an automatic Legal Text Summarizing system","year":2004,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":54,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Automatic summarization; Theme (computing); Computer science; Context (archaeology); Economic Justice; Multi-document summarization; Information retrieval; Natural language processing; Thematic map; Data science; Artificial intelligence; World Wide Web; Political science; Law; History; Cartography","score_opus":0.009957741616187935,"score_gpt":0.25501027668730847,"score_spread":0.24505253507112054,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2108664951","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.039546903,0.0028070477,0.72407466,0.0012015463,0.0006140223,0.0011331564,0.019110097,0.19809775,0.013414765],"genre_scores_gemma":[0.08143295,0.0009955146,0.856149,0.00030908323,0.00039444922,0.0006630371,0.044830598,0.002136911,0.013088388],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990145,0.0002659332,0.00014996208,0.00021355742,0.00032186357,0.00003419836],"domain_scores_gemma":[0.99784327,0.00085618463,0.00024290671,0.0002446303,0.00073892414,0.00007412752],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015284808,0.00091746723,0.00074033765,0.004001747,0.0010077797,0.0015726029,0.0011751613,0.00074379635,0.0091219125],"category_scores_gemma":[0.0058537503,0.0003652119,0.0005030715,0.0020284844,0.0002598203,0.0025248022,0.0010213602,0.0007612348,0.004662068],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00047779625,0.00017512932,0.0013417384,0.0015505932,0.00015480701,0.00037211645,0.0010037523,0.0041856193,0.038925458,0.005012723,0.13371497,0.8130853],"study_design_scores_gemma":[0.00070439803,0.00086686,0.0108923875,0.0003754934,0.00072149717,0.0014203864,0.0018461284,0.2363169,0.12801994,0.020198768,0.5983617,0.0002756197],"about_ca_topic_score_codex":0.002166767,"about_ca_topic_score_gemma":0.0038527036,"teacher_disagreement_score":0.0091219125,"about_ca_system_score_codex":0.0006088173,"about_ca_system_score_gemma":0.001012275,"threshold_uncertainty_score":0.03051585},"labels":[],"label_agreement":null},{"id":"W2108711363","doi":"","title":"A Strategy of Mapping Polish WordNet onto Princeton WordNet","year":2012,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":40,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"WordNet; Premise; Computer science; Set (abstract data type); Natural language processing; Focus (optics); Artificial intelligence; Lexical database; Range (aeronautics); Information retrieval; Linguistics; Programming language; Philosophy","score_opus":0.061792355341779494,"score_gpt":0.3410952297029126,"score_spread":0.2793028743611331,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2108711363","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020821882,0.00008530088,0.95314974,0.0007366914,0.00017438557,0.00054122874,0.00086886826,0.002520006,0.02110184],"genre_scores_gemma":[0.15012999,0.00028344494,0.8307717,0.00038639936,0.000046324847,0.0015656503,0.0020446081,0.0009836874,0.013788344],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99897134,0.00031907775,0.00010487304,0.00031504204,0.00021880194,0.00007083003],"domain_scores_gemma":[0.9986494,0.0002553635,0.00006167174,0.0006309948,0.0003364067,0.00006613094],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015546201,0.00074456417,0.00041947895,0.0038048774,0.001896846,0.0021924034,0.00095525663,0.0005701733,0.007832087],"category_scores_gemma":[0.0066505377,0.000815952,0.0007725426,0.0027350853,0.0012703557,0.0054285754,0.0043884846,0.0016520127,0.0034176542],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001356405,0.0001806922,0.0028933685,0.00026231856,0.000091639544,0.0004149135,0.002980398,0.004939711,0.014568421,0.57188624,0.012959554,0.3886872],"study_design_scores_gemma":[0.00006288382,0.0002736166,0.003636281,0.00017999866,0.000112112044,0.00086935185,0.0025020281,0.066011176,0.047235698,0.57558644,0.3033997,0.00013072084],"about_ca_topic_score_codex":0.004162129,"about_ca_topic_score_gemma":0.0061406265,"teacher_disagreement_score":0.007832087,"about_ca_system_score_codex":0.0007372301,"about_ca_system_score_gemma":0.0017060741,"threshold_uncertainty_score":0.02620089},"labels":[],"label_agreement":null},{"id":"W2109071706","doi":"","title":"Fully Abstractive Approach to Guided Summarization","year":2012,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":85,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Automatic summarization; Computer science; Abstraction; Selection (genetic algorithm); Natural language generation; Context (archaeology); Natural language processing; Artificial intelligence; Natural language; Information retrieval","score_opus":0.025620721938492772,"score_gpt":0.2896314398022767,"score_spread":0.2640107178637839,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2109071706","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0019028578,0.0005067708,0.99312603,0.00020867147,0.000068497095,0.00013083807,0.00022789794,0.0015886021,0.0022398233],"genre_scores_gemma":[0.08613984,0.0009155177,0.9026897,0.00044390597,0.00022774722,0.00049351,0.0017657342,0.0005455733,0.00677838],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99690765,0.0013506764,0.00025634328,0.0004432405,0.0008864417,0.00015562233],"domain_scores_gemma":[0.99588984,0.0017979749,0.0003352823,0.0010591928,0.0008234085,0.0000942634],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021778678,0.0012111362,0.0009356786,0.0020615023,0.00086998224,0.0022223303,0.0019868587,0.0009814323,0.005329366],"category_scores_gemma":[0.005513597,0.00048028142,0.0011408704,0.0014806399,0.0012298996,0.0029038608,0.0022183731,0.0020244266,0.0019833096],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035424577,0.00018032761,0.00043452717,0.0018516227,0.000280855,0.00043756072,0.0020811346,0.04259052,0.053731374,0.17475024,0.022964666,0.700343],"study_design_scores_gemma":[0.00011315057,0.00047822457,0.0008173175,0.00030225242,0.00034417622,0.0004639461,0.00040309833,0.2908583,0.053071946,0.43768167,0.21530232,0.00016358109],"about_ca_topic_score_codex":0.0010451204,"about_ca_topic_score_gemma":0.0022287725,"teacher_disagreement_score":0.005329366,"about_ca_system_score_codex":0.00064954127,"about_ca_system_score_gemma":0.0010264672,"threshold_uncertainty_score":0.017828524},"labels":[],"label_agreement":null},{"id":"W2109130931","doi":"10.7202/1024180ar","title":"Multilingual Chat through Machine Translation: A Case of English-Russian","year":2014,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Scripting language; Intelligibility (philosophy); Computer science; Machine translation; Linguistics; Natural language processing; Artificial intelligence; World Wide Web; Programming language","score_opus":0.03402003609065492,"score_gpt":0.29832070992163756,"score_spread":0.26430067383098266,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2109130931","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9725806,0.00030388663,0.014150726,0.0008980563,0.000055237935,0.00007409579,0.00007671954,0.00017404254,0.0116867535],"genre_scores_gemma":[0.9888142,0.00013177078,0.0062046577,0.0001037612,0.000026248581,0.000055830118,0.000054586562,0.000065644825,0.0045433417],"study_design_codex":"qualitative","study_design_gemma":"not_applicable","domain_scores_codex":[0.99251914,0.0055607003,0.00027204162,0.00055286114,0.0006727331,0.00042252714],"domain_scores_gemma":[0.9852165,0.011124168,0.000990605,0.0012845115,0.0008290251,0.0005552404],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0042754305,0.0007441623,0.0005140959,0.0009964957,0.0055177975,0.0029559038,0.0011272293,0.0022582235,0.002103442],"category_scores_gemma":[0.015587906,0.0004390962,0.00048161368,0.00080123416,0.0027911125,0.0028465984,0.0026284007,0.0013756139,0.0010041001],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00078616134,0.00039915906,0.024969757,0.0005506987,0.00008679655,0.097109646,0.80034065,0.0032267245,0.016229672,0.02040768,0.002126958,0.033766042],"study_design_scores_gemma":[0.00015302852,0.0010187245,0.04865794,0.00077408616,0.00033416343,0.09200504,0.6463362,0.0482882,0.030518232,0.018266423,0.113331884,0.0003161345],"about_ca_topic_score_codex":0.004894134,"about_ca_topic_score_gemma":0.0065029045,"teacher_disagreement_score":0.0055177975,"about_ca_system_score_codex":0.0012225361,"about_ca_system_score_gemma":0.00088706386,"threshold_uncertainty_score":0.022610903},"labels":[],"label_agreement":null},{"id":"W2109163800","doi":"10.1145/1187415.1187417","title":"A statistical model for near-synonym choice","year":2007,"lang":"en","type":"article","venue":"ACM Transactions on Speech and Language Processing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":39,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Synonym (taxonomy); Artificial intelligence; Natural language processing; Task (project management); Context (archaeology); Machine translation; Thesaurus; Mutual information; Information retrieval; Machine learning","score_opus":0.01971804255985616,"score_gpt":0.31857975632009145,"score_spread":0.2988617137602353,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2109163800","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016976943,0.00021293604,0.9791524,0.000506069,0.00006433646,0.00009699557,0.00049630535,0.0005595031,0.001934551],"genre_scores_gemma":[0.59344393,0.0005410455,0.39135447,0.00053533196,0.0004672578,0.0010174061,0.0028528816,0.00052279193,0.00926476],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9942364,0.0027181036,0.0003914164,0.0012748787,0.0010980079,0.00028132155],"domain_scores_gemma":[0.9796816,0.015149493,0.0015252143,0.0016194872,0.0016586988,0.0003654337],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0077102073,0.00088695186,0.001535097,0.0046470505,0.001216191,0.0023656366,0.0037324168,0.0020973834,0.006610387],"category_scores_gemma":[0.02908325,0.0009377501,0.0018020209,0.0047803647,0.0021323692,0.0065434733,0.0022221326,0.0030292806,0.0029168688],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005407126,0.00053368835,0.013181713,0.0003272263,0.0006339837,0.0007943607,0.00097515294,0.3685607,0.006033978,0.3656934,0.012716758,0.23000842],"study_design_scores_gemma":[0.000026786825,0.00004102195,0.00084774324,0.000014006671,0.000025491036,0.0001286284,0.000027773705,0.8600659,0.00049306755,0.13690296,0.001387926,0.000038727547],"about_ca_topic_score_codex":0.0051065097,"about_ca_topic_score_gemma":0.008466461,"teacher_disagreement_score":0.0077102073,"about_ca_system_score_codex":0.0014510335,"about_ca_system_score_gemma":0.0019570682,"threshold_uncertainty_score":0.040775955},"labels":[],"label_agreement":null},{"id":"W2110939313","doi":"10.1002/asi.10214","title":"Query expansion and query translation as logical inference","year":2003,"lang":"en","type":"article","venue":"Journal of the American Society for Information Science and Technology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Query expansion; Web query classification; Computer science; Sargable; RDF query language; Query optimization; Inference; Query language; Web search query; Information retrieval; Set (abstract data type); Translation (biology); Boolean conjunctive query; Natural language processing; Artificial intelligence; Search engine; Programming language","score_opus":0.015411966781037274,"score_gpt":0.2992683814853831,"score_spread":0.2838564147043458,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2110939313","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.043653324,0.0022586107,0.89647365,0.0066220844,0.0002203587,0.0002455678,0.00017367722,0.00056390784,0.049788844],"genre_scores_gemma":[0.79833037,0.0009830897,0.19217877,0.0012326004,0.00043168722,0.000214886,0.0002461097,0.00013010135,0.006252415],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.98824745,0.0070500067,0.00070175435,0.00097315334,0.0025988347,0.00042872108],"domain_scores_gemma":[0.97677714,0.01779139,0.0010180142,0.0017942829,0.0023655097,0.00025363176],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010409783,0.00054183294,0.0007472311,0.0022130483,0.0011186678,0.0043073953,0.0012574764,0.0012348137,0.0056914035],"category_scores_gemma":[0.024167903,0.0004651541,0.001232047,0.0021616442,0.0063193,0.011724024,0.0030071451,0.002610894,0.0006564401],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000083471685,0.000053518866,0.00037497314,0.00016103359,0.00003055526,0.0002536547,0.00074619305,0.0055367947,0.001538886,0.95372236,0.0020333328,0.03546521],"study_design_scores_gemma":[0.00003365539,0.00003531962,0.00024319088,0.00005269428,0.000038825026,0.00022582545,0.00023138545,0.062591635,0.002037068,0.92715377,0.0073357075,0.000020906753],"about_ca_topic_score_codex":0.0016093532,"about_ca_topic_score_gemma":0.00093924336,"teacher_disagreement_score":0.010409783,"about_ca_system_score_codex":0.0020018609,"about_ca_system_score_gemma":0.0012385992,"threshold_uncertainty_score":0.055052876},"labels":[],"label_agreement":null},{"id":"W2111727969","doi":"","title":"Disambiguation of Textual Data Typification for the Purpose of Categorial Analysis","year":2010,"lang":"en","type":"article","venue":"The Florida AI Research Society","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Trois-Rivières","funders":"","keywords":"Categorial grammar; Combinatory categorial grammar; Computer science; Typification; Natural language processing; Heuristics; Artificial intelligence; Link grammar; Sentence; Type (biology); Information retrieval; Rule-based machine translation; Generative grammar; Phrase structure rules; Mildly context-sensitive grammar formalism","score_opus":0.11014160332071683,"score_gpt":0.43904191722945457,"score_spread":0.32890031390873775,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2111727969","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02006221,0.00050359947,0.9657923,0.00061105174,0.0004469284,0.00035849208,0.0015990566,0.0065586483,0.0040677637],"genre_scores_gemma":[0.14086406,0.00028690873,0.8522553,0.00023544626,0.00017130372,0.00035963397,0.0022546437,0.0014539612,0.002118876],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9898718,0.0042078407,0.001410378,0.0019152223,0.0022123198,0.00038250035],"domain_scores_gemma":[0.9597494,0.021743888,0.0032411963,0.009776879,0.0048154034,0.0006732819],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009774829,0.0012956058,0.0017110028,0.012688947,0.0029147465,0.0043787286,0.0026967188,0.0015300972,0.007864308],"category_scores_gemma":[0.042934895,0.00095788745,0.0013435775,0.008950183,0.0021356356,0.007604981,0.0044452786,0.0031864054,0.0034211967],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007618974,0.00025235835,0.012718053,0.0013535036,0.00019796821,0.0016157597,0.0063784793,0.0087722,0.03729828,0.14698917,0.029107897,0.7545545],"study_design_scores_gemma":[0.00012307552,0.00012648404,0.008078468,0.0009841099,0.00025122325,0.0034063377,0.006282242,0.2319072,0.10544137,0.38527888,0.25773323,0.0003873303],"about_ca_topic_score_codex":0.0012035719,"about_ca_topic_score_gemma":0.002125326,"teacher_disagreement_score":0.012688947,"about_ca_system_score_codex":0.0015846005,"about_ca_system_score_gemma":0.0026710709,"threshold_uncertainty_score":0.05169487},"labels":[],"label_agreement":null},{"id":"W2111842129","doi":"10.3115/974147.974166","title":"Unit completion for a computer-aided translation typing system","year":2000,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Translation (biology); Context (archaeology); Word (group theory); Typing; Natural language processing; Completion (oil and gas wells); Artificial intelligence; Programming language; Speech recognition; Linguistics; Engineering","score_opus":0.03440169479145183,"score_gpt":0.27990749104873935,"score_spread":0.24550579625728752,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2111842129","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.061565567,0.0003282163,0.8077722,0.0003695213,0.00023378474,0.00050926843,0.0005702897,0.12447803,0.0041731643],"genre_scores_gemma":[0.18878344,0.00013376329,0.7939179,0.00013684375,0.00009318486,0.00026418787,0.0013383852,0.004069626,0.011262727],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99802905,0.0006182997,0.00021038516,0.00043905337,0.00058674923,0.00011648771],"domain_scores_gemma":[0.9942081,0.0026316177,0.0003166029,0.001220589,0.0012570034,0.00036614388],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028336344,0.000797946,0.0011310935,0.0006714106,0.0011693343,0.0016894178,0.0016279195,0.0013286973,0.019113578],"category_scores_gemma":[0.010594708,0.00083379226,0.0006776746,0.00075430615,0.0005343309,0.002157737,0.0010687063,0.0018583764,0.0068813656],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0049974225,0.00070655067,0.003950773,0.0010575154,0.00011844063,0.0015156425,0.0039816336,0.00976426,0.20952232,0.020753542,0.05630927,0.6873227],"study_design_scores_gemma":[0.0007260859,0.002331269,0.004063473,0.00025141603,0.0002060662,0.0023440127,0.0007814356,0.55122757,0.22957335,0.012536162,0.19555469,0.0004043798],"about_ca_topic_score_codex":0.0034374064,"about_ca_topic_score_gemma":0.0038879118,"teacher_disagreement_score":0.019113578,"about_ca_system_score_codex":0.0007743653,"about_ca_system_score_gemma":0.0014976464,"threshold_uncertainty_score":0.06394124},"labels":[],"label_agreement":null},{"id":"W2111856253","doi":"10.1162/coli.2007.33.1.9","title":"Word-Level Confidence Estimation for Machine Translation","year":2007,"lang":"en","type":"article","venue":"Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":106,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"National Research Council Canada; RWTH Aachen University; European Commission","keywords":"Computer science; Word (group theory); Phrase; Machine translation; Translation (biology); Natural language processing; Artificial intelligence; Example-based machine translation; Linguistics","score_opus":0.04620772119004527,"score_gpt":0.3427001006526349,"score_spread":0.2964923794625896,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2111856253","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013138332,0.001886627,0.9822025,0.00019254038,0.000062877545,0.00005381737,0.00014505326,0.0008574763,0.0014608563],"genre_scores_gemma":[0.45470232,0.0008907576,0.5411777,0.00018096821,0.0003304452,0.00030712198,0.0010834034,0.0005740719,0.0007533101],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.977032,0.010736436,0.0017733746,0.002266762,0.0076903426,0.00050105073],"domain_scores_gemma":[0.8278335,0.14869104,0.0071852445,0.006669271,0.00886376,0.0007571947],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.017739415,0.0015957851,0.001919982,0.0066141584,0.0011506682,0.0038674525,0.0023892086,0.0025456874,0.0029239294],"category_scores_gemma":[0.18177691,0.00081631675,0.0014772585,0.004514826,0.0024688896,0.0070575476,0.0028166426,0.0030842975,0.0011776631],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00093585165,0.00014725141,0.015393497,0.0010058116,0.00070966245,0.00021764298,0.0005819143,0.32882568,0.006068152,0.08984554,0.004041324,0.5522277],"study_design_scores_gemma":[0.000062244544,0.00018459639,0.0034899183,0.00014079454,0.00010628631,0.00036924964,0.00008343089,0.8641735,0.008650952,0.11940042,0.0031953808,0.00014322806],"about_ca_topic_score_codex":0.0014847012,"about_ca_topic_score_gemma":0.00070716674,"teacher_disagreement_score":0.017739415,"about_ca_system_score_codex":0.0015912016,"about_ca_system_score_gemma":0.0012053296,"threshold_uncertainty_score":0.0938161},"labels":[],"label_agreement":null},{"id":"W2112317934","doi":"10.5964/bioling.8881","title":"Solving the UG Problem","year":2012,"lang":"en","type":"article","venue":"Biolinguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Computer science; Word order; Grammar; Interrogative; Universal grammar; Linguistics; Inversion (geology); Interpretation (philosophy); Natural language processing; Appeal; Artificial intelligence; Cognitive science; Psychology; Generative grammar; Programming language; Philosophy","score_opus":0.017996745072897377,"score_gpt":0.27640298689717524,"score_spread":0.25840624182427785,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2112317934","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.044902217,0.003326393,0.7971381,0.060295455,0.0009576504,0.00015634092,0.0008052751,0.0021940358,0.09022456],"genre_scores_gemma":[0.49297565,0.0031725133,0.45757794,0.009506581,0.0022231871,0.0004472423,0.0017184424,0.0018247145,0.03055365],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9956573,0.0020709333,0.00017514835,0.001149924,0.0006119654,0.0003347062],"domain_scores_gemma":[0.98104525,0.0129691325,0.0005145394,0.004125837,0.00095743896,0.0003877275],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0055942894,0.0009206556,0.001542501,0.0015279002,0.0030769666,0.00391973,0.0022955844,0.003874329,0.014867064],"category_scores_gemma":[0.03439253,0.0009108477,0.0018085119,0.0012799163,0.008931659,0.014267576,0.006646566,0.008016084,0.0028922656],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003725322,0.000031176613,0.0006999793,0.0001301862,0.000034388075,0.00021325881,0.00046183856,0.0035694807,0.00027184054,0.92331797,0.016128954,0.055103656],"study_design_scores_gemma":[0.000011739579,0.0000046172686,0.00008859087,0.00002004728,0.0000060349084,0.000092148664,0.000107938286,0.0065859826,0.0001367211,0.983065,0.009872744,0.000008417701],"about_ca_topic_score_codex":0.0029136103,"about_ca_topic_score_gemma":0.00255972,"teacher_disagreement_score":0.014867064,"about_ca_system_score_codex":0.0014787116,"about_ca_system_score_gemma":0.0019627083,"threshold_uncertainty_score":0.049735308},"labels":[],"label_agreement":null},{"id":"W2112473332","doi":"","title":"Stacking for Statistical Machine Translation","year":2013,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Machine translation; Computer science; Phrase; Artificial intelligence; Translation (biology); On the fly; Natural language processing; Stacking; Training set; Ensemble learning; Statistical model; Language model; Example-based machine translation; Machine learning; Statistical learning","score_opus":0.020024394380041378,"score_gpt":0.2961899521339451,"score_spread":0.2761655577539037,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2112473332","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0026523161,0.00047286952,0.9900011,0.00015305223,0.00019808506,0.000034949757,0.0001513137,0.0041733133,0.002162938],"genre_scores_gemma":[0.16300645,0.0010734003,0.82490987,0.0003621194,0.00063289807,0.00034351283,0.0018991822,0.0013648218,0.006407809],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9977017,0.0011391909,0.00014427782,0.00040671954,0.00048394976,0.00012412168],"domain_scores_gemma":[0.99555933,0.0017728383,0.00023333263,0.001650078,0.00064926967,0.00013508931],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027880592,0.0017169105,0.0014713682,0.0018162696,0.0013526793,0.0021836534,0.0015763906,0.0012374269,0.00850176],"category_scores_gemma":[0.008470693,0.0008982654,0.0016246367,0.0022296782,0.0011152272,0.0036524362,0.0027973412,0.002653847,0.008585205],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019275966,0.000103590006,0.0012855013,0.00028652954,0.0002456644,0.00023373462,0.00026767695,0.12424391,0.011785948,0.110896036,0.016650911,0.73380786],"study_design_scores_gemma":[0.000015110004,0.00014085571,0.00032872785,0.00004558251,0.000059253034,0.00018986355,0.00003772402,0.74109054,0.014642439,0.22194201,0.02145389,0.000054081036],"about_ca_topic_score_codex":0.0014260169,"about_ca_topic_score_gemma":0.0025014835,"teacher_disagreement_score":0.00850176,"about_ca_system_score_codex":0.00065871817,"about_ca_system_score_gemma":0.0012504149,"threshold_uncertainty_score":0.02844125},"labels":[],"label_agreement":null},{"id":"W2113376122","doi":"","title":"Analysis of Summarization Evaluation Experiments","year":2007,"lang":"en","type":"article","venue":"North American Chapter of the Association for Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Automatic summarization; Computer science; Terminology; Presentation (obstetrics); Information retrieval; Focus (optics); Natural language processing; Multi-document summarization; Artificial intelligence; Data mining; Linguistics","score_opus":0.01988773488424193,"score_gpt":0.3238424057267155,"score_spread":0.3039546708424736,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2113376122","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9035753,0.0055488767,0.058416937,0.001237951,0.00072232843,0.0055715954,0.008591801,0.0026839525,0.013651322],"genre_scores_gemma":[0.9487934,0.0007594679,0.031649172,0.00040497942,0.00022345826,0.005344893,0.0091498615,0.000791805,0.0028829863],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.8534503,0.09847936,0.015856704,0.0057684663,0.024676742,0.0017685331],"domain_scores_gemma":[0.44347715,0.44125932,0.02582361,0.021934252,0.06543244,0.0020732386],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.061781432,0.0019325874,0.0019894862,0.004615422,0.0014315575,0.0028411774,0.0013103386,0.0013090151,0.0032324598],"category_scores_gemma":[0.32872027,0.0005187626,0.0013402125,0.0045381193,0.0011363697,0.0031533872,0.0016316879,0.0018365968,0.0010815241],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.033460792,0.00812998,0.07484902,0.015759567,0.0055181226,0.0007873436,0.010672129,0.03390468,0.08933765,0.009160733,0.042479377,0.6759406],"study_design_scores_gemma":[0.004653532,0.053626813,0.46463132,0.0036560334,0.007907915,0.0016996225,0.010378724,0.12220708,0.2367916,0.019008059,0.074002095,0.0014372391],"about_ca_topic_score_codex":0.0008541035,"about_ca_topic_score_gemma":0.00077109627,"teacher_disagreement_score":0.061781432,"about_ca_system_score_codex":0.0023365717,"about_ca_system_score_gemma":0.0013208418,"threshold_uncertainty_score":0.32673538},"labels":[],"label_agreement":null},{"id":"W2113958117","doi":"10.3115/1220175.1220277","title":"Names and similarities on the web","year":2006,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":83,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Ranking (information retrieval); Information retrieval; Quality (philosophy); Scale (ratio); Natural language processing; Artificial intelligence; World Wide Web; Geography","score_opus":0.007681838690379605,"score_gpt":0.22164541536433785,"score_spread":0.21396357667395824,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2113958117","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4901693,0.030440489,0.2596105,0.008462283,0.0009106358,0.00019420718,0.0096827885,0.0038756204,0.19665425],"genre_scores_gemma":[0.88775367,0.006725197,0.08033366,0.0005864557,0.0006214323,0.00010123197,0.0047779037,0.00037212,0.018728338],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99771935,0.00063249853,0.00018862232,0.00037884718,0.0009976712,0.00008305134],"domain_scores_gemma":[0.9938146,0.0033524097,0.0008415243,0.0008644547,0.0009434499,0.00018360055],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010762669,0.00023823691,0.00037249082,0.009661334,0.0014013902,0.0038340169,0.00045252868,0.0010442999,0.0063629798],"category_scores_gemma":[0.011612983,0.00027424959,0.00029236317,0.012797674,0.0016450235,0.012676498,0.0026307593,0.0005244249,0.0017484208],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00030058625,0.000066793786,0.025662255,0.00057794654,0.00011200383,0.001185534,0.0035104898,0.0066859243,0.0060274033,0.4850241,0.02309674,0.4477503],"study_design_scores_gemma":[0.00003225635,0.00007007433,0.026434692,0.0002662312,0.000107249376,0.002954624,0.0030412718,0.028958652,0.0052860165,0.67397237,0.25878826,0.00008831352],"about_ca_topic_score_codex":0.0022312372,"about_ca_topic_score_gemma":0.0025049413,"teacher_disagreement_score":0.009661334,"about_ca_system_score_codex":0.00080425525,"about_ca_system_score_gemma":0.00041544074,"threshold_uncertainty_score":0.02128625},"labels":[],"label_agreement":null},{"id":"W2114580886","doi":"10.1007/s10844-010-0130-7","title":"A vector-space dynamic feature for phrase-based statistical machine translation","year":2010,"lang":"en","type":"article","venue":"Journal of Intelligent Information Systems","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Institute for Infocomm Research; McGill University","keywords":"Computer science; Phrase; Machine translation; Feature vector; Artificial intelligence; Feature (linguistics); Translation (biology); Natural language processing; Context (archaeology); Feature selection; Support vector machine; Function (biology); Decoding methods; Vector space; Vector space model; Feature engineering; Algorithm; Linguistics","score_opus":0.011041794452586082,"score_gpt":0.28511334140405076,"score_spread":0.2740715469514647,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2114580886","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.027158815,0.0008017231,0.96136034,0.00024368745,0.0002766343,0.00012158528,0.0021062235,0.005954913,0.001976215],"genre_scores_gemma":[0.38726535,0.0006689263,0.5963452,0.0001768952,0.00028659267,0.00043825788,0.008728821,0.0006517966,0.0054382053],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993368,0.00017222189,0.000075536176,0.0001586994,0.00018084004,0.000075864365],"domain_scores_gemma":[0.9992107,0.00027125634,0.00006905888,0.00014804446,0.0002644408,0.00003637939],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007198855,0.0005418954,0.0009047428,0.0016096537,0.0005962663,0.00091290596,0.00084568537,0.0008281308,0.006909385],"category_scores_gemma":[0.0021250166,0.00022014011,0.00074247393,0.0028466892,0.0003003929,0.0014069752,0.0011118244,0.00085272186,0.0033270197],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007078913,0.00015461186,0.0007779434,0.00015701025,0.00006816407,0.00017639353,0.000050295603,0.009700661,0.044371847,0.008430603,0.010594917,0.92480975],"study_design_scores_gemma":[0.00012486208,0.00053304853,0.0037210037,0.000042090847,0.000113789305,0.00062427414,0.00009851355,0.9139028,0.040933635,0.017663812,0.022143982,0.00009807409],"about_ca_topic_score_codex":0.0028536848,"about_ca_topic_score_gemma":0.0030931896,"teacher_disagreement_score":0.006909385,"about_ca_system_score_codex":0.0003806266,"about_ca_system_score_gemma":0.0009388211,"threshold_uncertainty_score":0.023114145},"labels":[],"label_agreement":null},{"id":"W2114914460","doi":"10.1109/ccece.1993.332374","title":"Using random graphs in learning past tenses","year":2002,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Redundancy (engineering); Artificial intelligence; Probabilistic logic; Exploit; Natural language processing; Connectionism; Knowledge base; Process (computing); Machine learning; Artificial neural network; Programming language","score_opus":0.03135444347380327,"score_gpt":0.2731964397984948,"score_spread":0.2418419963246915,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2114914460","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.046789866,0.0010222599,0.9444831,0.0006377367,0.0000855616,0.000094591436,0.000453301,0.0019687295,0.004464897],"genre_scores_gemma":[0.6422368,0.0012942082,0.34971848,0.00035227617,0.00015347272,0.00022219546,0.0018963574,0.00031781473,0.003808363],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99891806,0.000480374,0.000063220155,0.00032037208,0.00016319245,0.00005485543],"domain_scores_gemma":[0.99274045,0.0057238988,0.00048007365,0.00058050256,0.00034870818,0.00012628434],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019230803,0.0007133105,0.0005936807,0.0030587572,0.00068628165,0.0012900605,0.0012346468,0.00095360447,0.0024008467],"category_scores_gemma":[0.010548747,0.00057398446,0.0008950061,0.0017814034,0.0012777093,0.0046995385,0.00087872794,0.0010793204,0.0007226976],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003681719,0.0002616041,0.008197405,0.00038753424,0.00023185724,0.00030463314,0.00044300238,0.3840134,0.0039794166,0.0905817,0.0060208254,0.5052104],"study_design_scores_gemma":[0.00003158212,0.00006915446,0.0012205872,0.000051141207,0.00004374957,0.00011622253,0.00006401253,0.8302155,0.0020602252,0.16065343,0.0054233293,0.000050976232],"about_ca_topic_score_codex":0.0037376832,"about_ca_topic_score_gemma":0.0066357204,"teacher_disagreement_score":0.0037376832,"about_ca_system_score_codex":0.00090370333,"about_ca_system_score_gemma":0.00064065086,"threshold_uncertainty_score":0.010170341},"labels":[],"label_agreement":null},{"id":"W2115056464","doi":"","title":"Vector Space Model for Adaptation in Statistical Machine Translation","year":2013,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":42,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Phrase; Artificial intelligence; Machine translation; Vector space model; Feature vector; Support vector machine; Translation (biology); Natural language processing; Set (abstract data type); NIST","score_opus":0.027001354648422488,"score_gpt":0.28125031550766166,"score_spread":0.2542489608592392,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2115056464","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0034153275,0.0006134216,0.9937465,0.00026154827,0.00010577204,0.000044267537,0.00010717226,0.00049731636,0.0012085243],"genre_scores_gemma":[0.4457032,0.002955881,0.52785116,0.00073823164,0.0009651592,0.0009906967,0.0017215081,0.0008478243,0.018226307],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9970927,0.0015219804,0.0001566282,0.0005838689,0.0004971836,0.00014756534],"domain_scores_gemma":[0.99703634,0.0018643313,0.00018517309,0.0004113663,0.0004413177,0.00006152528],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031205423,0.00124247,0.0015554152,0.0013922673,0.0005687393,0.0017280253,0.0021190937,0.0014045916,0.0037630158],"category_scores_gemma":[0.008675523,0.00067200814,0.0016801376,0.0025243042,0.0014264563,0.003734233,0.001776247,0.003211385,0.0026942934],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022088125,0.00011663471,0.0010098721,0.00019040103,0.00021095137,0.0001357093,0.00026126966,0.6363341,0.002795722,0.12985514,0.0052942135,0.22357513],"study_design_scores_gemma":[0.000012119956,0.00004375437,0.00015531085,0.000011149232,0.000013992942,0.000040053397,0.000015693337,0.93677,0.0004974687,0.059188705,0.0032298903,0.000021817898],"about_ca_topic_score_codex":0.00561336,"about_ca_topic_score_gemma":0.0035744095,"teacher_disagreement_score":0.00561336,"about_ca_system_score_codex":0.0014050737,"about_ca_system_score_gemma":0.0010873657,"threshold_uncertainty_score":0.016503215},"labels":[],"label_agreement":null},{"id":"W2115147883","doi":"10.7202/002751ar","title":"A Thing-bound Approach to the Practice and Teaching of Technical Translation","year":2002,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Referent; Linguistics; Source text; Computer science; Target text; Semiotics; Field (mathematics); Natural language processing; Artificial intelligence; Mathematics; Philosophy","score_opus":0.04471045527401331,"score_gpt":0.2966357214475949,"score_spread":0.25192526617358163,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2115147883","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009873447,0.0069431844,0.43748942,0.074225225,0.004354244,0.00044656015,0.000041405492,0.0008338578,0.4657926],"genre_scores_gemma":[0.48984897,0.0142792715,0.32274333,0.025217207,0.003125939,0.001997562,0.00010354556,0.0008713205,0.14181288],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9860167,0.01066154,0.0003622129,0.0010612365,0.001383127,0.00051519205],"domain_scores_gemma":[0.98852926,0.0075873253,0.0006742945,0.0013642156,0.0007985577,0.0010463202],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0074416003,0.0010582358,0.0005725837,0.0021906255,0.0057030916,0.010330847,0.0025674526,0.0048780413,0.010792374],"category_scores_gemma":[0.015211104,0.00047282738,0.0009746867,0.0018842268,0.024916232,0.008214726,0.007615579,0.010807768,0.0044148057],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00004353195,0.00018340259,0.0003618047,0.00048027956,0.000017085637,0.00026302566,0.06210637,0.0008948261,0.0013326207,0.83302164,0.0212518,0.08004362],"study_design_scores_gemma":[0.00003104862,0.00016150337,0.00033903038,0.00075432006,0.00001150505,0.0005429085,0.017262343,0.0013860163,0.0012466992,0.3992407,0.57898426,0.000039751845],"about_ca_topic_score_codex":0.0013696183,"about_ca_topic_score_gemma":0.0020908366,"teacher_disagreement_score":0.010792374,"about_ca_system_score_codex":0.0053612185,"about_ca_system_score_gemma":0.005107486,"threshold_uncertainty_score":0.039355397},"labels":[],"label_agreement":null},{"id":"W2115340919","doi":"","title":"A Clustering Approach for Nearly Unsupervised Recognition of Nonliteral Language","year":2005,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":196,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Literal (mathematical logic); Cluster analysis; Artificial intelligence; Natural language processing; Schema (genetic algorithms); Context (archaeology); Set (abstract data type); Information retrieval; Programming language","score_opus":0.021145858739292596,"score_gpt":0.27271926084365633,"score_spread":0.25157340210436374,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2115340919","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00712444,0.00017028842,0.9869178,0.00008426962,0.000042860407,0.000098178076,0.00036356144,0.0042190305,0.0009795937],"genre_scores_gemma":[0.04692849,0.00010359719,0.9470997,0.00010602218,0.00003197645,0.00021407632,0.0029669504,0.0005164992,0.0020326888],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99767727,0.000429956,0.00018855606,0.0010735126,0.000517453,0.000113320726],"domain_scores_gemma":[0.99721897,0.0008521452,0.00029319114,0.00081566273,0.0007251673,0.00009485538],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014844248,0.0011170454,0.00090029783,0.0037649183,0.0013833967,0.0013321342,0.0029692338,0.0013862342,0.0030223643],"category_scores_gemma":[0.0048654894,0.00066496327,0.00108453,0.0029627162,0.0009103232,0.0021975008,0.0014485326,0.0015177809,0.003314878],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019513603,0.00022738037,0.0036218395,0.000557442,0.00023845489,0.00037261317,0.0010256557,0.014846264,0.08139215,0.019135118,0.017208174,0.8611798],"study_design_scores_gemma":[0.00008161165,0.00015447593,0.0077521275,0.00012569409,0.00010978906,0.0016893812,0.00076559023,0.7245261,0.10342288,0.08341992,0.077704094,0.000248358],"about_ca_topic_score_codex":0.0022023062,"about_ca_topic_score_gemma":0.006997027,"teacher_disagreement_score":0.0037649183,"about_ca_system_score_codex":0.00067089027,"about_ca_system_score_gemma":0.0011144637,"threshold_uncertainty_score":0.010110855},"labels":[],"label_agreement":null},{"id":"W2115847145","doi":"","title":"An Ensemble Model that Combines Syntactic and Semantic Clustering for Discriminative Dependency Parsing","year":2011,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Discriminative model; Parsing; Dependency grammar; Artificial intelligence; Dependency (UML); Natural language processing; Cluster analysis; Word (group theory); Bottom-up parsing; Top-down parsing; Mathematics","score_opus":0.06793025544303595,"score_gpt":0.30348728179251583,"score_spread":0.2355570263494799,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2115847145","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.039287414,0.0004048906,0.95060426,0.0003032707,0.00016057723,0.00007484438,0.0007479077,0.0047802706,0.003636593],"genre_scores_gemma":[0.4528016,0.0006280949,0.5288838,0.0004714912,0.00020375005,0.0002649043,0.0065326546,0.0010252604,0.009188435],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992305,0.0001954779,0.00003479804,0.00028000437,0.00017576452,0.000083370425],"domain_scores_gemma":[0.9989778,0.00029864005,0.000046650428,0.00026335372,0.00035094126,0.00006263156],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011116088,0.0014465945,0.00144356,0.0017192406,0.0009373887,0.00094188075,0.0017995635,0.0014012854,0.0027442402],"category_scores_gemma":[0.0022388294,0.00065596664,0.0015294367,0.0023626417,0.00037861642,0.0029407567,0.0016275048,0.001962331,0.0020608108],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024948735,0.0004767445,0.007610992,0.00013766126,0.00053432834,0.00028583957,0.0002540395,0.25096512,0.021000704,0.01328795,0.02585268,0.6793445],"study_design_scores_gemma":[0.0000067130504,0.000033938828,0.0007131586,0.0000073440788,0.000078211204,0.00006291019,0.000017264669,0.9880057,0.0028567493,0.0060721063,0.002124653,0.000021255397],"about_ca_topic_score_codex":0.008603338,"about_ca_topic_score_gemma":0.022994,"teacher_disagreement_score":0.008603338,"about_ca_system_score_codex":0.0006583309,"about_ca_system_score_gemma":0.0015539408,"threshold_uncertainty_score":0.017106533},"labels":[],"label_agreement":null},{"id":"W2115848042","doi":"","title":"On Hierarchical Re-ordering and Permutation Parsing for Phrase-based Decoding","year":2012,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Parsing; Permutation (music); Decoding methods; Phrase; Rule-based machine translation; Artificial intelligence; Machine translation; Constraint (computer-aided design); Hierarchical database model; Theoretical computer science; Natural language processing; Algorithm; Data mining; Mathematics","score_opus":0.02318303471784058,"score_gpt":0.29767156648745846,"score_spread":0.2744885317696179,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2115848042","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004023273,0.00009444498,0.9907025,0.00030778092,0.000032389078,0.000057676927,0.00009484574,0.0018017725,0.0028853086],"genre_scores_gemma":[0.113017,0.00029782872,0.8818508,0.00029145158,0.00010904917,0.00020499554,0.0006942517,0.0010415622,0.0024932045],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9957001,0.0022199787,0.00029413152,0.00062077073,0.000939455,0.00022550921],"domain_scores_gemma":[0.9841427,0.009407822,0.0006396073,0.0044487445,0.0012014629,0.0001596224],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004380537,0.0010702794,0.0008306555,0.0014094663,0.0012084346,0.0022394399,0.0020472587,0.00169663,0.0070742574],"category_scores_gemma":[0.01689008,0.0010053666,0.0012118269,0.002420976,0.0032529584,0.006199708,0.0032344118,0.0030482307,0.0038976746],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002752967,0.0001625609,0.0013524758,0.0002943209,0.00006874602,0.0004647099,0.00069228763,0.089915514,0.025759203,0.49293336,0.009492562,0.37858894],"study_design_scores_gemma":[0.000037063444,0.00007126997,0.00030068425,0.000048386522,0.00003869652,0.000300377,0.00005720611,0.5545856,0.015373855,0.41766426,0.011455793,0.000066837805],"about_ca_topic_score_codex":0.0031977089,"about_ca_topic_score_gemma":0.005603751,"teacher_disagreement_score":0.0070742574,"about_ca_system_score_codex":0.0013277624,"about_ca_system_score_gemma":0.0028684307,"threshold_uncertainty_score":0.023665786},"labels":[],"label_agreement":null},{"id":"W2115982722","doi":"","title":"Joint Training of Dependency Parsing Filters through Latent Support Vector Machines","year":2011,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Parsing; Artificial intelligence; Dependency grammar; Classifier (UML); Pointwise; Pattern recognition (psychology); Support vector machine; Graph; Machine learning; Theoretical computer science; Mathematics","score_opus":0.073102836970246,"score_gpt":0.2755741511709338,"score_spread":0.2024713142006878,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2115982722","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017146384,0.00021346088,0.9784219,0.0001921761,0.000043541764,0.00004296084,0.00014052048,0.0032178424,0.0005810714],"genre_scores_gemma":[0.35335696,0.00024508507,0.6404589,0.00027967952,0.00011670047,0.00026175226,0.0014920622,0.00037056976,0.0034183348],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9978777,0.00069271453,0.00014401044,0.0006892167,0.0003460761,0.00025025159],"domain_scores_gemma":[0.9929381,0.004906482,0.00037610138,0.00051529333,0.0010983407,0.00016577565],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0044696634,0.0013674,0.00168006,0.0022464776,0.0007854133,0.0017577485,0.002526935,0.002286157,0.0035583475],"category_scores_gemma":[0.010820898,0.00085986435,0.0014286865,0.0017554465,0.0007494475,0.00408205,0.0012488413,0.0035559712,0.0023440544],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00030867342,0.00034832762,0.0032918474,0.00012116054,0.00015436402,0.0001206883,0.00015502686,0.15605445,0.006222018,0.012043387,0.0067450996,0.81443495],"study_design_scores_gemma":[0.000007490032,0.00001853578,0.00016548373,0.000008573469,0.000013183859,0.000010460009,0.00001121781,0.9927193,0.0018815642,0.004751266,0.00040631476,0.000006561181],"about_ca_topic_score_codex":0.0050730575,"about_ca_topic_score_gemma":0.007413572,"teacher_disagreement_score":0.0050730575,"about_ca_system_score_codex":0.0010052569,"about_ca_system_score_gemma":0.0019094234,"threshold_uncertainty_score":0.02363813},"labels":[],"label_agreement":null},{"id":"W2116445331","doi":"10.7202/003443ar","title":"Speech Proportion and Accuracy in Simultaneous Interpretation from English into Korean","year":2002,"lang":"fr","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":30,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Phrase; Philosophy; Linguistics","score_opus":0.02684860412469443,"score_gpt":0.2760153941673649,"score_spread":0.2491667900426705,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2116445331","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9920025,0.0005567147,0.0016108974,0.000060204024,0.00004192342,0.000013603599,0.000314789,0.000039796378,0.0053596306],"genre_scores_gemma":[0.9966066,0.00025513463,0.00095950573,0.000028151246,0.000023855906,0.000014386409,0.00046396273,0.00009304643,0.0015552705],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99818796,0.00049699284,0.00031165345,0.00049456337,0.0003642427,0.00014459172],"domain_scores_gemma":[0.97868675,0.015678328,0.002131977,0.0009182628,0.002278173,0.00030648266],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024674549,0.00057727465,0.0005112569,0.000909406,0.00054007646,0.0021391893,0.0003524193,0.00051935756,0.0056817597],"category_scores_gemma":[0.020478696,0.0004507023,0.0003328095,0.0006783207,0.0006071334,0.0020321144,0.0017353385,0.0007171312,0.0017857526],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0073681087,0.00011951257,0.49751416,0.0018463793,0.0005730047,0.0026153321,0.05319526,0.0026553106,0.18934214,0.0012850931,0.0009550147,0.24253064],"study_design_scores_gemma":[0.00006670226,0.0008002962,0.9297568,0.0002215315,0.0005762509,0.0032878956,0.019309565,0.0067009353,0.032205444,0.0010331207,0.005904265,0.00013724306],"about_ca_topic_score_codex":0.0032222883,"about_ca_topic_score_gemma":0.0047869002,"teacher_disagreement_score":0.0056817597,"about_ca_system_score_codex":0.00036297744,"about_ca_system_score_gemma":0.0003606277,"threshold_uncertainty_score":0.019007385},"labels":[],"label_agreement":null},{"id":"W2116532807","doi":"10.3115/1117601.1117610","title":"Incorporating position information into a Maximum Entropy/Minimum Divergence translation model","year":2000,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Perplexity; Trigram; Translation (biology); Computer science; Divergence (linguistics); Principle of maximum entropy; Machine translation; Artificial intelligence; Entropy (arrow of time); Natural language processing; Language model; Kullback–Leibler divergence; Linguistics; Physics","score_opus":0.009933826212167873,"score_gpt":0.24136729305918722,"score_spread":0.23143346684701935,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2116532807","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007775062,0.00015672123,0.9886671,0.0002551288,0.000081408696,0.00003670494,0.00009290332,0.001002257,0.0019327201],"genre_scores_gemma":[0.26968536,0.00030351698,0.72274107,0.0002878232,0.0001815455,0.00021839981,0.0008120993,0.0005591755,0.0052109705],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99873966,0.00059316494,0.00008804594,0.0001896732,0.0003172931,0.000072076866],"domain_scores_gemma":[0.99857366,0.00088373467,0.00005405854,0.0002160437,0.00022642329,0.000046150257],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024730866,0.0011110451,0.0010020325,0.0011637416,0.0008057588,0.0015017922,0.0013086413,0.0014866939,0.0034445815],"category_scores_gemma":[0.0053855353,0.00068575406,0.0010972472,0.0011651752,0.00069517316,0.0027830708,0.0016352884,0.0018717457,0.0027837113],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007945336,0.0003600522,0.0018908597,0.0003410333,0.00023317857,0.0004224784,0.0004115617,0.36176676,0.025189243,0.06959503,0.0064458763,0.53254944],"study_design_scores_gemma":[0.000050552764,0.00008961994,0.00021378958,0.000019014702,0.00004562116,0.00009207426,0.00002045169,0.9518911,0.0060739554,0.038778137,0.0026949148,0.000030855546],"about_ca_topic_score_codex":0.0018193672,"about_ca_topic_score_gemma":0.0029321793,"teacher_disagreement_score":0.0034445815,"about_ca_system_score_codex":0.0006415579,"about_ca_system_score_gemma":0.0011455794,"threshold_uncertainty_score":0.013079107},"labels":[],"label_agreement":null},{"id":"W2116866745","doi":"","title":"Combinators’ Introduction: an Enhanced Algorithm","year":2009,"lang":"en","type":"article","venue":"The Florida AI Research Society","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Trois-Rivières","funders":"","keywords":"Combinatory logic; Computer science; Algorithm; Expression (computer science); Process (computing); Programming language; Theoretical computer science","score_opus":0.026919527125806215,"score_gpt":0.3670952944581607,"score_spread":0.3401757673323545,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2116866745","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0064888904,0.0000685754,0.97202253,0.00033545116,0.000112991984,0.00029996756,0.0002674473,0.010978622,0.009425536],"genre_scores_gemma":[0.03669267,0.00005902506,0.95292217,0.00018667575,0.000048847378,0.00014486878,0.0005330101,0.0015292673,0.00788359],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99673355,0.0005233448,0.00042550513,0.0009910217,0.00094774086,0.00037895085],"domain_scores_gemma":[0.99535984,0.0015532381,0.00018994679,0.0014805612,0.0012480179,0.00016835496],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022049441,0.001511245,0.0010158167,0.0018739828,0.0011738619,0.0032467768,0.0031181523,0.0019582533,0.026922021],"category_scores_gemma":[0.0083856145,0.00085502805,0.0020382768,0.0015691642,0.0018120895,0.0044197924,0.0036723136,0.0021090133,0.010562955],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009825948,0.00045939322,0.0029627963,0.00060245866,0.00011889107,0.0007168443,0.00078560214,0.022898963,0.024074998,0.14626864,0.03134824,0.7687805],"study_design_scores_gemma":[0.00067818706,0.0004168824,0.0014439505,0.00021958805,0.00027493175,0.002088233,0.0003391747,0.45490143,0.060707077,0.20175436,0.27691337,0.00026277592],"about_ca_topic_score_codex":0.0025941902,"about_ca_topic_score_gemma":0.0037891853,"teacher_disagreement_score":0.026922021,"about_ca_system_score_codex":0.001335818,"about_ca_system_score_gemma":0.0028853405,"threshold_uncertainty_score":0.090063095},"labels":[],"label_agreement":null},{"id":"W2117026386","doi":"10.1017/s0022226704243237","title":"<b>Martine Coene &amp; Yves D'hulst (eds.)</b>, <i>From NP to DP</i>, vol. I: <i>The syntax and semantics of noun phrases</i>. Amsterdam: John Benjamins, 2002. Pp. vi+359. <i>From NP to DP</i>, vol. II: <i>The expression of possession in noun phrases</i>. Amsterdam: John Benjamins, 2002. Pp. viii+291.","year":2005,"lang":"en","type":"article","venue":"Journal of Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Syntax; Noun phrase; Linguistics; Semantics (computer science); Possession (linguistics); Philosophy; Noun; Humanities; Computer science; Programming language","score_opus":0.011470926589489585,"score_gpt":0.26812249613767125,"score_spread":0.2566515695481817,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2117026386","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0006455315,0.92197806,0.0076966174,0.015859522,0.009029092,0.000021380696,0.0006351338,0.00024649088,0.0438882],"genre_scores_gemma":[0.004937221,0.9068876,0.006067817,0.0021698119,0.0045275325,0.000043836524,0.0010384376,0.0002517088,0.074076],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9993901,0.00010586571,0.000040009443,0.0001628432,0.00024969014,0.000051423485],"domain_scores_gemma":[0.99840933,0.0008589491,0.000108727225,0.00007575747,0.0003993488,0.00014783841],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011568321,0.002504796,0.0017476568,0.0036476362,0.0012754191,0.0042838715,0.0013319707,0.0019886645,0.06818708],"category_scores_gemma":[0.0022680785,0.0012932762,0.0006547444,0.0050757756,0.0010982046,0.010305508,0.0014461675,0.0022739582,0.059862126],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005422048,0.000030703155,0.00032243,0.0015496652,0.000027649463,0.000096938806,0.0005418034,0.00027080678,0.00037107852,0.005505012,0.7531058,0.23812392],"study_design_scores_gemma":[0.0000069652256,0.0000090436015,0.0015198487,0.0006802828,0.000028099106,0.0005137212,0.00043794353,0.000186837,0.00027843943,0.0041753133,0.99214023,0.00002324234],"about_ca_topic_score_codex":0.017467566,"about_ca_topic_score_gemma":0.028672257,"teacher_disagreement_score":0.06818708,"about_ca_system_score_codex":0.0018399468,"about_ca_system_score_gemma":0.00227766,"threshold_uncertainty_score":0.22810853},"labels":[],"label_agreement":null},{"id":"W2117339222","doi":"","title":"Transductive learning for statistical machine translation","year":2007,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":120,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"Simon Fraser University; National Research Council Canada","funders":"","keywords":"Machine translation; Computer science; NIST; Artificial intelligence; Natural language processing; Translation (biology); Strengths and weaknesses; Training set; Evaluation of machine translation; Machine translation software usability; Set (abstract data type); Quality (philosophy); Example-based machine translation; Machine learning","score_opus":0.016725809446192265,"score_gpt":0.30246721136948596,"score_spread":0.28574140192329367,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2117339222","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0016460049,0.0020730018,0.9926696,0.00050564663,0.0001041468,0.000060916416,0.00013356884,0.0007094001,0.0020975946],"genre_scores_gemma":[0.28724805,0.0063033164,0.69395405,0.0008654317,0.0012478109,0.0010886149,0.0017362057,0.00054979813,0.0070068254],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99726474,0.0015902607,0.00014082824,0.0004152795,0.00052194303,0.000066978784],"domain_scores_gemma":[0.9932013,0.0051814597,0.00030656884,0.0008263674,0.00042356233,0.00006079521],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030836614,0.0012065696,0.0012699256,0.0015908564,0.0005815809,0.0020558469,0.0014968743,0.001397082,0.003527578],"category_scores_gemma":[0.011053113,0.00051538384,0.0009746036,0.0025224977,0.0018974057,0.0027890175,0.0018192433,0.0033324822,0.002483329],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014230286,0.00020001293,0.0007757171,0.00090025616,0.00027623985,0.00024682807,0.00032683788,0.2577559,0.004537883,0.28197297,0.014077728,0.43878734],"study_design_scores_gemma":[0.00001378622,0.000044823195,0.00019769717,0.00004066571,0.000017951837,0.00005396134,0.000023048637,0.64439994,0.0012701806,0.3475408,0.0063719945,0.000025086792],"about_ca_topic_score_codex":0.00093343743,"about_ca_topic_score_gemma":0.00084871217,"teacher_disagreement_score":0.003527578,"about_ca_system_score_codex":0.0014575343,"about_ca_system_score_gemma":0.0007365689,"threshold_uncertainty_score":0.016308129},"labels":[],"label_agreement":null},{"id":"W2117424365","doi":"","title":"Learning Noun Phrase Query Segmentation","year":2007,"lang":"en","type":"article","venue":"Empirical Methods in Natural Language Processing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":106,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Query expansion; Web query classification; Query language; Natural language processing; RDF query language; Information retrieval; Artificial intelligence; Precision and recall; Web search query; Noun phrase; Sargable; Query optimization; Segmentation; Phrase; Set (abstract data type); Noun; Search engine","score_opus":0.028399328598676603,"score_gpt":0.44222973249310743,"score_spread":0.4138304038944308,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2117424365","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.25050855,0.002030118,0.6917123,0.0010633345,0.00026298198,0.001148817,0.006863247,0.036268998,0.010141596],"genre_scores_gemma":[0.56066024,0.00050805433,0.40411666,0.000647672,0.00019136112,0.0005061697,0.025483562,0.0008227079,0.0070635485],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9975783,0.0005348501,0.00019392687,0.00094788714,0.000542721,0.00020221129],"domain_scores_gemma":[0.9960024,0.0019544782,0.00032582122,0.000390885,0.001186548,0.00013998036],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017649435,0.0014613075,0.0013603509,0.0025547903,0.0009095654,0.00168334,0.00208545,0.0016077085,0.0044530006],"category_scores_gemma":[0.008356786,0.00057998253,0.001203247,0.0022643995,0.00095044426,0.0048339474,0.001263835,0.0013949742,0.004793142],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012238083,0.0008117332,0.01734042,0.0012220303,0.0002690185,0.0008031325,0.0013679931,0.05544667,0.10955614,0.012651204,0.064100266,0.73520756],"study_design_scores_gemma":[0.00010419522,0.0003907371,0.004715692,0.00005155077,0.00009915551,0.00046752888,0.000544894,0.92465705,0.04218281,0.011568925,0.01515079,0.000066753084],"about_ca_topic_score_codex":0.010329683,"about_ca_topic_score_gemma":0.010213315,"teacher_disagreement_score":0.010329683,"about_ca_system_score_codex":0.0017981628,"about_ca_system_score_gemma":0.0027169113,"threshold_uncertainty_score":0.020539165},"labels":[],"label_agreement":null},{"id":"W2117745860","doi":"10.3115/1220575.1220670","title":"Translating with non-contiguous phrases","year":2005,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":74,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"NIST; Computer science; Phrase; Machine translation; Translation (biology); Artificial intelligence; Natural language processing; Metric (unit); Maximization; Word (group theory); Noun phrase; Speech recognition; Mathematics; Engineering","score_opus":0.007506883938218075,"score_gpt":0.24509182120984488,"score_spread":0.2375849372716268,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2117745860","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025990227,0.00063164555,0.96500313,0.0003272237,0.0002664291,0.00011560064,0.00070959656,0.0024855162,0.004470705],"genre_scores_gemma":[0.14062873,0.00081948104,0.84581006,0.00037715345,0.0002977364,0.00028099152,0.0036000127,0.0012879479,0.00689789],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9973846,0.0010155886,0.0002202213,0.00064440427,0.0006399684,0.000095122086],"domain_scores_gemma":[0.99551976,0.0021932826,0.00029469054,0.0012886156,0.0006490783,0.000054642336],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015181488,0.0009582338,0.0010203861,0.00082669826,0.0007442644,0.0012723907,0.0011074272,0.00087154127,0.0069514443],"category_scores_gemma":[0.0078066536,0.0006399271,0.0008533697,0.0017936254,0.0007770623,0.0028443728,0.0017063185,0.0014380079,0.006609129],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006748428,0.000217887,0.001096968,0.0015475549,0.00021069589,0.0011167369,0.001494902,0.04180059,0.1702115,0.06675779,0.014205605,0.700665],"study_design_scores_gemma":[0.00026377136,0.0010147577,0.0025473845,0.00026629562,0.00024839147,0.0031051938,0.00084340456,0.44900683,0.25751013,0.12344443,0.16157374,0.00017570508],"about_ca_topic_score_codex":0.0006318297,"about_ca_topic_score_gemma":0.0011706917,"teacher_disagreement_score":0.0069514443,"about_ca_system_score_codex":0.0002971519,"about_ca_system_score_gemma":0.0010960883,"threshold_uncertainty_score":0.023254931},"labels":[],"label_agreement":null},{"id":"W2117827367","doi":"10.3115/1610075.1610084","title":"Phrasetable smoothing for statistical machine translation","year":2006,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":113,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"Advanced Research Projects Agency; Defense Advanced Research Projects Agency","keywords":"Smoothing; Metric (unit); Machine translation; Computer science; Translation (biology); Range (aeronautics); Yield (engineering); Artificial intelligence; BLEU; Machine learning; Algorithm; Computer vision; Engineering","score_opus":0.013017031704459411,"score_gpt":0.2779080599458409,"score_spread":0.2648910282413815,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2117827367","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0028063478,0.0008225628,0.9907059,0.00018887341,0.00011342022,0.00005390608,0.00026970837,0.0036871019,0.0013522339],"genre_scores_gemma":[0.13763393,0.0014576606,0.8513775,0.0003536229,0.0004671467,0.000500577,0.0023223269,0.002631856,0.0032554097],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99628,0.002106748,0.00021795645,0.00046513203,0.0008131267,0.00011707729],"domain_scores_gemma":[0.98549116,0.008755589,0.00065378234,0.0037698515,0.00117677,0.00015288932],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005558216,0.00133248,0.0010214463,0.0017559028,0.001442425,0.0016913604,0.0016158065,0.0014806681,0.010219699],"category_scores_gemma":[0.02949609,0.0005804963,0.0011911388,0.0045211464,0.00092811516,0.0030722136,0.0018338994,0.0024707972,0.0063831196],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007996119,0.00017789433,0.0017292061,0.00091295695,0.00039412398,0.00023049353,0.00032694225,0.104566306,0.030303672,0.08634764,0.032853592,0.7413576],"study_design_scores_gemma":[0.00014030529,0.0003495932,0.0019093942,0.00013719086,0.00025461172,0.0005815953,0.00009694216,0.64828306,0.047011368,0.2575782,0.043482464,0.00017526123],"about_ca_topic_score_codex":0.0018295045,"about_ca_topic_score_gemma":0.0026565478,"teacher_disagreement_score":0.010219699,"about_ca_system_score_codex":0.00058255345,"about_ca_system_score_gemma":0.0008726519,"threshold_uncertainty_score":0.03418827},"labels":[],"label_agreement":null},{"id":"W2118105954","doi":"10.1162/ling_a_00186","title":"Lapsed Derivations: Ternary Stress in Harmonic Serialism","year":2015,"lang":"en","type":"article","venue":"Linguistic Inquiry","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Linguistics; Mathematics; Stress (linguistics); Ternary operation; Harmonic; Philosophy; Computer science; Physics; Acoustics; Programming language","score_opus":0.055545185275947956,"score_gpt":0.32267134797224134,"score_spread":0.2671261626962934,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2118105954","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2797297,0.0008200886,0.3150434,0.0028886786,0.00042745858,0.000083025865,0.00034964463,0.0006203249,0.40003765],"genre_scores_gemma":[0.9628733,0.00018526,0.02014317,0.00026208823,0.00017497478,0.000027504237,0.00011236297,0.00017587013,0.016045544],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99952924,0.00013990502,0.000036336165,0.000114905284,0.00011995412,0.000059585193],"domain_scores_gemma":[0.99924856,0.000190748,0.000080108905,0.00028217328,0.0001340132,0.00006433789],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005342942,0.00043173233,0.00028095572,0.00046003697,0.0010563094,0.0021506597,0.00081628136,0.0007794373,0.008886755],"category_scores_gemma":[0.0022691446,0.00026347736,0.00065587164,0.00059782347,0.0038352117,0.0036085423,0.002321861,0.0015599197,0.0011706621],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000014245468,0.000004977302,0.00025387437,0.00001537741,0.0000026068765,0.000117343465,0.00043000732,0.0006407832,0.000643217,0.9914464,0.00034885132,0.006082352],"study_design_scores_gemma":[0.000014296402,0.000020482303,0.0003432673,0.000012267338,0.000004558245,0.00016269782,0.00017115066,0.005534269,0.0006908033,0.984606,0.008426154,0.000013974019],"about_ca_topic_score_codex":0.001592192,"about_ca_topic_score_gemma":0.00172185,"teacher_disagreement_score":0.008886755,"about_ca_system_score_codex":0.0012622782,"about_ca_system_score_gemma":0.00046709314,"threshold_uncertainty_score":0.029729128},"labels":[],"label_agreement":null},{"id":"W2118156848","doi":"10.7202/1006175ar","title":"La lexicographie et l’analyse de corpus : nouvelles perspectives","year":2011,"lang":"fr","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"Uppsala Universitet","keywords":"Humanities; Philosophy","score_opus":0.06421047437459143,"score_gpt":0.30183415291703647,"score_spread":0.23762367854244504,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2118156848","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.048550833,0.18417926,0.5462463,0.0779941,0.002490525,0.0002948915,0.0026011406,0.0007438855,0.1368991],"genre_scores_gemma":[0.3416952,0.12717967,0.48470515,0.0065758643,0.00398794,0.0011066213,0.0026680918,0.0012930621,0.030788338],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.97350895,0.019083204,0.0012365272,0.0018176484,0.0039835977,0.00037011682],"domain_scores_gemma":[0.9410178,0.049341824,0.0012991413,0.003999836,0.004031478,0.0003099995],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.017903471,0.0011404705,0.0014678442,0.011687997,0.0029521736,0.02143429,0.0015629936,0.0026260791,0.0065198517],"category_scores_gemma":[0.031795513,0.0009517351,0.0009899451,0.015385186,0.015299204,0.024145423,0.0042768964,0.0040585413,0.0014338994],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001747215,0.00005331958,0.0027237888,0.002167863,0.00014135824,0.0003835548,0.02043916,0.0016327592,0.0034007542,0.78709847,0.007466147,0.17431816],"study_design_scores_gemma":[0.00004157087,0.00007806664,0.0067480397,0.0031005524,0.000077899975,0.0012694758,0.028265603,0.0063211373,0.004943268,0.43399873,0.5150103,0.00014528325],"about_ca_topic_score_codex":0.012134588,"about_ca_topic_score_gemma":0.014160273,"teacher_disagreement_score":0.02143429,"about_ca_system_score_codex":0.0057566017,"about_ca_system_score_gemma":0.0051999367,"threshold_uncertainty_score":0.09468377},"labels":[],"label_agreement":null},{"id":"W2118195382","doi":"10.3115/1220175.1220231","title":"Semi-supervised learning of partial cognates using bilingual bootstrapping","year":2006,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Natural language processing; Bootstrapping (finance); Artificial intelligence; Context (archaeology); Meaning (existential); Cognate; Machine translation; Labeled data; Supervised learning; Linguistics; Mathematics; Artificial neural network; Psychology","score_opus":0.01800472544288471,"score_gpt":0.2751417126019726,"score_spread":0.2571369871590879,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2118195382","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.100029685,0.00029354842,0.8948831,0.00016849772,0.000041504387,0.00016811746,0.00021907254,0.0020202503,0.0021761858],"genre_scores_gemma":[0.68834364,0.00012415294,0.30702937,0.00022518444,0.0000911773,0.0004314329,0.0020071035,0.000264009,0.0014838862],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9962806,0.0021169374,0.0002104408,0.00082858093,0.0004009091,0.00016241333],"domain_scores_gemma":[0.9902506,0.006012442,0.0006551853,0.0013480018,0.0014663473,0.00026735844],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035594383,0.0010164778,0.0016993621,0.0017467919,0.0010794022,0.0013320671,0.0016662033,0.0012306923,0.0013631762],"category_scores_gemma":[0.011582962,0.0005306248,0.0010868495,0.0010679485,0.001246711,0.0021054617,0.0018186079,0.0012132204,0.0009758222],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012908682,0.0009974533,0.013899629,0.00051580544,0.00059550494,0.00079100794,0.0008304655,0.2032555,0.024439499,0.015446787,0.007965435,0.7299722],"study_design_scores_gemma":[0.000054576067,0.00007807282,0.001032612,0.000023546907,0.000040909945,0.00012902376,0.00009108872,0.9731363,0.005739437,0.01841045,0.0012377921,0.000026151169],"about_ca_topic_score_codex":0.0012762594,"about_ca_topic_score_gemma":0.0029213226,"teacher_disagreement_score":0.0035594383,"about_ca_system_score_codex":0.00049260573,"about_ca_system_score_gemma":0.0014637107,"threshold_uncertainty_score":0.018824339},"labels":[],"label_agreement":null},{"id":"W2118337010","doi":"10.1075/ijcl.10.2.05lem","title":"Two methods for extracting “specific” single-word terms from specialized corpora","year":2005,"lang":"en","type":"article","venue":"International Journal of Corpus Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Word (group theory); Focus (optics); Term (time); Precision and recall; Noun phrase; Recall; Field (mathematics); Noun; Linguistics; Mathematics","score_opus":0.04948392902264137,"score_gpt":0.39423930483809483,"score_spread":0.34475537581545346,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2118337010","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.036620807,0.0007078954,0.9525789,0.00017413427,0.00014960578,0.0013858539,0.0014133112,0.0036019492,0.0033674724],"genre_scores_gemma":[0.032615293,0.00019301311,0.9623982,0.000031777825,0.000042613025,0.0011224045,0.0020328753,0.0003079086,0.0012560664],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99396014,0.0012750599,0.0010925154,0.0017536967,0.0016981566,0.00022040863],"domain_scores_gemma":[0.98679763,0.004253006,0.0017381189,0.0033575215,0.0035501171,0.00030366788],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005126683,0.0013192042,0.001432201,0.014356004,0.0011732677,0.0025360351,0.0015612176,0.0011463604,0.005834443],"category_scores_gemma":[0.029727785,0.00086803845,0.0014719694,0.014766437,0.001292303,0.003570419,0.0024572732,0.0011830364,0.0024919927],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005625097,0.00017957043,0.009317426,0.0009573276,0.00031742756,0.00032160836,0.0021379238,0.003250317,0.04004342,0.011880409,0.0054617343,0.92557037],"study_design_scores_gemma":[0.0008954068,0.0018314215,0.15901415,0.0005442246,0.0014654027,0.010389545,0.0053492216,0.26666147,0.23798579,0.060241695,0.2547033,0.00091835664],"about_ca_topic_score_codex":0.0022512656,"about_ca_topic_score_gemma":0.0050292704,"teacher_disagreement_score":0.014356004,"about_ca_system_score_codex":0.0010165316,"about_ca_system_score_gemma":0.0018265818,"threshold_uncertainty_score":0.027112842},"labels":[],"label_agreement":null},{"id":"W2118441253","doi":"10.3115/1118905.1118907","title":"ProAlign","year":2003,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Word (group theory); Rank (graph theory); Artificial intelligence; Quality (philosophy); Natural language processing; Mathematics","score_opus":0.01107401398270731,"score_gpt":0.2570696799712576,"score_spread":0.24599566598855033,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2118441253","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006783682,0.0015283407,0.40397698,0.0013877982,0.0027339372,0.0010113504,0.08168131,0.37297955,0.12791698],"genre_scores_gemma":[0.03472543,0.0014999581,0.44964924,0.0016052403,0.00061634404,0.0013510099,0.30737022,0.09061489,0.112567574],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99758303,0.00041465112,0.00031003347,0.0008404549,0.00065761054,0.00019412873],"domain_scores_gemma":[0.9962676,0.001019867,0.00025071134,0.001060197,0.0011803972,0.00022121651],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020619996,0.0030654096,0.0015792991,0.0041959053,0.0027328858,0.005319081,0.0035332476,0.0022013213,0.18344678],"category_scores_gemma":[0.009001646,0.0021372999,0.0015532152,0.0034181415,0.00079332566,0.005828557,0.004701638,0.0032237666,0.23375456],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00059952715,0.00017259829,0.0014171744,0.001591416,0.00011410792,0.00056311203,0.0007948279,0.0008091054,0.015228264,0.018131075,0.7261323,0.23444654],"study_design_scores_gemma":[0.0000796484,0.000085750995,0.0006529583,0.00013723268,0.000059980423,0.00095697877,0.00026435015,0.005070061,0.017042179,0.01445824,0.96110433,0.00008838684],"about_ca_topic_score_codex":0.001169494,"about_ca_topic_score_gemma":0.0022228803,"teacher_disagreement_score":0.18344678,"about_ca_system_score_codex":0.0006517131,"about_ca_system_score_gemma":0.0016122496,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W2118491422","doi":"10.1007/s10590-004-7693-4","title":"A Morphological Tagger for Korean: Statistical Tagging Combined with Corpus-Based Morphological Rule Application","year":2004,"lang":"en","type":"article","venue":"Machine Translation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":30,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Social Sciences and Humanities Research Council of Canada; University of Pennsylvania; Yale University","keywords":"Agglutinative language; Computer science; Artificial intelligence; Natural language processing; Lemma (botany); Trigram; Part-of-speech tagging; Part of speech; Component (thermodynamics); Tag system; Rule-based system; Word (group theory); Precision and recall; Computational linguistics; Speech recognition; Linguistics; Parsing; Algorithm; Biology","score_opus":0.015257152663782294,"score_gpt":0.27391012940859066,"score_spread":0.25865297674480836,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2118491422","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022710482,0.00042271405,0.90474087,0.00028226041,0.0003989277,0.00039017687,0.00667253,0.060579263,0.0038026941],"genre_scores_gemma":[0.067322336,0.0003689145,0.9075293,0.00022431358,0.000086843975,0.00031593305,0.0148074785,0.0060249576,0.0033199154],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9985713,0.00032030488,0.00035867142,0.00038796905,0.00027761637,0.00008413343],"domain_scores_gemma":[0.9940685,0.0018759196,0.00051371945,0.0015920199,0.0017553956,0.00019454802],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002606715,0.0016427445,0.0014446451,0.0030687745,0.0010058264,0.0025860881,0.002144214,0.0013025072,0.009194428],"category_scores_gemma":[0.0057294695,0.0015022991,0.0012390027,0.003511168,0.00066259893,0.0035480037,0.0021465896,0.0015110824,0.013921952],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00064776227,0.00042588197,0.0045694713,0.0014106422,0.0002923441,0.002026009,0.00069095864,0.0075755683,0.16892336,0.009498956,0.050383423,0.7535557],"study_design_scores_gemma":[0.0003464982,0.00052252493,0.008789343,0.0002993366,0.0008724215,0.0050222497,0.0009232895,0.40455756,0.38221884,0.02779903,0.16816278,0.000486039],"about_ca_topic_score_codex":0.0017393536,"about_ca_topic_score_gemma":0.0037741347,"teacher_disagreement_score":0.009194428,"about_ca_system_score_codex":0.0004765277,"about_ca_system_score_gemma":0.0021737732,"threshold_uncertainty_score":0.03075844},"labels":[],"label_agreement":null},{"id":"W2118813542","doi":"10.3115/1596276.1596282","title":"Improved large margin dependency parsing via local constraints and laplacian regularization","year":2006,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Treebank; Dependency grammar; Margin (machine learning); Parsing; Artificial intelligence; Dependency (UML); Natural language processing; Correctness; Regularization (linguistics); Top-down parsing; Bottom-up parsing; Machine learning; Algorithm","score_opus":0.003942921075461761,"score_gpt":0.21855651250564342,"score_spread":0.21461359143018166,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2118813542","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006729309,0.000091102826,0.98853195,0.00017934205,0.000028178847,0.000028601644,0.00013098141,0.0035203164,0.00076023454],"genre_scores_gemma":[0.16003926,0.00015197664,0.83032095,0.0004982827,0.00013197186,0.00022284774,0.0020613999,0.0013185248,0.00525472],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9980976,0.0007622673,0.000094050585,0.0004533848,0.00045338305,0.00013928066],"domain_scores_gemma":[0.9963128,0.0018700327,0.0001907687,0.0009017048,0.00061724585,0.000107513486],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002637873,0.0013299208,0.0013556675,0.0013609126,0.00085174415,0.0010914094,0.0024447027,0.0019876333,0.004964887],"category_scores_gemma":[0.007126535,0.0007690533,0.0014188733,0.00218473,0.001035842,0.0037276894,0.0028000202,0.0033498742,0.002647626],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028505383,0.0004188673,0.0015516941,0.00016368108,0.00017816483,0.0003483822,0.00029911826,0.34179774,0.023905404,0.030585004,0.025745176,0.57472163],"study_design_scores_gemma":[0.00001782537,0.000029125844,0.00018128067,0.0000059132794,0.00001372937,0.00003609237,0.000010808398,0.97826254,0.0038068192,0.016211834,0.0014039333,0.00002009161],"about_ca_topic_score_codex":0.0039128717,"about_ca_topic_score_gemma":0.00764764,"teacher_disagreement_score":0.004964887,"about_ca_system_score_codex":0.00064991653,"about_ca_system_score_gemma":0.0018910113,"threshold_uncertainty_score":0.016609192},"labels":[],"label_agreement":null},{"id":"W2118871236","doi":"10.3115/1687878.1687889","title":"Topological field parsing of German","year":2009,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Parsing; Computer science; Natural language processing; Parser combinator; Artificial intelligence; Bottom-up parsing; Top-down parsing; German; Sentence; Field (mathematics); S-attributed grammar; Word order; Linguistics; Mathematics; Pure mathematics","score_opus":0.01351561084332834,"score_gpt":0.31755318585511927,"score_spread":0.3040375750117909,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2118871236","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8173971,0.0007868098,0.13428971,0.0009656062,0.00012375374,0.00008354334,0.015646836,0.0060805627,0.024626082],"genre_scores_gemma":[0.932419,0.00026198485,0.052784335,0.00008563343,0.000020565432,0.000045132794,0.0110985385,0.00057672267,0.0027081626],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99962544,0.00014695922,0.000023741166,0.0000972676,0.00006679374,0.000039701186],"domain_scores_gemma":[0.99926573,0.00044933127,0.00005815793,0.00010733269,0.00010268966,0.00001682521],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005328054,0.0005412607,0.0002612217,0.0011869492,0.00051577104,0.000825805,0.00035396178,0.000403585,0.003494449],"category_scores_gemma":[0.0018168368,0.00027741294,0.00032864662,0.001089402,0.0006244146,0.0016142413,0.0005212198,0.00044973794,0.00069945585],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008648026,0.00019175117,0.015475034,0.0015147213,0.00017188153,0.0023143275,0.0072025773,0.079312645,0.09333547,0.42389995,0.055549134,0.32016772],"study_design_scores_gemma":[0.0002120329,0.00024002702,0.06058299,0.00025609782,0.00024454264,0.0020047198,0.003564944,0.3477038,0.17205305,0.22386967,0.18896341,0.00030457866],"about_ca_topic_score_codex":0.0038849893,"about_ca_topic_score_gemma":0.0065591233,"teacher_disagreement_score":0.0038849893,"about_ca_system_score_codex":0.0009369472,"about_ca_system_score_gemma":0.000548697,"threshold_uncertainty_score":0.01169008},"labels":[],"label_agreement":null},{"id":"W2118947254","doi":"","title":"Applying Many-to-Many Alignments and Hidden Markov Models to Letter-to-Phoneme Conversion","year":2007,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":186,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Bigram; Hidden Markov model; Computer science; Chunking (psychology); Speech recognition; Artificial intelligence; Preprocessor; Word (group theory); Training set; Conjunction (astronomy); Natural language processing; Viterbi algorithm; Trigram","score_opus":0.015816293560930495,"score_gpt":0.26100172716669645,"score_spread":0.24518543360576595,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2118947254","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.043971017,0.00060353015,0.9199556,0.00025885543,0.00018884975,0.00007168475,0.0005680666,0.030901713,0.0034806675],"genre_scores_gemma":[0.33847347,0.0005624297,0.6494698,0.0002610312,0.000120824436,0.00009984538,0.0024063908,0.0012233423,0.007382963],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993074,0.0002712148,0.000043757118,0.00020940036,0.0001249856,0.00004332465],"domain_scores_gemma":[0.99756,0.0014329581,0.00013821997,0.00050202594,0.0002831023,0.000083663654],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010561127,0.00072810595,0.000800651,0.0006988182,0.00063286995,0.0009931618,0.0010588645,0.0008864488,0.0043417513],"category_scores_gemma":[0.004595002,0.0005130439,0.00043697393,0.001058276,0.00038367364,0.0023186863,0.0010011242,0.001395681,0.004742158],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00045347173,0.00019449396,0.0039213304,0.00025759754,0.000107276785,0.0005342638,0.00040050584,0.032897547,0.037209116,0.0052436832,0.012845751,0.9059349],"study_design_scores_gemma":[0.00004507207,0.00012213048,0.0025526935,0.000053912936,0.00007457801,0.00062116707,0.00015835967,0.88779914,0.06948513,0.025423901,0.013571418,0.00009241603],"about_ca_topic_score_codex":0.003507531,"about_ca_topic_score_gemma":0.0062263208,"teacher_disagreement_score":0.0043417513,"about_ca_system_score_codex":0.0003636809,"about_ca_system_score_gemma":0.00067724846,"threshold_uncertainty_score":0.014524579},"labels":[],"label_agreement":null},{"id":"W2119029180","doi":"10.1109/icassp.1994.389724","title":"Learning consistent semantics from training data","year":2002,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Computer Research Institute of Montréal","funders":"","keywords":"Utterance; Computer science; Context (archaeology); Sentence; Task (project management); Component (thermodynamics); Programmer; Semantics (computer science); Artificial intelligence; Natural language processing; Parsing; Natural language; Representation (politics); Word (group theory); Programming language; Linguistics","score_opus":0.10584488557106479,"score_gpt":0.28854655451308786,"score_spread":0.18270166894202305,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2119029180","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.068668835,0.0015919054,0.9068879,0.0015144179,0.000264814,0.00032643505,0.005142006,0.011296275,0.004307423],"genre_scores_gemma":[0.5621454,0.00095218053,0.4003584,0.0008016011,0.0002776834,0.0008183102,0.03049424,0.0009841088,0.0031679776],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99674404,0.0010383704,0.00022389596,0.0013765872,0.00045340965,0.00016375157],"domain_scores_gemma":[0.98941946,0.0074147875,0.0003501923,0.0018910314,0.00077197305,0.00015254362],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003785634,0.0027867255,0.0018097552,0.0028809113,0.00093535864,0.00282299,0.0032820096,0.0023664874,0.0051017925],"category_scores_gemma":[0.018489426,0.0011823254,0.002064845,0.0020839483,0.0015734282,0.0060227877,0.0022491438,0.0040425025,0.0026687535],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007431154,0.00067879906,0.007813914,0.0009039338,0.00043331194,0.0006453331,0.0005230951,0.17380767,0.007359029,0.01688329,0.021993343,0.7682152],"study_design_scores_gemma":[0.00015183864,0.00022140566,0.00093535136,0.00015837779,0.0001235344,0.00015654367,0.00033580547,0.8803619,0.008429357,0.10022171,0.008862754,0.000041292034],"about_ca_topic_score_codex":0.0029219254,"about_ca_topic_score_gemma":0.006239195,"teacher_disagreement_score":0.0051017925,"about_ca_system_score_codex":0.0013777057,"about_ca_system_score_gemma":0.0023445296,"threshold_uncertainty_score":0.020020604},"labels":[],"label_agreement":null},{"id":"W2119168550","doi":"10.1162/0891201042544884","title":"The Alignment Template Approach to Statistical Machine Translation","year":2004,"lang":"en","type":"article","venue":"Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":933,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; NIST; Machine translation; Evaluation of machine translation; Natural language processing; Example-based machine translation; Artificial intelligence; Phrase; Machine translation software usability; Generalization; Transfer-based machine translation; Translation (biology); Task (project management); Context (archaeology); Feature (linguistics); Rule-based machine translation; Word (group theory); Linguistics","score_opus":0.021864809993940373,"score_gpt":0.2953628508639949,"score_spread":0.2734980408700545,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2119168550","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00032148973,0.0006344055,0.99529654,0.0001381774,0.000114338414,0.00006476677,0.00011830217,0.0012274872,0.0020845283],"genre_scores_gemma":[0.032028638,0.002575733,0.95582753,0.00031317843,0.0005394523,0.00062630436,0.0012090727,0.00087788515,0.0060021975],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99611264,0.0018079082,0.00028585255,0.0006366392,0.0010363548,0.00012065727],"domain_scores_gemma":[0.996811,0.0013544149,0.00025269474,0.0009296159,0.0005798683,0.00007237609],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028895345,0.0015358625,0.0018877751,0.0025173943,0.0008623259,0.0027955528,0.0025713756,0.0019072617,0.0071182833],"category_scores_gemma":[0.007980834,0.00095844525,0.001889473,0.004376983,0.0012772084,0.0030653833,0.0020925466,0.0029320035,0.00947412],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010321522,0.00011957382,0.00048448256,0.00068706775,0.00037799738,0.0002821159,0.00020486473,0.062465634,0.013545807,0.31667095,0.020619951,0.5844383],"study_design_scores_gemma":[0.00004647604,0.00018106682,0.00047337727,0.00008944585,0.00011151533,0.00068993075,0.00005223907,0.46759695,0.011836116,0.41726968,0.101519525,0.00013364301],"about_ca_topic_score_codex":0.0018464096,"about_ca_topic_score_gemma":0.0016765089,"teacher_disagreement_score":0.0071182833,"about_ca_system_score_codex":0.0009297866,"about_ca_system_score_gemma":0.0020747269,"threshold_uncertainty_score":0.02381301},"labels":[],"label_agreement":null},{"id":"W2119388680","doi":"10.3115/1220175.1220240","title":"Improved discriminative bilingual word alignment","year":2006,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":72,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Discriminative model; Generative grammar; Computer science; Word (group theory); Word error rate; Artificial intelligence; Machine translation; Natural language processing; Bilingual dictionary; Generative model; Selection (genetic algorithm); Translation (biology); Training set; Statistical model; Machine learning; Speech recognition; Linguistics","score_opus":0.00950746876754677,"score_gpt":0.2645663555949464,"score_spread":0.25505888682739963,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2119388680","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07741404,0.0009654794,0.8787384,0.00058404455,0.00040400954,0.0001279735,0.0026841157,0.023551868,0.015530109],"genre_scores_gemma":[0.5195685,0.00047695628,0.44212118,0.0005952179,0.0002148005,0.0002051697,0.017800009,0.0028126417,0.01620551],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9981104,0.00063578907,0.00009329621,0.0006857203,0.00029976695,0.0001751043],"domain_scores_gemma":[0.9983071,0.00045418175,0.00009903803,0.00060500414,0.00043188987,0.00010284673],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014664109,0.0018002257,0.0018891403,0.0018909511,0.001203729,0.0010853186,0.0012304425,0.0009354455,0.0087263],"category_scores_gemma":[0.004823803,0.0006926701,0.000937282,0.002915068,0.00051423925,0.0027927202,0.00217125,0.0019098661,0.011663247],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006520112,0.0006921864,0.007996449,0.00041214627,0.00021072193,0.0004761292,0.00052086235,0.12321185,0.053821478,0.030794743,0.06569216,0.71551925],"study_design_scores_gemma":[0.0000940295,0.000111081696,0.001915374,0.000024580486,0.00007758222,0.00053756265,0.00009764326,0.93438816,0.020585394,0.020621648,0.021492518,0.000054444234],"about_ca_topic_score_codex":0.006270446,"about_ca_topic_score_gemma":0.021260884,"teacher_disagreement_score":0.0087263,"about_ca_system_score_codex":0.0006904691,"about_ca_system_score_gemma":0.0021949115,"threshold_uncertainty_score":0.029192388},"labels":[],"label_agreement":null},{"id":"W2119516574","doi":"10.1016/j.scico.2013.11.007","title":"Parse views with Boolean grammars","year":2013,"lang":"en","type":"article","venue":"Science of Computer Programming","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Tree-adjoining grammar; Parsing; Parsing expression grammar; Phrase structure grammar; Rule-based machine translation; Programming language; Context-free grammar; L-attributed grammar; Natural language processing; Indexed grammar; S-attributed grammar; Negation; Artificial intelligence","score_opus":0.01395245520905636,"score_gpt":0.261069871613509,"score_spread":0.24711741640445262,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2119516574","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018570796,0.00041890933,0.94268775,0.0012516845,0.00025070625,0.00011399691,0.001785199,0.011023678,0.02389716],"genre_scores_gemma":[0.34648567,0.0011193338,0.61754245,0.0009529688,0.00027887663,0.0002479869,0.0073239673,0.0075420663,0.018506652],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9975848,0.00067993585,0.00018152792,0.00046928643,0.00084735925,0.00023717331],"domain_scores_gemma":[0.99655735,0.0018138706,0.0001225224,0.0009503295,0.00043025205,0.00012573287],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018746922,0.0008613679,0.0011441169,0.0017756011,0.0011887159,0.005185645,0.0016599289,0.0016638117,0.016134696],"category_scores_gemma":[0.008916422,0.0014203781,0.0022440453,0.0022035567,0.0021714712,0.012207511,0.004858282,0.0041037668,0.003916131],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015396548,0.000065608416,0.0005739429,0.00018422844,0.000058903785,0.00020187451,0.0007056415,0.0049335505,0.002331151,0.8973828,0.015973412,0.077434935],"study_design_scores_gemma":[0.00004286393,0.000012636741,0.00011775431,0.000059600872,0.00005833181,0.00008305722,0.00012281981,0.026630582,0.003914357,0.9451626,0.023761483,0.00003393116],"about_ca_topic_score_codex":0.0029826122,"about_ca_topic_score_gemma":0.0053572278,"teacher_disagreement_score":0.016134696,"about_ca_system_score_codex":0.0011310601,"about_ca_system_score_gemma":0.001174128,"threshold_uncertainty_score":0.053976},"labels":[],"label_agreement":null},{"id":"W2119652412","doi":"10.3138/cmlr.1726.436","title":"A Corpus-Based Assessment of French CEFR Lexical Content","year":2013,"lang":"en","type":"article","venue":"Canadian Modern Language Review/ La Revue canadienne des langues vivantes","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Vocabulary; Computer science; Corpus linguistics; Natural language processing; Intuition; Artificial intelligence; Introspection; Vocabulary development; Perspective (graphical); Linguistics; Psychology; Cognitive psychology","score_opus":0.021572081074702893,"score_gpt":0.2600412722368445,"score_spread":0.23846919116214157,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2119652412","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8824684,0.0071329055,0.0224223,0.0012292376,0.00018105812,0.0006353347,0.020972377,0.00064332556,0.064315066],"genre_scores_gemma":[0.948799,0.0017496232,0.02440949,0.00016541255,0.0000849538,0.00061086816,0.019504167,0.00018876624,0.004487805],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99498063,0.0021889915,0.0003804519,0.000754111,0.0015190019,0.0001767436],"domain_scores_gemma":[0.9780471,0.008535112,0.0014490226,0.0017828689,0.009825972,0.0003599614],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007447587,0.0004523646,0.00052558107,0.016504243,0.0016775356,0.0025766569,0.00083079515,0.0006048382,0.0052567315],"category_scores_gemma":[0.027690489,0.0001789017,0.00026720524,0.010240775,0.001151682,0.0017928043,0.0017475637,0.000499293,0.0011296985],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008620133,0.00028912668,0.18960789,0.0039617307,0.0003373768,0.0016111565,0.05053228,0.0027978178,0.025411045,0.024427699,0.048777983,0.6513838],"study_design_scores_gemma":[0.00012312274,0.0004929933,0.58350855,0.0020262806,0.00033545785,0.003292848,0.03695297,0.009664577,0.016357495,0.0032967504,0.34373125,0.00021782306],"about_ca_topic_score_codex":0.09098182,"about_ca_topic_score_gemma":0.091054,"teacher_disagreement_score":0.09098182,"about_ca_system_score_codex":0.0032600733,"about_ca_system_score_gemma":0.0027508838,"threshold_uncertainty_score":0.18090445},"labels":[],"label_agreement":null},{"id":"W2119707623","doi":"","title":"Classifying arguments by scheme","year":2011,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":178,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Argumentation theory; Scheme (mathematics); Argument (complex analysis); Classification scheme; Pairwise comparison; Baseline (sea); Computer science; Artificial intelligence; Mathematics; Pattern recognition (psychology); Theoretical computer science; Machine learning; Epistemology; Medicine","score_opus":0.03213522819208245,"score_gpt":0.26389262181178097,"score_spread":0.23175739361969852,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2119707623","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.18100043,0.0011847056,0.7410713,0.0028971874,0.00046291205,0.0022630629,0.006880882,0.01284676,0.051392812],"genre_scores_gemma":[0.37088916,0.0005603703,0.60218036,0.00029278774,0.00012643992,0.00052501966,0.010045598,0.0008089319,0.014571307],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9948908,0.0012389345,0.0007049588,0.0009387084,0.0019117617,0.00031487865],"domain_scores_gemma":[0.9888888,0.0050272904,0.0010370488,0.0022735198,0.0023776812,0.00039562298],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0047257044,0.00096695876,0.00093410874,0.008104629,0.0014368321,0.005016909,0.0018854332,0.0019943907,0.0125347525],"category_scores_gemma":[0.023490308,0.0005405369,0.0018802099,0.0036614083,0.0011137653,0.007952103,0.0022276295,0.0017831994,0.004451709],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007084246,0.0004364586,0.0485242,0.0010403852,0.0001818953,0.00079642114,0.004415552,0.0044991043,0.029458445,0.18088101,0.033023424,0.69603467],"study_design_scores_gemma":[0.00016346778,0.0003619939,0.03693354,0.00087574095,0.00037612813,0.0019684094,0.0035497658,0.2892547,0.05352001,0.24211252,0.37067616,0.00020763805],"about_ca_topic_score_codex":0.001690499,"about_ca_topic_score_gemma":0.0019525924,"teacher_disagreement_score":0.0125347525,"about_ca_system_score_codex":0.0014185314,"about_ca_system_score_gemma":0.0016135635,"threshold_uncertainty_score":0.041933},"labels":[],"label_agreement":null},{"id":"W2120128214","doi":"10.1007/3-540-44686-9_28","title":"Experiments on Extracting Knowledge from a Machine-Readable Dictionary of Synonym Differences","year":2001,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Synonym (taxonomy); Natural language processing; Artificial intelligence; Word (group theory); Sentence; Knowledge base; Set (abstract data type); Part of speech; Machine translation; Bilingual dictionary; Linguistics; Programming language","score_opus":0.024582961851504867,"score_gpt":0.28740466966711753,"score_spread":0.26282170781561265,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2120128214","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.94513786,0.0025366968,0.023346484,0.0006988359,0.00040317007,0.0010174166,0.011806749,0.0053444146,0.009708396],"genre_scores_gemma":[0.8278163,0.0016319022,0.11212359,0.0004266592,0.0001385998,0.0007956791,0.050035503,0.0005937484,0.006438007],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99685514,0.0012066129,0.0006165605,0.0007781671,0.00038150867,0.00016213313],"domain_scores_gemma":[0.93183964,0.059759397,0.001162618,0.0039974735,0.0024632162,0.00077770127],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031510722,0.0017353112,0.0014717025,0.0020031817,0.0010398993,0.0015133482,0.0025541952,0.0024956437,0.008088108],"category_scores_gemma":[0.031998783,0.00075321714,0.0012415,0.0032055383,0.00090075686,0.006491916,0.0016243575,0.0020171853,0.0030560496],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.013756188,0.0143374065,0.021955308,0.00960255,0.0014786727,0.0034538733,0.0067768465,0.025643025,0.07320584,0.0026678015,0.043674868,0.78344756],"study_design_scores_gemma":[0.007691179,0.021568023,0.09713891,0.0014716703,0.0045552957,0.007360591,0.015215398,0.49171612,0.25113449,0.019194216,0.08205715,0.000897055],"about_ca_topic_score_codex":0.0062590265,"about_ca_topic_score_gemma":0.0058598383,"teacher_disagreement_score":0.008088108,"about_ca_system_score_codex":0.0007106267,"about_ca_system_score_gemma":0.0010291983,"threshold_uncertainty_score":0.02705741},"labels":[],"label_agreement":null},{"id":"W2120158994","doi":"10.5539/cis.v5n1p13","title":"Unsupervised Query Segmentation Using Monolingual Word Alignment Method","year":2011,"lang":"en","type":"article","venue":"Computer and Information Science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Query expansion; Segmentation; Artificial intelligence; Query language; Web query classification; Natural language processing; Query optimization; Sargable; Text segmentation; RDF query language; Word (group theory); Language model; Market segmentation; Query by Example; Web search query; Information retrieval; Search engine","score_opus":0.03692806809415005,"score_gpt":0.3103869872140255,"score_spread":0.27345891911987547,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2120158994","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.027047116,0.0008217327,0.96075475,0.00018895291,0.000076458484,0.00029413952,0.0005806524,0.0065042414,0.0037319902],"genre_scores_gemma":[0.2928455,0.00061976456,0.6900895,0.00045648764,0.00023620689,0.000625644,0.006327012,0.0013416592,0.007458213],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9977323,0.0004676711,0.00020765014,0.00084195094,0.00055799,0.00019240228],"domain_scores_gemma":[0.99839693,0.00043939473,0.00016709573,0.0002620432,0.0006667696,0.00006780624],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007132316,0.0013800287,0.0015109911,0.003945894,0.0010910181,0.0013324629,0.0015524967,0.000945747,0.0041639223],"category_scores_gemma":[0.002771203,0.0004922586,0.0012068892,0.0035792158,0.0007989484,0.0026852945,0.0015225628,0.0011523861,0.0034832673],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005913322,0.0004154673,0.0031728027,0.00065698155,0.00019705595,0.0006436759,0.0013550756,0.018598143,0.229191,0.013959041,0.016176997,0.7150425],"study_design_scores_gemma":[0.00018290752,0.00065442635,0.007209769,0.000062710205,0.00031403973,0.0017402556,0.0014598892,0.7391043,0.15974389,0.028505659,0.060770053,0.0002522081],"about_ca_topic_score_codex":0.004960249,"about_ca_topic_score_gemma":0.006010144,"teacher_disagreement_score":0.004960249,"about_ca_system_score_codex":0.0008739096,"about_ca_system_score_gemma":0.0021645527,"threshold_uncertainty_score":0.013929725},"labels":[],"label_agreement":null},{"id":"W2120344709","doi":"","title":"Beyond the Transfer-and-Merge Wordnet Construction: plWordNet and a Comparison with WordNet","year":2013,"lang":"en","type":"article","venue":"Recent Advances in Natural Language Processing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"WordNet; Computer science; Merge (version control); Natural language processing; Artificial intelligence; Perspective (graphical); Information retrieval","score_opus":0.005192038935594544,"score_gpt":0.2609067758639921,"score_spread":0.2557147369283975,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2120344709","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06801297,0.020627018,0.75269884,0.017691584,0.0018069282,0.00024666957,0.0009060086,0.0013089532,0.13670105],"genre_scores_gemma":[0.63319135,0.011159034,0.33170703,0.0018407069,0.0012054872,0.000493242,0.0015581347,0.000853185,0.017991913],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9975777,0.001327453,0.0001388153,0.00036275422,0.00045501246,0.00013834877],"domain_scores_gemma":[0.99524045,0.0027100628,0.00040815005,0.0007413118,0.00062303507,0.00027706506],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0040845457,0.00093786314,0.001074031,0.006172776,0.0027389769,0.00705978,0.0017056843,0.0020728824,0.007441776],"category_scores_gemma":[0.013165693,0.0004061795,0.0008318539,0.007909585,0.006519688,0.025694797,0.004417378,0.002568766,0.0015059478],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006908691,0.00003065193,0.0008566981,0.00015221308,0.000031071475,0.000118277385,0.00057711813,0.0020039044,0.00046289366,0.94830066,0.002657265,0.04474027],"study_design_scores_gemma":[0.0000077768345,0.000041243045,0.00060823513,0.00011883793,0.000018158615,0.00027794184,0.00057991315,0.01628547,0.0009659378,0.94824064,0.032833543,0.000022307062],"about_ca_topic_score_codex":0.0018134157,"about_ca_topic_score_gemma":0.0026929842,"teacher_disagreement_score":0.007441776,"about_ca_system_score_codex":0.002088344,"about_ca_system_score_gemma":0.001249616,"threshold_uncertainty_score":0.02489525},"labels":[],"label_agreement":null},{"id":"W2120502101","doi":"10.1145/1148170.1148303","title":"Swordfish","year":2006,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Morpheme; Prefix; Computer science; Natural language processing; Word (group theory); Artificial intelligence; Turkish; Task (project management); Linguistics","score_opus":0.005190354507070602,"score_gpt":0.23160760409238165,"score_spread":0.22641724958531106,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2120502101","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04379799,0.0014047383,0.2998209,0.001578235,0.0011633558,0.0009198729,0.051090166,0.27711746,0.32310727],"genre_scores_gemma":[0.119171016,0.0009764778,0.37346527,0.0011903862,0.00019872739,0.0006757723,0.12956665,0.04455709,0.3301986],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99908876,0.000070447,0.00009366112,0.0002609055,0.00040828387,0.000078027704],"domain_scores_gemma":[0.99829084,0.00027987396,0.00010373444,0.0006727832,0.00052746804,0.00012531847],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.00079100335,0.0010198127,0.0007005782,0.002376068,0.0010086794,0.0018470725,0.0019372081,0.00096022745,0.08141172],"category_scores_gemma":[0.0035271824,0.00073455926,0.00087348884,0.0020217183,0.0006024848,0.003688871,0.0034753908,0.0008131914,0.068777986],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006757829,0.000086726286,0.0043611233,0.00081794424,0.00008878489,0.00047754578,0.0006515625,0.0009290048,0.019083276,0.016107172,0.3657468,0.59097433],"study_design_scores_gemma":[0.00008157724,0.00013739803,0.0051273247,0.00011909772,0.000045390632,0.0010331866,0.00034079535,0.008355298,0.022587206,0.0144539215,0.9476385,0.00008017911],"about_ca_topic_score_codex":0.0038845688,"about_ca_topic_score_gemma":0.008540292,"teacher_disagreement_score":0.9185883,"about_ca_system_score_codex":0.0006121869,"about_ca_system_score_gemma":0.0014000816,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W2121274514","doi":"10.3115/1218955.1218992","title":"Unsupervised sense disambiguation using bilingual probabilistic models","year":2004,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":29,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"National Science Foundation","keywords":"Computer science; Probabilistic logic; Natural language processing; Artificial intelligence; Language model; Task (project management); Statistical model; Word (group theory); Word-sense disambiguation; Unsupervised learning; Latent variable; Machine learning; Linguistics","score_opus":0.04102658423590184,"score_gpt":0.2908114587039068,"score_spread":0.24978487446800496,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2121274514","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0087087145,0.00024160609,0.9872132,0.00029403728,0.00004605899,0.000058633075,0.0002750388,0.0011131718,0.0020494962],"genre_scores_gemma":[0.30324152,0.0006738852,0.68697315,0.00049044716,0.00023266909,0.0005170648,0.0025858346,0.00051061163,0.004774724],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9971917,0.0011998735,0.00015335811,0.0007431071,0.0005887031,0.00012327032],"domain_scores_gemma":[0.99555737,0.0025231799,0.0004360474,0.0007807927,0.00054905826,0.0001536758],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031862007,0.0012877561,0.0012217538,0.0026521701,0.0013340114,0.002012401,0.0023796174,0.0013148804,0.0030108627],"category_scores_gemma":[0.009911559,0.0011273533,0.0019377725,0.0031594515,0.0013520406,0.005017878,0.0031408318,0.0022295658,0.001930428],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00045041263,0.0005354575,0.0059345677,0.00044003534,0.0004825372,0.000702617,0.0010792284,0.39127517,0.009430074,0.19591068,0.017430242,0.3763289],"study_design_scores_gemma":[0.000043290813,0.000032626926,0.00056929264,0.00002223631,0.00004197443,0.00020821347,0.00005628376,0.84166795,0.0019885288,0.14990772,0.005416038,0.00004587079],"about_ca_topic_score_codex":0.0039235055,"about_ca_topic_score_gemma":0.008682734,"teacher_disagreement_score":0.0039235055,"about_ca_system_score_codex":0.0010422062,"about_ca_system_score_gemma":0.0022138592,"threshold_uncertainty_score":0.016850412},"labels":[],"label_agreement":null},{"id":"W2121351739","doi":"","title":"Syntax-aware Phrase-based Statistical Machine Translation: System Description","year":2012,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Phrase; Natural language processing; Parsing; Machine translation; Artificial intelligence; Syntax; Word order; Lattice (music); Translation (biology); Word (group theory); Algorithm; Speech recognition; Linguistics","score_opus":0.02629563168076906,"score_gpt":0.26968867744228103,"score_spread":0.24339304576151197,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2121351739","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0054873074,0.0018510922,0.90577686,0.00070408074,0.00015373346,0.0011386105,0.008766005,0.0702151,0.0059072394],"genre_scores_gemma":[0.066425644,0.001595721,0.8961379,0.00064768427,0.0002263058,0.0025088994,0.022305937,0.003603457,0.0065485104],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99900275,0.00016703043,0.00012843519,0.00034842952,0.0002816805,0.00007171937],"domain_scores_gemma":[0.99927133,0.00021850226,0.000051982002,0.00018321203,0.00022907424,0.00004591295],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011768608,0.001257253,0.0017614095,0.0017929595,0.0009074529,0.0035039044,0.0030078886,0.0016253879,0.02093772],"category_scores_gemma":[0.002945083,0.0012566497,0.00097653735,0.0025919469,0.0005537041,0.0024615442,0.0021124713,0.0014435077,0.025459142],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00061280205,0.00043474897,0.0021297713,0.0027920755,0.00033521987,0.0014361097,0.0004198509,0.035333756,0.065283336,0.019379847,0.12803486,0.7438077],"study_design_scores_gemma":[0.00029262746,0.0004937573,0.0026904682,0.0002283584,0.0003749452,0.003795951,0.00024604035,0.6670151,0.0928575,0.030568823,0.20107132,0.00036502155],"about_ca_topic_score_codex":0.0044215657,"about_ca_topic_score_gemma":0.0040350487,"teacher_disagreement_score":0.02093772,"about_ca_system_score_codex":0.001006871,"about_ca_system_score_gemma":0.0020885048,"threshold_uncertainty_score":0.07004362},"labels":[],"label_agreement":null},{"id":"W2121479142","doi":"","title":"Collocation Extraction for Machine Translation","year":2003,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":30,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Collocation (remote sensing); Computer science; Natural language processing; Translation (biology); Artificial intelligence; Machine translation; Embedding; Information extraction; Extraction (chemistry); Machine translation software usability; Rule-based machine translation; Example-based machine translation; Machine learning","score_opus":0.021996887847106135,"score_gpt":0.3105580019753942,"score_spread":0.28856111412828805,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2121479142","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0021796292,0.004861423,0.9554985,0.0013308078,0.00086460746,0.00021055342,0.0014521247,0.009943677,0.023658708],"genre_scores_gemma":[0.08743307,0.0046123685,0.8795481,0.00054111436,0.00068100565,0.0003791152,0.0058217295,0.0019692748,0.01901433],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99672675,0.0014390026,0.0003331258,0.0006448875,0.0007345557,0.00012165257],"domain_scores_gemma":[0.9964881,0.0011480553,0.00021939875,0.0011237689,0.0009500446,0.00007055403],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020580394,0.0012498492,0.0009674121,0.0033078794,0.0018973047,0.00322559,0.001417869,0.0014022914,0.034919385],"category_scores_gemma":[0.008089895,0.00069558737,0.00091327826,0.0044369465,0.0011111763,0.004450906,0.0024987985,0.0016182291,0.02547152],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013568623,0.000042555002,0.000440701,0.0012097656,0.000115912706,0.00063175865,0.0006108735,0.0047801863,0.018066611,0.16312908,0.088321455,0.7225154],"study_design_scores_gemma":[0.00004610462,0.00005893514,0.0010440751,0.00042883813,0.00007677265,0.0013318382,0.0003847067,0.06681004,0.026062118,0.2686687,0.6349491,0.00013869406],"about_ca_topic_score_codex":0.002565479,"about_ca_topic_score_gemma":0.0023258254,"teacher_disagreement_score":0.034919385,"about_ca_system_score_codex":0.0012860616,"about_ca_system_score_gemma":0.0014363787,"threshold_uncertainty_score":0.11681694},"labels":[],"label_agreement":null},{"id":"W2121524912","doi":"10.3115/1654449.1654462","title":"Nukti","year":2005,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Word (group theory); Machine translation; Task (project management); Natural language processing; Artificial intelligence; Translation (biology); Linguistics; Engineering","score_opus":0.009055433428015328,"score_gpt":0.26519073263332404,"score_spread":0.2561352992053087,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2121524912","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011313316,0.006960705,0.06328316,0.004499401,0.005349539,0.00038511134,0.021327538,0.024602296,0.8622788],"genre_scores_gemma":[0.06620523,0.0038542075,0.06574607,0.0015676293,0.00072010746,0.0004938528,0.050942536,0.0055391053,0.8049313],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9988695,0.0001660337,0.0001064885,0.0003130453,0.00039735815,0.00014762036],"domain_scores_gemma":[0.99842983,0.00013823986,0.00009692777,0.00041623935,0.0006452715,0.00027353436],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0012609577,0.0011419681,0.0011293698,0.0019945607,0.0023202454,0.0046247304,0.0013883502,0.0013439155,0.28137732],"category_scores_gemma":[0.0031103343,0.00060084526,0.00054519816,0.0019529075,0.000506236,0.0039306455,0.0037114308,0.0017143706,0.28988495],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007117981,0.00015849815,0.0021486098,0.00077474053,0.000042129646,0.00038103914,0.00047158025,0.0007838252,0.012841196,0.035257757,0.36895248,0.5774764],"study_design_scores_gemma":[0.000035992714,0.000044940592,0.0009562808,0.00009562097,0.000021651089,0.00035077074,0.0000937731,0.0016898082,0.0048727132,0.0063193226,0.98548746,0.00003169683],"about_ca_topic_score_codex":0.0017880984,"about_ca_topic_score_gemma":0.0023858664,"teacher_disagreement_score":0.7186227,"about_ca_system_score_codex":0.0013583836,"about_ca_system_score_gemma":0.0018153103,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W2121784613","doi":"10.7202/009790ar","title":"Translating Computer Abbreviations from English into Spanish: Main Types and Problems","year":2005,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Terminology; Computer science; Natural language processing; Linguistics; Artificial intelligence","score_opus":0.020121691104262398,"score_gpt":0.2523694093530894,"score_spread":0.23224771824882698,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2121784613","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2272289,0.06999566,0.5636665,0.030467598,0.008689875,0.0010631451,0.0047010276,0.0077405456,0.08644676],"genre_scores_gemma":[0.4885993,0.04763673,0.40047202,0.0060813157,0.0048827576,0.0010440401,0.0059928405,0.0084779225,0.03681304],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.98135155,0.008135215,0.0032796194,0.0017452827,0.0048280847,0.0006602454],"domain_scores_gemma":[0.9646658,0.018540895,0.0038588152,0.003447649,0.009044443,0.0004424769],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0059175193,0.0017251637,0.0016344368,0.004928033,0.0021761733,0.0054040877,0.0020334716,0.0017323359,0.0057686255],"category_scores_gemma":[0.040330376,0.0008051283,0.0011027855,0.009178569,0.002802891,0.0068734875,0.0030604226,0.0021506832,0.00491734],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00050868536,0.00013993119,0.013024148,0.0042690197,0.000113318005,0.0031130938,0.016767379,0.0020195819,0.016740717,0.073287055,0.067651354,0.80236584],"study_design_scores_gemma":[0.00012191363,0.0002516076,0.010617591,0.002353139,0.00024815983,0.0120385615,0.02540475,0.010340432,0.045039766,0.07942959,0.8138335,0.00032105707],"about_ca_topic_score_codex":0.0037791529,"about_ca_topic_score_gemma":0.0028398011,"teacher_disagreement_score":0.0059175193,"about_ca_system_score_codex":0.0023476633,"about_ca_system_score_gemma":0.002350974,"threshold_uncertainty_score":0.03129518},"labels":[],"label_agreement":null},{"id":"W2121929533","doi":"10.1017/s135132491300003x","title":"Designing a machine translation system for Canadian weather warnings: A case study","year":2013,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Université de Montréal; Université du Québec à Montréal","funders":"","keywords":"Computer science; Machine translation; Natural language processing; Metric (unit); Task (project management); Quality (philosophy); Domain (mathematical analysis); Evaluation of machine translation; Translation (biology); Artificial intelligence; Machine translation software usability; Presentation (obstetrics); Example-based machine translation; Machine learning; Information retrieval; Systems engineering","score_opus":0.007204560510228801,"score_gpt":0.2309438070816249,"score_spread":0.2237392465713961,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2121929533","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.73459977,0.001326954,0.17345683,0.006377378,0.0007575516,0.003353867,0.0066814437,0.01415146,0.059294738],"genre_scores_gemma":[0.7016811,0.0008062366,0.24925809,0.00061286945,0.00009801871,0.00041512426,0.00822524,0.0014401025,0.03746321],"study_design_codex":"design_other","study_design_gemma":"qualitative","domain_scores_codex":[0.9975339,0.0008231773,0.00016741462,0.00032483868,0.00088698504,0.00026367797],"domain_scores_gemma":[0.9959973,0.0013383925,0.00013370282,0.00029101182,0.0019692103,0.00027037834],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027109513,0.0009796236,0.00055642024,0.0011394726,0.004927296,0.0019204371,0.0011298265,0.0015872229,0.0059156884],"category_scores_gemma":[0.007593164,0.00034825347,0.00053884904,0.002144947,0.0012425245,0.0013677087,0.00081979576,0.0013516163,0.002190729],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012587791,0.0015573917,0.018690666,0.0035643186,0.00016400634,0.017066132,0.03291257,0.064693995,0.15131333,0.01636749,0.0978449,0.5945664],"study_design_scores_gemma":[0.0009629357,0.0012806387,0.042835575,0.00035451984,0.00036597726,0.0056727096,0.024682162,0.1524724,0.23042305,0.0034870063,0.536755,0.00070803793],"about_ca_topic_score_codex":0.521345,"about_ca_topic_score_gemma":0.54722166,"teacher_disagreement_score":0.47865498,"about_ca_system_score_codex":0.008337611,"about_ca_system_score_gemma":0.014768796,"threshold_uncertainty_score":0.9629477},"labels":[],"label_agreement":null},{"id":"W2121963504","doi":"","title":"Morphological acquisition by Formal Analogy","year":2009,"lang":"en","type":"article","venue":"CLEF (Working Notes)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Analogy; Morpheme; Lexicon; Computer science; Artificial intelligence; Relation (database); Natural language processing; Reading (process); Task (project management); Linguistics","score_opus":0.016276388832739,"score_gpt":0.27136484243506614,"score_spread":0.2550884536023271,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2121963504","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025866373,0.00015024055,0.9597486,0.00025364544,0.0000317592,0.0001293112,0.00017666385,0.005715782,0.007927525],"genre_scores_gemma":[0.33950958,0.00024686332,0.6517268,0.00028526608,0.000065237524,0.00027502273,0.0012637762,0.00093929266,0.0056881458],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99652195,0.0009828552,0.00026093275,0.0013316489,0.00076157355,0.000140951],"domain_scores_gemma":[0.9927377,0.0026958538,0.00054804335,0.0029393497,0.0009149378,0.00016412772],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022573187,0.0007871951,0.0011511687,0.0025277091,0.0013643074,0.0025949874,0.0026355004,0.0011332618,0.008210323],"category_scores_gemma":[0.013243145,0.0008990831,0.001780415,0.001731791,0.0027806277,0.010675803,0.007960602,0.0024821162,0.0039141616],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001372648,0.00020845702,0.0040690587,0.00045261288,0.00009251565,0.00031172618,0.0014038567,0.012789138,0.03783691,0.156138,0.005335362,0.7812251],"study_design_scores_gemma":[0.000058785336,0.00023982757,0.0042798375,0.000109398025,0.00006718389,0.0016065477,0.0006309541,0.29476538,0.047573645,0.60884017,0.041677393,0.00015097273],"about_ca_topic_score_codex":0.00063975976,"about_ca_topic_score_gemma":0.0009562119,"teacher_disagreement_score":0.008210323,"about_ca_system_score_codex":0.00093612855,"about_ca_system_score_gemma":0.0011164173,"threshold_uncertainty_score":0.027466297},"labels":[],"label_agreement":null},{"id":"W2122225311","doi":"10.22329/il.v29i1.683","title":"Implicit dialogical premises, explanation as argument: A corpus-based reconstruction","year":2009,"lang":"en","type":"article","venue":"Informal Logic","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Dialogical self; Argument (complex analysis); Premise; Newspaper; Epistemology; Sociology; Linguistics; Philosophy; Media studies","score_opus":0.013400877473480257,"score_gpt":0.2619267581770227,"score_spread":0.24852588070354245,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2122225311","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16444145,0.0013350513,0.74263626,0.005437669,0.00023619458,0.00046445354,0.0013156424,0.0009315992,0.08320173],"genre_scores_gemma":[0.80988663,0.000520736,0.18072201,0.00023741223,0.00007748979,0.00035263604,0.0011411664,0.00035825977,0.0067036315],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"qualitative","domain_scores_codex":[0.998314,0.0010028493,0.00007080219,0.00026823973,0.00025535224,0.00008864363],"domain_scores_gemma":[0.992176,0.0055638854,0.00030914717,0.00135553,0.0005142117,0.0000811793],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033704482,0.00047060856,0.0005032257,0.0034502724,0.0025401008,0.0057897135,0.0019323161,0.0017378333,0.009100718],"category_scores_gemma":[0.014204894,0.000719664,0.0005423966,0.0030486188,0.007445634,0.010140315,0.0035154351,0.0033825934,0.0010882211],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017359706,0.000065879205,0.002473375,0.00028102417,0.000033525386,0.0011142474,0.041081402,0.0026976995,0.0029292551,0.8801031,0.002842865,0.06620401],"study_design_scores_gemma":[0.000098283024,0.000091952184,0.005245657,0.0005663586,0.00009764036,0.0016575413,0.042060755,0.08183695,0.010168279,0.7161026,0.14197859,0.00009541774],"about_ca_topic_score_codex":0.0031527772,"about_ca_topic_score_gemma":0.0041237413,"teacher_disagreement_score":0.009100718,"about_ca_system_score_codex":0.0020723636,"about_ca_system_score_gemma":0.0013583442,"threshold_uncertainty_score":0.03044492},"labels":[],"label_agreement":null},{"id":"W2122373020","doi":"10.2478/v10108-012-0004-y","title":"Kriya - An end-to-end Hierarchical Phrase-based MT System","year":2012,"lang":"en","type":"article","venue":"The Prague Bulletin of Mathematical Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"End-to-end principle; Phrase; Computer science; Artificial intelligence","score_opus":0.01778073816134857,"score_gpt":0.2739424006951001,"score_spread":0.25616166253375156,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2122373020","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.032745056,0.00096808275,0.50465775,0.0004644863,0.00025376302,0.0004412973,0.012997942,0.42768764,0.019783946],"genre_scores_gemma":[0.24629498,0.00049144594,0.6688317,0.00063280266,0.00019121863,0.00038356244,0.039117906,0.013205474,0.03085084],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99932337,0.00012929192,0.0001035742,0.00023441235,0.00015055867,0.000058845137],"domain_scores_gemma":[0.99871373,0.0002966587,0.00009438727,0.00047568863,0.0003369781,0.00008253452],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010116898,0.00074519514,0.0011964228,0.0012623444,0.0007680697,0.0016567949,0.0020616257,0.0012084628,0.020668693],"category_scores_gemma":[0.0027742116,0.00073755434,0.0005834022,0.0008008975,0.00047048862,0.004353434,0.0023893781,0.0011575597,0.025062254],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0026906824,0.00030894738,0.0019806556,0.0013193534,0.00028851352,0.0010942246,0.00069053884,0.0052145314,0.16972002,0.010577099,0.14782436,0.658291],"study_design_scores_gemma":[0.0013082525,0.0010551455,0.008270503,0.00025399667,0.0006186498,0.0040380782,0.0005833475,0.40725216,0.29081747,0.03658488,0.24869242,0.0005250075],"about_ca_topic_score_codex":0.001878032,"about_ca_topic_score_gemma":0.003189022,"teacher_disagreement_score":0.020668693,"about_ca_system_score_codex":0.00037519156,"about_ca_system_score_gemma":0.000715543,"threshold_uncertainty_score":0.06914365},"labels":[],"label_agreement":null},{"id":"W2122497464","doi":"10.3115/1564144.1564150","title":"<i>Yawat</i>","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":54,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Markup language; Visualization; Sentence; Word (group theory); Focus (optics); Phrase; JavaScript; Annotation; Artificial intelligence; Natural language processing; Information retrieval; XML; Programming language; World Wide Web","score_opus":0.014275069383512126,"score_gpt":0.24422248265506086,"score_spread":0.22994741327154872,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2122497464","genre_codex":"software","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.001901741,0.00038286348,0.23702274,0.0008298878,0.0012481804,0.0002687784,0.08763214,0.6119172,0.058796365],"genre_scores_gemma":[0.030179456,0.00071423774,0.33938223,0.0018539695,0.00074157043,0.0011247136,0.22229816,0.29485765,0.10884794],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9994925,0.00007696332,0.000051831717,0.00012014079,0.00019839023,0.0000601216],"domain_scores_gemma":[0.9984993,0.00047320293,0.00013858636,0.0002933837,0.00042419072,0.00017135407],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.00079456304,0.0019711694,0.0010869094,0.0026558847,0.0007091039,0.0025355443,0.0020015885,0.00086498196,0.35097376],"category_scores_gemma":[0.0039062854,0.00094394793,0.0009886671,0.0020554585,0.00041882286,0.0030043027,0.002539783,0.0014745536,0.1769483],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017844659,0.00001936189,0.000357484,0.00046798482,0.000028040447,0.00019326794,0.00020443514,0.00021025224,0.007909753,0.0025824436,0.909107,0.078741446],"study_design_scores_gemma":[0.00008712708,0.000034489858,0.0013528814,0.000121030906,0.000020100802,0.00043527933,0.000071798895,0.0038172298,0.016557548,0.004522174,0.97287446,0.00010596658],"about_ca_topic_score_codex":0.0025064955,"about_ca_topic_score_gemma":0.004290328,"teacher_disagreement_score":0.35097376,"about_ca_system_score_codex":0.00055603584,"about_ca_system_score_gemma":0.00067150337,"threshold_uncertainty_score":0.92575717},"labels":[],"label_agreement":null},{"id":"W2123215530","doi":"10.1162/coli.08-010-r1-07-048","title":"Unsupervised Type and Token Identification of Idiomatic Expressions","year":2009,"lang":"en","type":"article","venue":"Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":201,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canada Research Chairs; University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada; Atomic Energy of Canada Limited; University of Toronto","keywords":"Linguistics; Identification (biology); Interpretation (philosophy); Computer science; Natural language processing; Literal (mathematical logic); Task (project management); Context (archaeology); Security token; Artificial intelligence; Semantic property; Expression (computer science); Psychology; History","score_opus":0.015640231814246133,"score_gpt":0.29692181395150163,"score_spread":0.2812815821372555,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2123215530","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.40255097,0.0005095452,0.5849693,0.00036363796,0.00019201373,0.00030781658,0.0020724307,0.004041636,0.00499272],"genre_scores_gemma":[0.63266,0.00020066522,0.3592974,0.00010093834,0.00009571162,0.0002496105,0.003979844,0.0004924405,0.0029234274],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9975382,0.0006523495,0.00028638315,0.000813653,0.00048783387,0.00022169942],"domain_scores_gemma":[0.99042165,0.0047399914,0.0012976965,0.0013652724,0.0018841925,0.0002911735],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019525915,0.00061215396,0.0009856471,0.0036055981,0.00067212345,0.0016489405,0.0014604228,0.00083563867,0.001613579],"category_scores_gemma":[0.010129006,0.0003166714,0.00071070495,0.002530779,0.0008239504,0.0029319143,0.0010636562,0.0011263314,0.0015348976],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012724007,0.00029305345,0.08236528,0.0005782445,0.00014848409,0.0006804428,0.0013156913,0.0057836697,0.08303012,0.016922254,0.010545213,0.79706526],"study_design_scores_gemma":[0.000086785156,0.00026726333,0.07606187,0.00009126233,0.00020441393,0.002858156,0.001630861,0.7434557,0.11256799,0.049640454,0.012930754,0.00020444687],"about_ca_topic_score_codex":0.001350142,"about_ca_topic_score_gemma":0.0021323499,"teacher_disagreement_score":0.0036055981,"about_ca_system_score_codex":0.00061215134,"about_ca_system_score_gemma":0.0011096236,"threshold_uncertainty_score":0.0103263855},"labels":[],"label_agreement":null},{"id":"W2123269682","doi":"10.1007/978-3-540-78135-6_52","title":"Real-Word Spelling Correction with Trigrams: A Reconsideration of the Mays, Damerau, and Mercer Model","year":2008,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":93,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Trigram; Spelling; WordNet; Computer science; Artificial intelligence; Word (group theory); Natural language processing; Speech recognition; Mathematics; Linguistics","score_opus":0.018039904374920597,"score_gpt":0.24648636253333012,"score_spread":0.22844645815840953,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2123269682","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016252166,0.0038461846,0.95333815,0.0061759804,0.0010972555,0.00008404909,0.0005048952,0.0028671445,0.01583423],"genre_scores_gemma":[0.49155632,0.005345786,0.47374094,0.0025819277,0.0016931688,0.0001216474,0.0008469917,0.0016845164,0.0224287],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99765956,0.00083693827,0.00018607508,0.00041431966,0.000790309,0.00011280579],"domain_scores_gemma":[0.988267,0.0063013965,0.00045330223,0.0030740178,0.0017145438,0.00018975834],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0042994227,0.000922337,0.0018644655,0.0021220471,0.0010147416,0.0036780927,0.005849073,0.001788192,0.0049971985],"category_scores_gemma":[0.02370256,0.00062848954,0.001192318,0.0024540317,0.0019571967,0.008832628,0.002056199,0.0032861196,0.003396007],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00040096522,0.00017202346,0.0037040792,0.0005143068,0.000247911,0.0005553033,0.0011651306,0.024106983,0.0046377573,0.30741858,0.023075668,0.6340013],"study_design_scores_gemma":[0.000059486407,0.00016170915,0.0017751395,0.00021399929,0.00016459188,0.0009932667,0.00035813788,0.46741256,0.007258431,0.50031096,0.021166215,0.00012563428],"about_ca_topic_score_codex":0.0055743903,"about_ca_topic_score_gemma":0.007847086,"teacher_disagreement_score":0.005849073,"about_ca_system_score_codex":0.00080268877,"about_ca_system_score_gemma":0.0021756059,"threshold_uncertainty_score":0.022737741},"labels":[],"label_agreement":null},{"id":"W2123551014","doi":"10.3115/1220835.1220839","title":"Segment choice models","year":2006,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Perplexity; Computer science; Phrase; Distortion (music); Machine translation; Artificial intelligence; Decoding methods; Metric (unit); Probabilistic logic; Sentence; Translation (biology); Natural language processing; Sequence (biology); Language model; Speech recognition; Algorithm","score_opus":0.012764601618364652,"score_gpt":0.2531484317530582,"score_spread":0.24038383013469353,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2123551014","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.033129606,0.0012456141,0.9362863,0.0017522656,0.00028479847,0.00030525142,0.004654159,0.0017085691,0.020633465],"genre_scores_gemma":[0.7295245,0.0014831664,0.19573258,0.0009476426,0.0004203872,0.00097274873,0.010604318,0.00096271327,0.059351936],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9977411,0.00091417646,0.0001151784,0.0006481285,0.00038394524,0.00019747773],"domain_scores_gemma":[0.99402785,0.0040730555,0.00035305863,0.0008024874,0.0005095582,0.00023408925],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034434376,0.0012342775,0.0012580083,0.0015274588,0.00071764766,0.0019093488,0.003597176,0.0020022353,0.025662575],"category_scores_gemma":[0.01201911,0.00073749956,0.001810411,0.0024034611,0.0012624051,0.0047540967,0.00199629,0.0025395309,0.006459724],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012798146,0.00023666222,0.007970359,0.00035449344,0.00028954295,0.00044246655,0.00063059985,0.33387077,0.002553703,0.4528959,0.026738903,0.17273682],"study_design_scores_gemma":[0.000053677257,0.000075866905,0.0006916524,0.00002752086,0.00005949531,0.00014738663,0.000052886917,0.7812365,0.00085284765,0.20374233,0.013024674,0.000035150006],"about_ca_topic_score_codex":0.00565215,"about_ca_topic_score_gemma":0.0069720773,"teacher_disagreement_score":0.025662575,"about_ca_system_score_codex":0.0017449278,"about_ca_system_score_gemma":0.001246816,"threshold_uncertainty_score":0.08584988},"labels":[],"label_agreement":null},{"id":"W2124136969","doi":"10.3115/1075096.1075108","title":"A probability model to improve word alignment","year":2003,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":91,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Word (group theory); Natural language processing; Machine translation; Artificial intelligence; Translation (biology); Sentence; Context (archaeology); Statistical model; Probability model; Linguistics; Mathematics","score_opus":0.02319660031581302,"score_gpt":0.2710961346077686,"score_spread":0.24789953429195558,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2124136969","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00462114,0.0002922539,0.9913284,0.00035994052,0.00013048871,0.000047057765,0.00015049352,0.0015446012,0.0015257269],"genre_scores_gemma":[0.2511634,0.0011416564,0.73016214,0.0007171181,0.00068009016,0.0006567749,0.0022286498,0.0018505518,0.011399657],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9971711,0.0009493232,0.00018845178,0.0007134041,0.0008083565,0.0001694224],"domain_scores_gemma":[0.99109393,0.005729805,0.00042959323,0.0012581084,0.001296344,0.000192286],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030817797,0.0013110708,0.0018000717,0.0017803583,0.0010857716,0.001969774,0.002646386,0.0020059978,0.007853429],"category_scores_gemma":[0.021017725,0.0010449955,0.0019144007,0.0029827368,0.0010492787,0.007803707,0.0026411517,0.0035898378,0.0048967428],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003440134,0.000289205,0.0018144344,0.0002549668,0.00022072053,0.00028076163,0.0002613133,0.590591,0.007540181,0.12078838,0.015765004,0.26185],"study_design_scores_gemma":[0.000019671783,0.000035484492,0.00012488135,0.000007903704,0.00002474117,0.00004707051,0.000006684197,0.9548926,0.0011069009,0.04134581,0.00237155,0.000016671376],"about_ca_topic_score_codex":0.0039783968,"about_ca_topic_score_gemma":0.004496162,"teacher_disagreement_score":0.007853429,"about_ca_system_score_codex":0.0012522354,"about_ca_system_score_gemma":0.0018551367,"threshold_uncertainty_score":0.026272297},"labels":[],"label_agreement":null},{"id":"W2124258519","doi":"10.7202/602538ar","title":"Syntactic Analysis and Semantic Processing","year":2009,"lang":"fr","type":"article","venue":"Revue québécoise de linguistique","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Philosophy","score_opus":0.012384159046657798,"score_gpt":0.2918935166915102,"score_spread":0.2795093576448524,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2124258519","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023925405,0.00813027,0.8528337,0.010429353,0.0005908352,0.00019250378,0.0015645417,0.0029017972,0.09943164],"genre_scores_gemma":[0.48159695,0.0084757,0.46431887,0.0023130488,0.00093312317,0.00027962797,0.0036646836,0.0017684826,0.036649518],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9970975,0.0013066293,0.00020472222,0.00042329097,0.0007886106,0.00017921257],"domain_scores_gemma":[0.9964373,0.0018347054,0.00025360024,0.0006304581,0.0007748114,0.000069155925],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030289434,0.00093845703,0.00077330973,0.0056771575,0.0019618287,0.007880513,0.0012060957,0.0013823846,0.012316623],"category_scores_gemma":[0.006606393,0.0006482315,0.0018144767,0.004685975,0.0054122135,0.008564972,0.0023633263,0.002324363,0.0034330983],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000085437016,0.00004151168,0.001825388,0.0005023678,0.00010480449,0.00041829422,0.0026329695,0.0031343733,0.0092139905,0.73930424,0.011567948,0.23116861],"study_design_scores_gemma":[0.000014179787,0.000027645108,0.0026399028,0.00027203312,0.000081164806,0.0004350839,0.001434474,0.022125995,0.007234241,0.8267336,0.1389415,0.000060156057],"about_ca_topic_score_codex":0.009881838,"about_ca_topic_score_gemma":0.0097550405,"teacher_disagreement_score":0.012316623,"about_ca_system_score_codex":0.0038791075,"about_ca_system_score_gemma":0.0034087414,"threshold_uncertainty_score":0.0412032},"labels":[],"label_agreement":null},{"id":"W2124474181","doi":"10.1016/j.system.2004.04.001","title":"Can learners use concordance feedback for writing errors?","year":2004,"lang":"en","type":"article","venue":"System","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":346,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Concordance; Notice; Sentence; Computer science; Natural language processing; Linguistics; Mathematics education; Psychology","score_opus":0.02152784306383158,"score_gpt":0.2726472343495525,"score_spread":0.2511193912857209,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2124474181","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.73192406,0.00081349077,0.16767216,0.010003928,0.0014154846,0.00079977064,0.0012236431,0.025545787,0.06060167],"genre_scores_gemma":[0.9036042,0.00037142323,0.079836525,0.0009396658,0.00020908543,0.00020144535,0.00065299426,0.0009916944,0.013193017],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9854864,0.0065360023,0.0011946883,0.0016677601,0.0046812757,0.00043386588],"domain_scores_gemma":[0.87089956,0.083014235,0.007920008,0.013012165,0.022485686,0.0026683894],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00978669,0.0009476082,0.0009456281,0.0014246772,0.0009303218,0.0038432667,0.0019195983,0.0029167442,0.013462891],"category_scores_gemma":[0.16451117,0.00048438853,0.0003884422,0.00092615624,0.0006040014,0.00602962,0.002017555,0.0016644781,0.0068761534],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014393202,0.0014537768,0.058979265,0.00097615784,0.00009016829,0.0016761889,0.014022604,0.0014955667,0.03272572,0.0033571897,0.03428494,0.84949917],"study_design_scores_gemma":[0.0016943156,0.0072336765,0.1206443,0.0028775411,0.0011101954,0.015476505,0.041757017,0.18156882,0.3709917,0.055920847,0.19967435,0.001050706],"about_ca_topic_score_codex":0.0011255227,"about_ca_topic_score_gemma":0.0016934945,"teacher_disagreement_score":0.013462891,"about_ca_system_score_codex":0.00043771646,"about_ca_system_score_gemma":0.0018808891,"threshold_uncertainty_score":0.051757634},"labels":[],"label_agreement":null},{"id":"W2124618303","doi":"","title":"Substring-Based Transliteration","year":2007,"lang":"en","type":"article","venue":"Meeting of the Association for Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":61,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Transliteration; Substring; Computer science; Margin (machine learning); Artificial intelligence; Word (group theory); Natural language processing; Machine translation; Speech recognition; Programming language; Machine learning; Data structure; Mathematics","score_opus":0.011120757289096535,"score_gpt":0.27550062212974663,"score_spread":0.26437986484065007,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2124618303","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00989526,0.00021458624,0.97913456,0.00019275234,0.00011342203,0.000068893736,0.0002115368,0.0040569734,0.0061119725],"genre_scores_gemma":[0.29259974,0.00048006402,0.6900888,0.00031042102,0.000096452626,0.00016968451,0.001151221,0.001358785,0.013744816],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99867404,0.0003185599,0.00011551832,0.00042234,0.00037779397,0.00009185897],"domain_scores_gemma":[0.9973017,0.0011431545,0.00015328554,0.00077299005,0.0005809411,0.000048051068],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008406619,0.0005936558,0.00077990995,0.00059186574,0.00056406256,0.0012589344,0.0015133035,0.00094375457,0.009057875],"category_scores_gemma":[0.004397772,0.0003364872,0.0006686663,0.0009635891,0.0008478089,0.0023829797,0.0012297847,0.0014868461,0.0061921887],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029293157,0.00018093486,0.0011563381,0.00082545285,0.00010602876,0.0004762609,0.00080566603,0.054993704,0.19509092,0.13800567,0.014618146,0.593448],"study_design_scores_gemma":[0.00005306216,0.0002872222,0.00055128866,0.000068978385,0.00009866714,0.0008506349,0.00017494347,0.50192326,0.30414367,0.100856446,0.09090339,0.000088565765],"about_ca_topic_score_codex":0.0007096136,"about_ca_topic_score_gemma":0.0008742723,"teacher_disagreement_score":0.009057875,"about_ca_system_score_codex":0.00053664966,"about_ca_system_score_gemma":0.0010239854,"threshold_uncertainty_score":0.03030163},"labels":[],"label_agreement":null},{"id":"W2124794726","doi":"10.18653/v1/w14-0141","title":"Terminology in WordNet and in plWordNet","year":2014,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"WordNet; Denotation (semiotics); Terminology; Computer science; Natural language processing; Artificial intelligence; Lexical database; Connotation; Term (time); Linguistics; Information retrieval","score_opus":0.00814702927022443,"score_gpt":0.2529311196684496,"score_spread":0.24478409039822518,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2124794726","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010153584,0.004018155,0.88485587,0.0049034175,0.0025868376,0.00028148075,0.0020210121,0.002126927,0.089052714],"genre_scores_gemma":[0.16924848,0.0057784733,0.78423804,0.002476758,0.0015183947,0.0014063206,0.004773703,0.0020620106,0.028497767],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9962813,0.00164907,0.00059946475,0.00071321183,0.0005144476,0.00024245502],"domain_scores_gemma":[0.9959947,0.0014985244,0.00061378913,0.0008739699,0.0007999519,0.0002190888],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037852617,0.0012219952,0.0009395875,0.0063992264,0.0027405065,0.008443914,0.0018927491,0.0017910124,0.01573592],"category_scores_gemma":[0.010038134,0.0010664469,0.00082115585,0.008643104,0.00591457,0.02750895,0.004032698,0.003832303,0.008547668],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000029562516,0.000010773184,0.00025053357,0.00018779834,0.000010520973,0.000111986796,0.0008557081,0.00066072965,0.0005221443,0.9622111,0.0076834774,0.027465718],"study_design_scores_gemma":[0.000009818979,0.000022247596,0.00028219042,0.00019517048,0.000018095208,0.00045638203,0.0007276964,0.0043567554,0.0011944079,0.69187856,0.3008252,0.000033460372],"about_ca_topic_score_codex":0.001572564,"about_ca_topic_score_gemma":0.0022877934,"teacher_disagreement_score":0.01573592,"about_ca_system_score_codex":0.0019225708,"about_ca_system_score_gemma":0.0016949291,"threshold_uncertainty_score":0.05264193},"labels":[],"label_agreement":null},{"id":"W2125268123","doi":"","title":"From the Definitions of the \"Trésor de la Langue Française\" To a Semantic Database of the French Language","year":2010,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Lexicon; Lexical database; Semantics (computer science); Natural language processing; XML; Artificial intelligence; Meaning (existential); Information retrieval; WordNet; World Wide Web; Programming language","score_opus":0.015547312105096073,"score_gpt":0.2775662658126556,"score_spread":0.2620189537075595,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2125268123","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022648066,0.019156976,0.7510746,0.020067973,0.0030148302,0.0003357599,0.009670881,0.0020827267,0.17194825],"genre_scores_gemma":[0.41499478,0.013799629,0.48962563,0.0061681955,0.0017189711,0.00094401155,0.016816089,0.0014752487,0.054457422],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.996416,0.0016150775,0.00039372905,0.00072782254,0.00052017707,0.00032716422],"domain_scores_gemma":[0.9965934,0.001117133,0.00032946328,0.0005458561,0.0011832541,0.00023081942],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00391922,0.0010773953,0.00076764455,0.0062953467,0.0035926874,0.008761996,0.002081227,0.0024619387,0.011629469],"category_scores_gemma":[0.0064017517,0.0006212661,0.0013907023,0.0067632073,0.007413852,0.012464606,0.002774048,0.0033707444,0.003214259],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000032077834,0.000006112699,0.0002387343,0.00011247451,0.000013434587,0.00015097849,0.0018725588,0.00034465565,0.00049613375,0.9693276,0.010695243,0.016709868],"study_design_scores_gemma":[0.00001653435,0.000029501653,0.0006898547,0.00031367963,0.000035116802,0.0007164767,0.001699544,0.0021570555,0.0010645638,0.24955878,0.7436547,0.000064259504],"about_ca_topic_score_codex":0.059064075,"about_ca_topic_score_gemma":0.040997997,"teacher_disagreement_score":0.059064075,"about_ca_system_score_codex":0.0066569513,"about_ca_system_score_gemma":0.006195097,"threshold_uncertainty_score":0.11744058},"labels":[],"label_agreement":null},{"id":"W2125913977","doi":"10.7202/004143ar","title":"L'interjection dans la BD : réflexions sur sa traduction","year":2002,"lang":"fr","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Philosophy; Art","score_opus":0.051033829143496356,"score_gpt":0.2777521262095514,"score_spread":0.22671829706605506,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2125913977","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2934166,0.006353894,0.39701676,0.007893101,0.0021441048,0.00022042362,0.002060572,0.002490939,0.28840366],"genre_scores_gemma":[0.91054523,0.002221451,0.046842694,0.0006061446,0.0003677102,0.00012097576,0.0010667876,0.0013816189,0.036847346],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99485344,0.0026684916,0.00037792057,0.0007430007,0.0011398246,0.00021729509],"domain_scores_gemma":[0.9888888,0.007244019,0.00053725654,0.0016628923,0.00152014,0.00014697431],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004453493,0.0012373397,0.0006112375,0.0019915274,0.0034239308,0.00596478,0.0008880757,0.0013415797,0.011686624],"category_scores_gemma":[0.014523617,0.0007260438,0.00055081106,0.0022293727,0.0071698627,0.008852985,0.0038040187,0.0043592686,0.0028309368],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00048927823,0.00006286319,0.005511314,0.0010376385,0.000051929517,0.0013813005,0.1223914,0.0008797082,0.023123361,0.6572006,0.014167312,0.1737033],"study_design_scores_gemma":[0.000068992864,0.00015740599,0.00874206,0.00088389986,0.000100330624,0.0058033112,0.056440726,0.0066638067,0.032702073,0.101149015,0.7871025,0.00018583305],"about_ca_topic_score_codex":0.0072069177,"about_ca_topic_score_gemma":0.006838251,"teacher_disagreement_score":0.011686624,"about_ca_system_score_codex":0.0023506158,"about_ca_system_score_gemma":0.0017867346,"threshold_uncertainty_score":0.03909564},"labels":[],"label_agreement":null},{"id":"W2126351286","doi":"","title":"Automatic Generation of English Respellings","year":2013,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Spelling; Pronunciation; Task (project management); Natural language processing; Word (group theory); Artificial intelligence; Representation (politics); Quality (philosophy); Speech recognition; Linguistics; Engineering","score_opus":0.017077393558384082,"score_gpt":0.2504887760205293,"score_spread":0.23341138246214524,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2126351286","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.32426813,0.0012875508,0.5519864,0.0005688647,0.0012508294,0.0012851035,0.011208428,0.074909545,0.033235125],"genre_scores_gemma":[0.52685726,0.00042474145,0.43238539,0.00021459894,0.00022323657,0.0003927276,0.021114772,0.0038481087,0.014539199],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9982857,0.0005363223,0.0001860465,0.00051050395,0.00040191025,0.00007954951],"domain_scores_gemma":[0.99244165,0.0041091153,0.00041255815,0.0009497732,0.0018855209,0.00020141521],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00089899485,0.0015124321,0.00097030133,0.001141052,0.0005364561,0.0013374276,0.0013859075,0.000792191,0.013484727],"category_scores_gemma":[0.010506716,0.00044357456,0.0005783987,0.0007227589,0.00034612144,0.0010177187,0.0013198283,0.000722396,0.006102347],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00097220234,0.0003171143,0.0031807977,0.0019391354,0.0001138021,0.0024249912,0.001046637,0.007824347,0.112727985,0.003334463,0.04437794,0.82174057],"study_design_scores_gemma":[0.0009019873,0.0011858469,0.018847378,0.0004071203,0.0003333015,0.009095705,0.0025724086,0.38267457,0.38043135,0.012722207,0.19037955,0.00044867428],"about_ca_topic_score_codex":0.0009670367,"about_ca_topic_score_gemma":0.0014793602,"teacher_disagreement_score":0.013484727,"about_ca_system_score_codex":0.0002668081,"about_ca_system_score_gemma":0.0005465348,"threshold_uncertainty_score":0.04511088},"labels":[],"label_agreement":null},{"id":"W2126784652","doi":"","title":"Bayesian Extraction of Minimal SCFG Rules for Hierarchical Phrase-based Translation","year":2011,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Synchronous context-free grammar; Artificial intelligence; Natural language processing; Machine translation; Phrase; Machine learning; Example-based machine translation","score_opus":0.04062912719782167,"score_gpt":0.29837046445499366,"score_spread":0.25774133725717197,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2126784652","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0059223347,0.00007581813,0.992324,0.000061230574,0.000011796825,0.00005274476,0.00021392311,0.00065009826,0.00068811164],"genre_scores_gemma":[0.16134675,0.00019177367,0.83463454,0.0001642522,0.000050768755,0.00034195653,0.0017619176,0.00054788624,0.00096022413],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.997198,0.0010687249,0.00023236808,0.00045895448,0.0009178678,0.00012409828],"domain_scores_gemma":[0.9952219,0.0026549145,0.00046526105,0.0007994974,0.00076076575,0.00009764072],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022498672,0.0008838018,0.0011742483,0.0020034572,0.00083279377,0.0013915024,0.0012990686,0.0012392381,0.0029841794],"category_scores_gemma":[0.010588353,0.0008938506,0.0012855883,0.0016235001,0.0010907006,0.0018979611,0.0016363661,0.0015071945,0.002512135],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028635768,0.00032887602,0.0036860236,0.00085713,0.00022907129,0.0006884711,0.00074019417,0.1884571,0.07386753,0.13589253,0.0086869085,0.58627975],"study_design_scores_gemma":[0.000038188882,0.000080657184,0.00076608476,0.00007201355,0.000052083742,0.00026532414,0.000058415088,0.82978845,0.01999634,0.14365107,0.005176384,0.000054965352],"about_ca_topic_score_codex":0.0015486813,"about_ca_topic_score_gemma":0.004356945,"teacher_disagreement_score":0.0029841794,"about_ca_system_score_codex":0.00066087843,"about_ca_system_score_gemma":0.0023171108,"threshold_uncertainty_score":0.011898577},"labels":[],"label_agreement":null},{"id":"W2126795320","doi":"10.3115/1706543.1706559","title":"An expectation maximization approach to pronoun resolution","year":2005,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Pronoun; Artificial intelligence; Probabilistic logic; Maximization; Context (archaeology); Expectation–maximization algorithm; Natural language processing; Resolution (logic); Feature (linguistics); Task (project management); Machine learning; Mathematics; Maximum likelihood; Linguistics; Statistics","score_opus":0.013243323667606612,"score_gpt":0.2703732971784499,"score_spread":0.25712997351084327,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2126795320","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0012600811,0.00014990906,0.99682295,0.0002721971,0.00003201986,0.00002829723,0.000047410547,0.000410501,0.000976626],"genre_scores_gemma":[0.10851396,0.0004719761,0.8837582,0.00070407306,0.00026552175,0.00022552664,0.00047701068,0.00023392511,0.005349817],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9970969,0.0014776179,0.00014587611,0.00052513025,0.0006197787,0.0001347178],"domain_scores_gemma":[0.9966281,0.0021953802,0.000216616,0.00032786804,0.00053843803,0.00009354732],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0038204906,0.0008786199,0.0012748714,0.001077499,0.00074068276,0.0012365922,0.0034045486,0.0014862905,0.003856785],"category_scores_gemma":[0.0089593185,0.00061750604,0.0010052046,0.0016322386,0.0010190211,0.0020904606,0.0014221307,0.0022546204,0.0020175695],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019862605,0.00042188235,0.0017730041,0.0004063883,0.0004138609,0.0004281853,0.0005289027,0.2586692,0.01476703,0.11430665,0.021479834,0.58660644],"study_design_scores_gemma":[0.000018136667,0.00003865331,0.00026022265,0.000013498724,0.00002403832,0.000121777935,0.00002796227,0.90778196,0.0035620972,0.08261165,0.00551116,0.000028854512],"about_ca_topic_score_codex":0.0018095254,"about_ca_topic_score_gemma":0.0033205322,"teacher_disagreement_score":0.003856785,"about_ca_system_score_codex":0.0008304778,"about_ca_system_score_gemma":0.0013499813,"threshold_uncertainty_score":0.020204961},"labels":[],"label_agreement":null},{"id":"W2126914874","doi":"10.7202/004127ar","title":"Simplifying the Complexity of Machine Translation","year":2002,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Formalism (music); Computer science; Machine translation; Simplicity; Dynamic and formal equivalence; Translation (biology); Rule-based machine translation; Artificial intelligence; Theoretical computer science; Epistemology","score_opus":0.11743525485587651,"score_gpt":0.30103809248714786,"score_spread":0.18360283763127133,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2126914874","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021701168,0.0015338756,0.95345885,0.004734576,0.00020169865,0.00013216045,0.00022948449,0.0015394421,0.016468767],"genre_scores_gemma":[0.3861721,0.002925993,0.59633535,0.0009995325,0.00036985017,0.00037289495,0.00075345207,0.00091087876,0.011159962],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9946144,0.0027578974,0.00038608102,0.0005462689,0.0014618941,0.00023343017],"domain_scores_gemma":[0.9888103,0.0070053944,0.00039459756,0.0029707565,0.000668613,0.00015027942],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0047196345,0.001034655,0.0012533051,0.0016808185,0.002377703,0.0066579483,0.002163135,0.0023668343,0.0068134954],"category_scores_gemma":[0.016308397,0.0012385106,0.002625125,0.0018999189,0.0041050236,0.014732566,0.0055868803,0.0040298053,0.0026532123],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016975959,0.000045834106,0.0006593383,0.00044297724,0.000075615826,0.000510519,0.0014285366,0.052845612,0.0055415365,0.8777482,0.005265602,0.055266485],"study_design_scores_gemma":[0.000033331362,0.00005923106,0.00018607716,0.00010214133,0.000046363566,0.0004848614,0.00023147764,0.14770277,0.0029959131,0.8048246,0.04327861,0.000054544562],"about_ca_topic_score_codex":0.003910891,"about_ca_topic_score_gemma":0.004128281,"teacher_disagreement_score":0.0068134954,"about_ca_system_score_codex":0.0024294264,"about_ca_system_score_gemma":0.0027524908,"threshold_uncertainty_score":0.0249601},"labels":[],"label_agreement":null},{"id":"W2126936149","doi":"","title":"Letter-Phoneme Alignment: An Exploration","year":2010,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":29,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Integer programming; Phonetics; Integer (computer science); State (computer science); Algorithm; Artificial intelligence; Linear programming; Programming language","score_opus":0.020938569112589525,"score_gpt":0.27964481194644836,"score_spread":0.25870624283385885,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2126936149","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.055442438,0.0035722326,0.9315644,0.0007527407,0.00007451255,0.000058576054,0.00052914035,0.0024765297,0.0055294377],"genre_scores_gemma":[0.44129804,0.0017818211,0.5502418,0.0002336317,0.00014556898,0.00015350374,0.0021140862,0.0013453901,0.002686091],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99587077,0.0021990174,0.00025640972,0.0009865313,0.0005617578,0.00012550862],"domain_scores_gemma":[0.9906827,0.006210084,0.00037771187,0.0017417747,0.0008161281,0.00017149402],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005786207,0.0015443894,0.0015914539,0.0031451061,0.001244988,0.004183116,0.002049017,0.0015996959,0.0042838044],"category_scores_gemma":[0.0187714,0.0007331647,0.0009667644,0.0048113326,0.0010236474,0.0064822263,0.0029789885,0.0018581685,0.002777056],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00053179474,0.00027510955,0.011696544,0.00064625446,0.00030887767,0.00032447654,0.00072055537,0.0700955,0.018066613,0.036744885,0.0067540454,0.85383534],"study_design_scores_gemma":[0.00004963333,0.0001702506,0.0059568896,0.00013451616,0.00009660077,0.00081119913,0.0005687709,0.84320533,0.022107964,0.10660315,0.020211572,0.00008409037],"about_ca_topic_score_codex":0.0014319897,"about_ca_topic_score_gemma":0.0030034904,"teacher_disagreement_score":0.005786207,"about_ca_system_score_codex":0.0006145381,"about_ca_system_score_gemma":0.0014407426,"threshold_uncertainty_score":0.030600786},"labels":[],"label_agreement":null},{"id":"W2127566020","doi":"","title":"Cross-lingual propagation for morphological analysis","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Morpheme; Task (project management); Hebrew; Bootstrapping (finance); Language model; Word (group theory); Segmentation; Projection (relational algebra); Linguistics","score_opus":0.03134928163952207,"score_gpt":0.3299259960160213,"score_spread":0.29857671437649924,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2127566020","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0052816556,0.00024212553,0.989145,0.00013253992,0.00003545574,0.000046623027,0.00018092146,0.0037992143,0.0011365546],"genre_scores_gemma":[0.21893409,0.00066841324,0.7691502,0.00019129571,0.00011867947,0.00035127366,0.0019384897,0.0010649378,0.0075826286],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9987619,0.00041286222,0.00006621335,0.0003453307,0.00034136497,0.000072393705],"domain_scores_gemma":[0.99646616,0.001942253,0.00023418214,0.00068879663,0.0005914694,0.00007718663],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033152623,0.0009469525,0.0007106585,0.0020261921,0.0009832976,0.0012262664,0.002151169,0.0012869071,0.0064636027],"category_scores_gemma":[0.0074549927,0.0009588745,0.000984399,0.0029025092,0.0011181696,0.003219231,0.0028555335,0.0020810291,0.0035446163],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024896977,0.00021742344,0.0018783675,0.00023006271,0.00026036723,0.00028932068,0.00028605445,0.16190208,0.013317854,0.035540026,0.009426425,0.776403],"study_design_scores_gemma":[0.000015681399,0.000020337295,0.00051948184,0.000012611911,0.000022948325,0.000089879235,0.000026400288,0.95144856,0.0048841806,0.03881638,0.0041243983,0.000019124169],"about_ca_topic_score_codex":0.006711199,"about_ca_topic_score_gemma":0.015165906,"teacher_disagreement_score":0.006711199,"about_ca_system_score_codex":0.00085123174,"about_ca_system_score_gemma":0.0015267971,"threshold_uncertainty_score":0.021622956},"labels":[],"label_agreement":null},{"id":"W2127990768","doi":"10.1109/imcsit.2008.4747222","title":"Sense-based clustering of Polish nouns in the extraction of semantic relatedness","year":2008,"lang":"en","type":"article","venue":"Proceedings of the International Multiconference on Computer Science and Information Technology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"WordNet; Computer science; Noun; Cluster analysis; Natural language processing; Artificial intelligence; Resource (disambiguation); Semantic similarity; Lexical database; Software; Information retrieval; Programming language","score_opus":0.013346336159328252,"score_gpt":0.2522200487354629,"score_spread":0.23887371257613463,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2127990768","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.26613948,0.001528724,0.7044731,0.00036126655,0.00021291303,0.00075391826,0.0044873673,0.0029226518,0.019120581],"genre_scores_gemma":[0.48734742,0.0009758101,0.49275756,0.000096726704,0.00007972807,0.00082061655,0.010917774,0.0008406215,0.0061636996],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99638474,0.0010567082,0.0006580586,0.0010677642,0.0006788188,0.00015396172],"domain_scores_gemma":[0.9967205,0.0010647344,0.0003496506,0.0006032258,0.001118045,0.00014379926],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025079455,0.0009303432,0.0009553901,0.011840946,0.0018223015,0.0027019691,0.0007869303,0.00083874,0.003970146],"category_scores_gemma":[0.011137657,0.0006369611,0.0010038887,0.00898109,0.0013205531,0.0060453126,0.0027905009,0.0006958459,0.003526512],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013559116,0.00035140582,0.030786663,0.0030556812,0.00045683066,0.0020411073,0.012781898,0.014593602,0.106938556,0.106279366,0.01879381,0.7025652],"study_design_scores_gemma":[0.0003200253,0.0007307205,0.078066066,0.0010228872,0.0010137585,0.007094587,0.021135654,0.25717518,0.14124747,0.25595826,0.23571771,0.00051771005],"about_ca_topic_score_codex":0.0023614077,"about_ca_topic_score_gemma":0.0045782086,"teacher_disagreement_score":0.011840946,"about_ca_system_score_codex":0.00060273026,"about_ca_system_score_gemma":0.0016303648,"threshold_uncertainty_score":0.013281465},"labels":[],"label_agreement":null},{"id":"W2128255315","doi":"10.7202/602620ar","title":"Génération automatique de rapports boursiers français et anglais","year":2009,"lang":"fr","type":"article","venue":"Revue québécoise de linguistique","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Humanities; Philosophy; Art","score_opus":0.011692881114277305,"score_gpt":0.2926784905371099,"score_spread":0.2809856094228326,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2128255315","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3411788,0.001918769,0.5377284,0.0023861164,0.0008770074,0.0012948986,0.010929397,0.04380349,0.059883043],"genre_scores_gemma":[0.46142808,0.00084849005,0.46982542,0.0002629111,0.00015535355,0.0005788851,0.017757501,0.0032553226,0.045888104],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9977182,0.0008141316,0.00019652126,0.00069639354,0.0004656918,0.00010913777],"domain_scores_gemma":[0.9899597,0.0053593316,0.00065763737,0.0011961657,0.0026058292,0.00022124665],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028239756,0.0011073289,0.000748101,0.0025147146,0.0014395673,0.002351238,0.00072083704,0.0008137213,0.011145952],"category_scores_gemma":[0.013418934,0.00061558513,0.0010322379,0.0014699453,0.00078919565,0.0021969425,0.0016152351,0.00096440053,0.0055183866],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014262955,0.00026480271,0.023679378,0.0014002468,0.00026444686,0.0014792021,0.034005396,0.008077424,0.074190564,0.022354912,0.045224793,0.7876325],"study_design_scores_gemma":[0.0004058162,0.00059490907,0.041234117,0.00087788264,0.00074478827,0.0026765023,0.02057734,0.1365264,0.13023444,0.017756196,0.6477808,0.0005908793],"about_ca_topic_score_codex":0.030858815,"about_ca_topic_score_gemma":0.033458147,"teacher_disagreement_score":0.030858815,"about_ca_system_score_codex":0.0013957303,"about_ca_system_score_gemma":0.0024230105,"threshold_uncertainty_score":0.061358392},"labels":[],"label_agreement":null},{"id":"W2128826758","doi":"10.1109/icdar.2009.78","title":"Handling Out-of-Vocabulary Words and Recognition Errors Based on Word Linguistic Context for Handwritten Sentence Recognition","year":2009,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure","funders":"","keywords":"Computer science; Speech recognition; Word error rate; Artificial intelligence; Natural language processing; Vocabulary; Focus (optics); Word (group theory); Sentence; Classifier (UML); Language model; Linguistics","score_opus":0.04528160092575161,"score_gpt":0.29893052694745137,"score_spread":0.25364892602169975,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2128826758","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19250602,0.0018286207,0.7997931,0.00023413186,0.00015956617,0.00013547373,0.0002164484,0.004270974,0.0008556546],"genre_scores_gemma":[0.6112373,0.0007753894,0.38442653,0.00014047994,0.00014241647,0.00009708653,0.0005986204,0.0004309438,0.002151243],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986601,0.00035373302,0.00012617155,0.00032993907,0.0004441988,0.00008591631],"domain_scores_gemma":[0.99353254,0.0031627344,0.0012394609,0.0010922422,0.0008332353,0.00013974274],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00085674477,0.0013019579,0.0013517998,0.0010093657,0.00035259675,0.0010577545,0.0011006744,0.00084258534,0.0016487122],"category_scores_gemma":[0.0068662125,0.00039511482,0.0005337877,0.00071554026,0.00051306613,0.002110753,0.0007903297,0.0009101372,0.0011439965],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006932523,0.00023036735,0.0043840352,0.0006307724,0.00011281903,0.0008096991,0.00041046116,0.022960452,0.20938122,0.0010747547,0.0013576789,0.7579545],"study_design_scores_gemma":[0.00005344686,0.00077330647,0.009549556,0.00012331497,0.00025385947,0.0021575391,0.00034233302,0.7041557,0.27215546,0.005577827,0.004737078,0.00012063058],"about_ca_topic_score_codex":0.0009296205,"about_ca_topic_score_gemma":0.001770952,"teacher_disagreement_score":0.0016487122,"about_ca_system_score_codex":0.00024523406,"about_ca_system_score_gemma":0.00065955333,"threshold_uncertainty_score":0.005515516},"labels":[],"label_agreement":null},{"id":"W2128952623","doi":"10.7202/004059ar","title":"Eurotra: the Philosophy Behind it","year":2002,"lang":"fr","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Philosophy; Political science; Sociology","score_opus":0.09870606123358246,"score_gpt":0.2906411182233662,"score_spread":0.19193505698978375,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2128952623","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00399825,0.028855007,0.24759476,0.08089862,0.0036521927,0.0001613865,0.0002750644,0.0015857307,0.632979],"genre_scores_gemma":[0.31611922,0.05026233,0.1427009,0.022216216,0.006849711,0.00093448185,0.00065933744,0.0017985444,0.45845923],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.992767,0.003274386,0.0004250589,0.00096540153,0.0019617886,0.0006062871],"domain_scores_gemma":[0.99471176,0.0020702237,0.00037238083,0.0012641553,0.0009956516,0.0005858479],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009383765,0.00069406437,0.0004948194,0.0024517213,0.0028567472,0.017562741,0.00170225,0.0048475,0.019472737],"category_scores_gemma":[0.009269957,0.00064009975,0.00080955104,0.0033889343,0.01240399,0.028278215,0.006081703,0.006116334,0.0072614537],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000010573414,0.000006279015,0.000056870667,0.00005222624,0.000004992632,0.00001741942,0.00036165008,0.00015191495,0.00010457833,0.9727956,0.010160609,0.016277365],"study_design_scores_gemma":[0.000016230244,0.000017851362,0.00015871767,0.00020718582,0.0000074291743,0.000107232256,0.00039230142,0.00075071264,0.00028065956,0.49401557,0.5040324,0.000013675137],"about_ca_topic_score_codex":0.0031351799,"about_ca_topic_score_gemma":0.0017201009,"teacher_disagreement_score":0.019472737,"about_ca_system_score_codex":0.006179399,"about_ca_system_score_gemma":0.0062419185,"threshold_uncertainty_score":0.06514281},"labels":[],"label_agreement":null},{"id":"W2129123581","doi":"10.3115/1641976.1641983","title":"Evaluation of several phonetic similarity algorithms on the task of cognate identification","year":2006,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":45,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Similarity (geometry); Context (archaeology); Hidden Markov model; Identification (biology); Task (project management); Artificial intelligence; Set (abstract data type); Cognate; Cluster analysis; Natural language processing; Speech recognition; Pattern recognition (psychology); Image (mathematics); Linguistics","score_opus":0.025835460491572527,"score_gpt":0.2935216319955214,"score_spread":0.26768617150394886,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2129123581","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8143561,0.0059916563,0.16420238,0.00047238806,0.00032945248,0.00051383616,0.0008471738,0.0053781928,0.00790873],"genre_scores_gemma":[0.8107142,0.00071949064,0.18298762,0.0001571219,0.00010279395,0.0002131759,0.0030814814,0.00031111945,0.0017130021],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99084395,0.0035785814,0.0010845418,0.0021449693,0.0018703536,0.00047750789],"domain_scores_gemma":[0.9734163,0.017748252,0.0008792107,0.0020506815,0.0049918033,0.0009137011],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010631811,0.0019404616,0.001999336,0.005301379,0.0016064935,0.0021001676,0.0025387916,0.003365715,0.0015252815],"category_scores_gemma":[0.029967982,0.00037956637,0.0009248203,0.0030116,0.0008555969,0.0036446461,0.0023171944,0.0011557158,0.0009671666],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0031300108,0.001602826,0.030920668,0.00069483067,0.0010941941,0.00016994754,0.00047116526,0.10694468,0.009268522,0.0030841317,0.0039313626,0.8386876],"study_design_scores_gemma":[0.00037160754,0.0019306068,0.013936175,0.00006297805,0.0003022886,0.0003593299,0.0006523729,0.9527168,0.023146177,0.0041500577,0.0022874053,0.00008413165],"about_ca_topic_score_codex":0.0071795867,"about_ca_topic_score_gemma":0.007905433,"teacher_disagreement_score":0.010631811,"about_ca_system_score_codex":0.0015115055,"about_ca_system_score_gemma":0.0019010765,"threshold_uncertainty_score":0.056227088},"labels":[],"label_agreement":null},{"id":"W2129388736","doi":"10.1515/pralin-2015-0008","title":"A Python-based Interface for Wide Coverage Lexicalized Tree-adjoining Grammars","year":2015,"lang":"en","type":"article","venue":"The Prague Bulletin of Mathematical Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Python (programming language); Parsing; Programming language; Artificial intelligence; L-attributed grammar; Rule-based machine translation; Adjunction; Tree-adjoining grammar; Natural language processing; Tree (set theory); Abstract syntax tree; Tree structure; Grammar; Data structure; Context-free grammar; Mathematics; Linguistics","score_opus":0.03020264464375103,"score_gpt":0.29662733924707696,"score_spread":0.2664246946033259,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2129388736","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.003481917,0.00006588502,0.5522884,0.000213896,0.00008801217,0.00049138383,0.0046361913,0.4314214,0.0073129344],"genre_scores_gemma":[0.14681624,0.00034599815,0.59276414,0.0014983506,0.00013351571,0.0030560538,0.025504896,0.20487498,0.025005853],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9977508,0.00043670068,0.00034485763,0.00044680017,0.00073090236,0.00028995977],"domain_scores_gemma":[0.9964886,0.0017515264,0.00019692216,0.0006417846,0.0005901108,0.00033107548],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003633167,0.0017894767,0.0013112649,0.0016537973,0.0008197383,0.0027671813,0.0045890342,0.0013352507,0.065681346],"category_scores_gemma":[0.008036399,0.0018242941,0.00214816,0.0010450503,0.0018013024,0.0050716135,0.0076807067,0.0029185656,0.029895619],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0032535314,0.00081811356,0.0077237953,0.0023536056,0.00031193145,0.003815711,0.004303603,0.025921794,0.038010854,0.096556984,0.42780778,0.3891223],"study_design_scores_gemma":[0.0009778203,0.00019471395,0.0037612158,0.0004956051,0.00010223849,0.0020099152,0.00037442896,0.29282886,0.047437675,0.10071102,0.55060786,0.0004986959],"about_ca_topic_score_codex":0.0032520972,"about_ca_topic_score_gemma":0.0025837154,"teacher_disagreement_score":0.065681346,"about_ca_system_score_codex":0.0015771675,"about_ca_system_score_gemma":0.0024122244,"threshold_uncertainty_score":0.21972597},"labels":[],"label_agreement":null},{"id":"W2129423172","doi":"10.1017/s0261444809990218","title":"Colloquium – Bridging computational and applied linguistics: Implementation challenges and benefits","year":2009,"lang":"en","type":"article","venue":"Language Teaching","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Bridging (networking); Applied linguistics; Linguistics; Computer science; Sociology; Cognitive science; Psychology; Philosophy","score_opus":0.014460125863062324,"score_gpt":0.29583597555123164,"score_spread":0.2813758496881693,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2129423172","genre_codex":"methods","genre_gemma":"commentary","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"commentary","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023332968,0.0017172523,0.8393724,0.03717087,0.010091759,0.00085660303,0.0014349987,0.035012934,0.05101018],"genre_scores_gemma":[0.1790196,0.0010181281,0.6819656,0.008731398,0.0018749099,0.0009357672,0.0047853785,0.017907387,0.10376185],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9873186,0.0069113364,0.0011689627,0.0023727368,0.0015392734,0.00068915647],"domain_scores_gemma":[0.9465758,0.011484964,0.0007237471,0.022965547,0.012928584,0.005321285],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.023764426,0.0014737732,0.0016312031,0.001835182,0.0033638203,0.009011328,0.00616794,0.004063712,0.10698002],"category_scores_gemma":[0.04894508,0.0021328693,0.0012042124,0.002724349,0.0033216167,0.025353359,0.017266873,0.00604073,0.05278351],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011643834,0.00077144196,0.003579905,0.00060803676,0.00015936642,0.0006925889,0.0037494209,0.0017599883,0.013996236,0.10236139,0.387372,0.4837853],"study_design_scores_gemma":[0.00057088985,0.00055611756,0.002425596,0.00039882114,0.00010581136,0.0014654752,0.0025873205,0.04639181,0.013840802,0.11483948,0.81653273,0.00028519772],"about_ca_topic_score_codex":0.011911609,"about_ca_topic_score_gemma":0.0144842975,"teacher_disagreement_score":0.10698002,"about_ca_system_score_codex":0.002860353,"about_ca_system_score_gemma":0.012384799,"threshold_uncertainty_score":0.3578838},"labels":[],"label_agreement":null},{"id":"W2129576665","doi":"10.1007/s00426-015-0660-2","title":"Information theory and artificial grammar learning: inferring grammaticality from redundancy","year":2015,"lang":"en","type":"article","venue":"Psychological Research","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University; University of Manitoba","funders":"","keywords":"Grammaticality; Grammar; Natural language processing; Redundancy (engineering); Computer science; Artificial intelligence; Linguistics; Psychology; Philosophy","score_opus":0.1916290607786856,"score_gpt":0.4685378858944448,"score_spread":0.2769088251157592,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2129576665","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15523255,0.0035100623,0.8298979,0.0054435832,0.00011672068,0.000044696502,0.00021157567,0.00024751795,0.005295353],"genre_scores_gemma":[0.8515465,0.0021197665,0.14454365,0.00044977732,0.00024864182,0.00008242112,0.0002910206,0.00007934711,0.0006389382],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.9963899,0.0021980782,0.000213237,0.00051355455,0.0005848852,0.00010028742],"domain_scores_gemma":[0.91531855,0.07677514,0.0022230595,0.0036964878,0.0015307991,0.00045597868],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007060361,0.0006933986,0.00133611,0.003594834,0.0007783959,0.0036372584,0.0018969482,0.001788896,0.0019626233],"category_scores_gemma":[0.06770005,0.00080639817,0.0011457786,0.0022153966,0.006585216,0.012696357,0.0021570001,0.003006306,0.00027691564],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00040653872,0.00027609433,0.013832847,0.0010032797,0.0005113609,0.00036733566,0.0017964104,0.06284579,0.004837886,0.7001761,0.002577385,0.21136893],"study_design_scores_gemma":[0.000013016763,0.000019958856,0.00063412206,0.000030227813,0.00003118497,0.0000636852,0.000059956248,0.04605724,0.0007078978,0.9520078,0.000355996,0.000018938375],"about_ca_topic_score_codex":0.0014586298,"about_ca_topic_score_gemma":0.0013440556,"teacher_disagreement_score":0.007060361,"about_ca_system_score_codex":0.0013487873,"about_ca_system_score_gemma":0.0013181254,"threshold_uncertainty_score":0.03733921},"labels":[],"label_agreement":null},{"id":"W2129909699","doi":"10.1162/coli_a_00085","title":"Learning Entailment Relations by Global Graph Structure Optimization","year":2011,"lang":"en","type":"article","venue":"Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":39,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"University of Haifa; Azrieli Foundation; Israel Science Foundation","keywords":"Logical consequence; Computer science; Textual entailment; Inference; Graph; Transitive relation; Artificial intelligence; Constraint (computer-aided design); Theoretical computer science; Natural language processing; Mathematics; Combinatorics","score_opus":0.010916148327742984,"score_gpt":0.2556109568737796,"score_spread":0.24469480854603662,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2129909699","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017605506,0.0002164884,0.97826195,0.00031445155,0.000015694923,0.00009843785,0.0002908559,0.002016207,0.0011804512],"genre_scores_gemma":[0.24425685,0.00033846265,0.7493918,0.00023783158,0.00005225518,0.00018614356,0.003054315,0.0005597311,0.0019225727],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99865615,0.00038649692,0.000076238146,0.00056519767,0.00023677161,0.00007920425],"domain_scores_gemma":[0.996351,0.0026239697,0.00023125479,0.00048777257,0.00025038296,0.000055553945],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017179848,0.0015965627,0.0015907557,0.002204729,0.000857718,0.001560144,0.0019881527,0.0015550939,0.004457093],"category_scores_gemma":[0.0064744526,0.0009055479,0.0020964353,0.0018143611,0.0014829104,0.0060425685,0.0019846635,0.0029076238,0.000997568],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021868455,0.00033638754,0.0028270888,0.000493319,0.0002410617,0.00022068135,0.00026852908,0.44304982,0.009199821,0.06211682,0.009620806,0.47140694],"study_design_scores_gemma":[0.000032005188,0.000041473246,0.00025164307,0.000016619599,0.000051766066,0.000045450037,0.000049444167,0.88457036,0.0032932912,0.1103105,0.0013277332,0.000009776571],"about_ca_topic_score_codex":0.00375409,"about_ca_topic_score_gemma":0.010602019,"teacher_disagreement_score":0.004457093,"about_ca_system_score_codex":0.0016925765,"about_ca_system_score_gemma":0.0018427847,"threshold_uncertainty_score":0.0149104595},"labels":[],"label_agreement":null},{"id":"W2130327685","doi":"10.71781/10891","title":"Résumé automatique de texte arabe","year":2004,"lang":"fr","type":"dissertation","venue":"Open MIND","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"National Institute of Standards and Technology","keywords":"Humanities; Philosophy","score_opus":0.028099308222055003,"score_gpt":0.3493259385768133,"score_spread":0.3212266303547583,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2130327685","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.042904142,0.019801598,0.51562196,0.0086273,0.008488825,0.0017881943,0.0931899,0.076762825,0.23281522],"genre_scores_gemma":[0.12898123,0.0105268955,0.47779676,0.0008649928,0.002106294,0.0010216634,0.0703993,0.010113704,0.29818916],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9982686,0.00032991037,0.00015495198,0.00040812808,0.0007285918,0.00010986379],"domain_scores_gemma":[0.9958289,0.0012341918,0.00016487275,0.00052647595,0.002140425,0.00010515841],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013207356,0.0016604143,0.000993053,0.0082576135,0.0019055043,0.0053978837,0.00087486184,0.0006711678,0.08080713],"category_scores_gemma":[0.009076867,0.0005897889,0.0012520106,0.0046432824,0.0010364342,0.0024535975,0.0017078137,0.0014694919,0.053355444],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003606906,0.000065041466,0.0023192754,0.0017051937,0.000102424965,0.00051638123,0.001140291,0.0021404203,0.02231549,0.019838141,0.2084281,0.7410685],"study_design_scores_gemma":[0.00004493998,0.000042404918,0.0051908223,0.00036440877,0.00006556775,0.0006704942,0.00054086506,0.0044608926,0.022741258,0.006090498,0.95972323,0.00006455454],"about_ca_topic_score_codex":0.07026108,"about_ca_topic_score_gemma":0.04428531,"teacher_disagreement_score":0.08080713,"about_ca_system_score_codex":0.0024751483,"about_ca_system_score_gemma":0.0036960386,"threshold_uncertainty_score":0.27032673},"labels":[],"label_agreement":null},{"id":"W2131152792","doi":"10.1109/skg.2007.255","title":"Resolving Quantifier and Number Restriction to Question OWL Ontologies","year":2007,"lang":"en","type":"article","venue":"Third International Conference on Semantics, Knowledge and Grid (SKG 2007)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Quantifier (linguistics); Parsing; Natural language; Rank (graph theory); Question answering; Feature (linguistics); Natural language user interface; Ontology; Artificial intelligence; Information retrieval; Natural language processing; Programming language; Mathematics; Linguistics; Combinatorics","score_opus":0.03609097007582002,"score_gpt":0.3510652886381209,"score_spread":0.3149743185623009,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2131152792","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.042811908,0.00024108546,0.9408194,0.0011440852,0.000117453885,0.00033982666,0.0011471706,0.009184749,0.004194481],"genre_scores_gemma":[0.21936047,0.00022447658,0.76935834,0.0007342888,0.0001404078,0.00026927004,0.0034107245,0.0017177954,0.004784271],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9871349,0.003886894,0.0015616609,0.0018240524,0.0049665775,0.00062590535],"domain_scores_gemma":[0.9739648,0.016622972,0.0022145389,0.0032262693,0.0034884051,0.00048308936],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009218916,0.0010375145,0.0015594972,0.0033823715,0.0019133658,0.0033607825,0.0022893134,0.0017473792,0.0060618613],"category_scores_gemma":[0.03206128,0.0009605098,0.0018683438,0.0017521874,0.0017603047,0.008386911,0.0052123615,0.0027269593,0.0010829556],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00091497524,0.0006271356,0.013267909,0.002409007,0.00036725562,0.0027591048,0.013580952,0.031146927,0.058667235,0.32383862,0.039223265,0.51319766],"study_design_scores_gemma":[0.0001749526,0.0001625271,0.005409483,0.0003745901,0.00020605345,0.0020710805,0.0041595115,0.4149862,0.07385917,0.27582356,0.22249582,0.00027702437],"about_ca_topic_score_codex":0.010684293,"about_ca_topic_score_gemma":0.013681213,"teacher_disagreement_score":0.010684293,"about_ca_system_score_codex":0.002303962,"about_ca_system_score_gemma":0.003587785,"threshold_uncertainty_score":0.04875487},"labels":[],"label_agreement":null},{"id":"W2131462252","doi":"","title":"A Scalable Hierarchical Distributed Language Model","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":852,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Language model; Computer science; Hierarchical database model; Cache language model; Artificial intelligence; Scalability; Probabilistic logic; Simple (philosophy); Feature (linguistics); Tree (set theory); Word (group theory); Binary tree; Machine learning; Natural language processing; Data mining; Algorithm; Natural language; Universal Networking Language; Mathematics","score_opus":0.014993362832299193,"score_gpt":0.2616674500190941,"score_spread":0.2466740871867949,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2131462252","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01640158,0.00037121642,0.9723628,0.0006421527,0.00011968867,0.000098505996,0.000862209,0.0038160468,0.0053259004],"genre_scores_gemma":[0.5149992,0.0005761949,0.4649035,0.0005244656,0.00020536137,0.0005436594,0.002698427,0.0004911366,0.01505804],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994161,0.00014177062,0.000031594896,0.0001894454,0.00015643524,0.00006472189],"domain_scores_gemma":[0.99924254,0.00033436881,0.000048871978,0.0001575045,0.00017172759,0.0000449955],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00087130425,0.00057128817,0.0010056923,0.0004925264,0.00050452276,0.0010730543,0.0023058008,0.0010995808,0.005750282],"category_scores_gemma":[0.002722734,0.00050690764,0.0008783713,0.0009679801,0.00050162233,0.002466597,0.0014698887,0.0013818016,0.002350551],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016294025,0.000103378115,0.00070412783,0.000118807075,0.000086143664,0.00019214406,0.00008350994,0.8154629,0.0048193154,0.036597777,0.012824063,0.1288449],"study_design_scores_gemma":[0.000007829938,0.0000064652368,0.00003865876,0.0000015431742,0.000003982342,0.000009696153,0.0000031408333,0.99288344,0.00018234299,0.006255966,0.0006039061,0.0000029031896],"about_ca_topic_score_codex":0.013109822,"about_ca_topic_score_gemma":0.015273387,"teacher_disagreement_score":0.013109822,"about_ca_system_score_codex":0.0010422866,"about_ca_system_score_gemma":0.0018995319,"threshold_uncertainty_score":0.026067078},"labels":[],"label_agreement":null},{"id":"W2131526828","doi":"10.1162/coli.2006.32.2.223","title":"Building and Using a Lexical Knowledge Base of Near-Synonym Differences","year":2006,"lang":"en","type":"article","venue":"Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":82,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; University of Toronto; University of Ottawa; University of Pennsylvania","keywords":"Computer science; Synonym (taxonomy); Natural language processing; Knowledge base; Artificial intelligence; Mistake; Natural language; Bilingual dictionary; Word (group theory); Linguistics","score_opus":0.026568550280453792,"score_gpt":0.3117453974176719,"score_spread":0.2851768471372181,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2131526828","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.042655732,0.00031900854,0.93723714,0.00065424194,0.00012439296,0.00052908587,0.0044864584,0.0050703264,0.008923642],"genre_scores_gemma":[0.17336993,0.00039361144,0.8101042,0.00027997943,0.000053966258,0.00037722106,0.012436222,0.00045217856,0.0025327452],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99781287,0.0005226783,0.0002644403,0.0007227087,0.0005798735,0.00009743702],"domain_scores_gemma":[0.9941181,0.0031782163,0.00031960165,0.0012107544,0.0010440074,0.00012928329],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016403534,0.000929211,0.0010799026,0.0058903093,0.0014389327,0.0021862057,0.0025782466,0.0013080119,0.007434053],"category_scores_gemma":[0.011976357,0.0009000483,0.0011033425,0.0037573427,0.0009858871,0.0072645308,0.0026233378,0.0020393168,0.0034695494],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003297457,0.0006179366,0.007016207,0.0009936413,0.00023710703,0.0024884543,0.0023010732,0.01915643,0.030324498,0.04218738,0.022027684,0.87231976],"study_design_scores_gemma":[0.00032748954,0.00046256202,0.010886828,0.0010020147,0.0006826536,0.0046719736,0.0032770995,0.4281553,0.095396265,0.24101353,0.21367195,0.00045232664],"about_ca_topic_score_codex":0.0033661341,"about_ca_topic_score_gemma":0.0069226124,"teacher_disagreement_score":0.007434053,"about_ca_system_score_codex":0.001021726,"about_ca_system_score_gemma":0.0019797932,"threshold_uncertainty_score":0.024869323},"labels":[],"label_agreement":null},{"id":"W2131809738","doi":"10.1080/01690960344000152","title":"Admitting that admitting verb sense into corpus analyses makes sense","year":2004,"lang":"en","type":"article","venue":"Language and Cognitive Processes","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":76,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Verb; Meaning (existential); Linguistics; Ambiguity; Consistency (knowledge bases); Psychology; Exploit; Computer science; Natural language processing; Artificial intelligence; Philosophy","score_opus":0.0252927054938304,"score_gpt":0.32188646631154894,"score_spread":0.29659376081771854,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2131809738","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0804068,0.001830895,0.8627625,0.026254544,0.005253189,0.00028012772,0.0014625579,0.0017540856,0.019995388],"genre_scores_gemma":[0.6876191,0.000916393,0.29081002,0.008336926,0.002732617,0.00087871257,0.0015676761,0.0013014284,0.005837016],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.96137,0.021841006,0.004238483,0.0051473263,0.0067455,0.00065769197],"domain_scores_gemma":[0.8189863,0.11374816,0.011568152,0.044744816,0.010016492,0.00093602494],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.049826425,0.001510735,0.0013663638,0.0038354776,0.0029555988,0.010771837,0.0032598206,0.002512062,0.004551259],"category_scores_gemma":[0.2537901,0.0010478826,0.0009500522,0.0034056604,0.009392519,0.01562845,0.0059492136,0.006746686,0.0016351459],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00043227285,0.00018857994,0.037179805,0.0013197692,0.00090886565,0.0011154818,0.026345992,0.0023207292,0.022079766,0.63901156,0.033598717,0.23549852],"study_design_scores_gemma":[0.000032731514,0.000102399055,0.008170146,0.00032640665,0.00017028929,0.0010996056,0.0041174856,0.01074909,0.006355741,0.8993974,0.06936923,0.000109582325],"about_ca_topic_score_codex":0.0013844076,"about_ca_topic_score_gemma":0.0037054536,"teacher_disagreement_score":0.049826425,"about_ca_system_score_codex":0.0008721619,"about_ca_system_score_gemma":0.0020395701,"threshold_uncertainty_score":0.26351047},"labels":[],"label_agreement":null},{"id":"W2132001515","doi":"10.3115/1626355.1626372","title":"Mixture-model adaptation for SMT","year":2007,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":253,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Computer science; Machine translation; Granularity; Adaptation (eye); BLEU; Translation (biology); Domain adaptation; Log-linear model; Artificial intelligence; Baseline (sea); Language model; Natural language processing; Algorithm; Linear model; Machine learning; Programming language","score_opus":0.02364179146064855,"score_gpt":0.3049751683204618,"score_spread":0.28133337685981324,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2132001515","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0013077854,0.00026496098,0.9946031,0.000098765835,0.000086944376,0.000038374772,0.00013859369,0.002433228,0.0010282199],"genre_scores_gemma":[0.1298004,0.0008173825,0.8548298,0.0003309304,0.00024400579,0.00050111784,0.0020568708,0.002110157,0.009309384],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99810433,0.0008678616,0.000110582936,0.00040911065,0.00042174396,0.00008634387],"domain_scores_gemma":[0.9977738,0.0011206936,0.000093614326,0.0005488326,0.00040304588,0.000059915277],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002298851,0.0014157062,0.001237829,0.0011669202,0.00065900787,0.0012020298,0.0017780681,0.0014244757,0.0078118145],"category_scores_gemma":[0.008119074,0.00085299206,0.0017912034,0.0020868678,0.0005953886,0.002339152,0.001960437,0.0033135943,0.008248528],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00033089166,0.00016294826,0.00085997145,0.0003346719,0.0005486141,0.00017410336,0.00018174706,0.3547237,0.022554774,0.028432522,0.017775234,0.5739208],"study_design_scores_gemma":[0.000016361817,0.000038463164,0.00024759868,0.000015736852,0.000043966502,0.000114662915,0.000014035222,0.963197,0.0069857975,0.018918386,0.010371387,0.000036611127],"about_ca_topic_score_codex":0.0036817435,"about_ca_topic_score_gemma":0.0052509326,"teacher_disagreement_score":0.0078118145,"about_ca_system_score_codex":0.00087955676,"about_ca_system_score_gemma":0.0009200506,"threshold_uncertainty_score":0.02613312},"labels":[],"label_agreement":null},{"id":"W2132130844","doi":"10.1007/s10726-006-9024-z","title":"Comparative Analysis of Text Data in Successful Face-to-Face and Electronic Negotiations","year":2006,"lang":"en","type":"article","venue":"Group Decision and Negotiation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Negotiation; Vocabulary; Face (sociological concept); Generalization; Artificial intelligence; Natural language processing; Similarity (geometry); Face-to-face; Linguistics; Sociology; Epistemology; Social science","score_opus":0.0173132733298819,"score_gpt":0.31303933452525046,"score_spread":0.29572606119536854,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2132130844","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.96886665,0.00076354237,0.016147166,0.0006789765,0.00011922297,0.00023081533,0.003739123,0.0003549625,0.009099632],"genre_scores_gemma":[0.98263043,0.00025272628,0.009039586,0.0000693206,0.0000735876,0.00024455157,0.0052339467,0.0001353682,0.0023204442],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99143785,0.0050437623,0.0007569334,0.00055875536,0.001904006,0.0002987434],"domain_scores_gemma":[0.69671386,0.2815628,0.0061753327,0.0036696768,0.011139312,0.00073892495],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0072700256,0.00033247587,0.0005614148,0.004834551,0.0011052544,0.0030240456,0.0009605143,0.0012638867,0.0056151035],"category_scores_gemma":[0.101162314,0.00021280476,0.0004006002,0.005524377,0.0010771563,0.0036527694,0.0010716476,0.0010643825,0.002054024],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.021621397,0.0032822092,0.19439319,0.0055885795,0.00071675313,0.005399357,0.039307177,0.024100624,0.06085574,0.027856667,0.019977676,0.5969006],"study_design_scores_gemma":[0.0005287653,0.001994454,0.51249534,0.0009013579,0.00090112496,0.0033728713,0.045285836,0.2763295,0.06687828,0.048421185,0.04234282,0.0005484025],"about_ca_topic_score_codex":0.0013479784,"about_ca_topic_score_gemma":0.0016105484,"teacher_disagreement_score":0.0072700256,"about_ca_system_score_codex":0.0007565021,"about_ca_system_score_gemma":0.000616395,"threshold_uncertainty_score":0.038448036},"labels":[],"label_agreement":null},{"id":"W2132187905","doi":"10.1109/tasl.2010.2040793","title":"Integration of Statistical Models for Dictation of Document Translations in a Machine-Aided Human Translation Task","year":2010,"lang":"en","type":"article","venue":"IEEE Transactions on Audio Speech and Language Processing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":34,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"McGill University","keywords":"Computer science; Dictation; Machine translation; Natural language processing; NIST; Language model; Artificial intelligence; Word error rate; Speech translation; Baseline (sea); Task (project management); Speech recognition; Vocabulary; Machine translation software usability; Translation (biology); Word (group theory); Evaluation of machine translation; Example-based machine translation; Linguistics","score_opus":0.015678018987896825,"score_gpt":0.30618815517328546,"score_spread":0.2905101361853886,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2132187905","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.052065946,0.0007353849,0.93966305,0.0006331813,0.00017547363,0.00013666578,0.00020688686,0.0035060123,0.0028773805],"genre_scores_gemma":[0.6463843,0.00093436986,0.3409968,0.00028574924,0.00028181722,0.0004077763,0.00096678466,0.0005821359,0.009160266],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99866235,0.00055475434,0.00010738437,0.00027241057,0.0003515244,0.00005156527],"domain_scores_gemma":[0.9957432,0.0030924233,0.00021828071,0.0002417959,0.0006351262,0.00006913306],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022484912,0.00081736484,0.00093830604,0.0007032774,0.0005140929,0.0013963259,0.0008961304,0.0010891296,0.0024292264],"category_scores_gemma":[0.008142592,0.0006625342,0.0009924297,0.0005902914,0.0005175057,0.001740997,0.0006908821,0.0012072715,0.0025357804],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008397632,0.00029793452,0.003566107,0.00027618353,0.00029464407,0.00029283232,0.0003578028,0.57575876,0.019345284,0.0059071253,0.003780955,0.38928258],"study_design_scores_gemma":[0.000017301089,0.000097907,0.0005242456,0.0000075296657,0.000037963193,0.00009889893,0.000018837662,0.992706,0.0035502003,0.0017162727,0.0011980348,0.000026738655],"about_ca_topic_score_codex":0.00467923,"about_ca_topic_score_gemma":0.006781792,"teacher_disagreement_score":0.00467923,"about_ca_system_score_codex":0.00066780054,"about_ca_system_score_gemma":0.0012200146,"threshold_uncertainty_score":0.011891305},"labels":[],"label_agreement":null},{"id":"W2132891320","doi":"10.1007/s10590-011-9112-y","title":"Improved Arabic-to-English statistical machine translation by reordering post-verbal subjects for word alignment","year":2011,"lang":"en","type":"article","venue":"Machine Translation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":44,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"University of Pennsylvania","keywords":"Computer science; Machine translation; Natural language processing; Word order; Artificial intelligence; Phrase; Parsing; Word (group theory); Translation (biology); Speech recognition; Arabic; Computational linguistics; Linguistics","score_opus":0.017936208754936653,"score_gpt":0.2589407352169916,"score_spread":0.24100452646205497,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2132891320","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.038659018,0.0008216235,0.9219839,0.0006267181,0.001312724,0.0003420261,0.0032961585,0.024622727,0.008335003],"genre_scores_gemma":[0.12603785,0.00052320433,0.8511017,0.00022422256,0.00031806043,0.0002606301,0.008393636,0.0039330428,0.009207746],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99834514,0.00064253586,0.00024777374,0.00035832287,0.000288697,0.000117474585],"domain_scores_gemma":[0.99495846,0.0013354803,0.00020784406,0.0009093176,0.0024814168,0.00010759008],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012368226,0.0021084042,0.0011467417,0.0017043866,0.0011237453,0.0019948313,0.0008943033,0.00069762074,0.013944941],"category_scores_gemma":[0.005473094,0.0005519444,0.0009084572,0.0021834904,0.0003872928,0.0016009476,0.0012272946,0.0018188701,0.016049307],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008858418,0.00041575864,0.0012758123,0.0008385934,0.00017822055,0.0007709028,0.0008544646,0.011413508,0.13497552,0.013230573,0.037865773,0.79729503],"study_design_scores_gemma":[0.00036387288,0.0008872119,0.0042176354,0.00018282446,0.00052083976,0.0015444619,0.001192793,0.46325538,0.3407238,0.026408296,0.16036834,0.00033443628],"about_ca_topic_score_codex":0.0036562977,"about_ca_topic_score_gemma":0.0073693544,"teacher_disagreement_score":0.013944941,"about_ca_system_score_codex":0.0004973943,"about_ca_system_score_gemma":0.0024252245,"threshold_uncertainty_score":0.04665053},"labels":[],"label_agreement":null},{"id":"W2132922574","doi":"10.7202/003549ar","title":"Extraction d’une phraséologie bilingue en langue de spécialité : corpus parallèles et corpus comparables","year":2002,"lang":"fr","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Art; Philosophy","score_opus":0.1070598189298972,"score_gpt":0.3338831408659913,"score_spread":0.2268233219360941,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2132922574","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20252092,0.003222275,0.7516086,0.0010821288,0.00056567736,0.0009864507,0.008836114,0.006340687,0.024837157],"genre_scores_gemma":[0.2636817,0.0016366746,0.6994427,0.00021871686,0.0002411656,0.0013198677,0.020317728,0.0018173927,0.011324105],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9960711,0.0013295278,0.00058790017,0.0010995894,0.00076711085,0.00014486224],"domain_scores_gemma":[0.992288,0.0037400885,0.00040607783,0.0015384303,0.0018823869,0.00014507292],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033161098,0.0009770492,0.001355474,0.005669244,0.0017484665,0.0041458476,0.00082902896,0.0012149562,0.008357241],"category_scores_gemma":[0.01094401,0.0009221661,0.0011135808,0.0071698455,0.0014916174,0.0045645256,0.0023449175,0.00184032,0.002635973],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001099799,0.00040886694,0.0062075797,0.003197568,0.0003665339,0.0028007866,0.010263403,0.0043473113,0.24447589,0.06434793,0.014338232,0.64814615],"study_design_scores_gemma":[0.0004332096,0.00067288097,0.028633736,0.0007257497,0.00078380556,0.0071782246,0.007936801,0.04476624,0.22253627,0.06491697,0.6210489,0.0003671294],"about_ca_topic_score_codex":0.0046658246,"about_ca_topic_score_gemma":0.0047688293,"teacher_disagreement_score":0.008357241,"about_ca_system_score_codex":0.0013958493,"about_ca_system_score_gemma":0.0022470218,"threshold_uncertainty_score":0.027957737},"labels":[],"label_agreement":null},{"id":"W2132959801","doi":"","title":"Incremental Segmentation and Decoding Strategies for Simultaneous Translation","year":2013,"lang":"en","type":"article","venue":"International Joint Conference on Natural Language Processing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Decoding methods; Segmentation; Artificial intelligence; Machine translation; Speech recognition; Interpreter; Natural language processing; Speech translation; Active listening; Phrase; Task (project management); Latency (audio); Translation (biology); Algorithm; Programming language; Communication","score_opus":0.02833968079378524,"score_gpt":0.31117202112435544,"score_spread":0.2828323403305702,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2132959801","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024094405,0.00034842975,0.9680388,0.00010611767,0.00006866621,0.00009135614,0.00012511578,0.0036116326,0.0035156012],"genre_scores_gemma":[0.24746576,0.00032039563,0.7465414,0.00014260692,0.000067034365,0.00018230868,0.0006952788,0.0010482526,0.0035368926],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985215,0.0004960381,0.00011192211,0.00032317307,0.0004087095,0.0001386537],"domain_scores_gemma":[0.9955238,0.002467539,0.0001526054,0.00074126694,0.0010002253,0.000114557624],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011897598,0.0015458799,0.00077491975,0.0009082578,0.0006734453,0.0017151721,0.0015983689,0.0017236826,0.00477539],"category_scores_gemma":[0.0073826835,0.0005481722,0.0007531188,0.0012518151,0.0008316542,0.0022309918,0.001337794,0.0015519202,0.0034423347],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00091657834,0.00018231024,0.0011499221,0.0004554378,0.00007999012,0.0004687773,0.0013523614,0.038247354,0.1724898,0.018121706,0.0038408826,0.7626949],"study_design_scores_gemma":[0.00014293555,0.0005070817,0.001483224,0.000056867873,0.00018899584,0.0013915533,0.00055310375,0.6521763,0.30290237,0.024560168,0.0158797,0.00015773899],"about_ca_topic_score_codex":0.0027102798,"about_ca_topic_score_gemma":0.005075884,"teacher_disagreement_score":0.00477539,"about_ca_system_score_codex":0.0005516809,"about_ca_system_score_gemma":0.0014685609,"threshold_uncertainty_score":0.015975237},"labels":[],"label_agreement":null},{"id":"W2133243346","doi":"10.7202/013942ar","title":"Inupiaq writing and international Inuit relations","year":2006,"lang":"en","type":"article","venue":"Études/Inuit/Studies","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Indigenous; Vitality; Reading (process); Linguistics; Indigenous language; Writing system; The arctic; Arctic; Cultural exchange; History; Geography; Ethnology; Oceanography; Ecology; Biology; Philosophy; Geology","score_opus":0.029170605411231114,"score_gpt":0.3214450824098563,"score_spread":0.2922744769986252,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2133243346","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8850031,0.0016506729,0.0019560694,0.0018356373,0.000058164856,0.00003023369,0.00005405108,0.000031185613,0.10938088],"genre_scores_gemma":[0.990348,0.00061313494,0.0009803934,0.000076344666,0.0000085050815,0.000015700032,0.000032085598,0.000011691533,0.007914162],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9987992,0.0007071197,0.00007366691,0.00013528967,0.00014044762,0.00014420832],"domain_scores_gemma":[0.9964616,0.0020361831,0.00064362277,0.00020031637,0.00042652074,0.00023180002],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014405525,0.00025888265,0.00019926386,0.0011747074,0.004828576,0.0045218286,0.00055643934,0.00036617863,0.007492],"category_scores_gemma":[0.006948091,0.0001248418,0.00013935236,0.0022191564,0.002273493,0.0023299826,0.002095503,0.0008473336,0.0002916808],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013040214,0.00019081333,0.060593642,0.00028514708,0.000039157523,0.0017925714,0.7100968,0.00025056166,0.00178952,0.090337634,0.0033466304,0.13114718],"study_design_scores_gemma":[0.0000213487,0.00016160775,0.11968386,0.00035959977,0.00007990587,0.0014018228,0.75105166,0.0020492906,0.0052941237,0.024718193,0.09513078,0.000047755035],"about_ca_topic_score_codex":0.07407271,"about_ca_topic_score_gemma":0.13778265,"teacher_disagreement_score":0.07407271,"about_ca_system_score_codex":0.0027031833,"about_ca_system_score_gemma":0.0019546575,"threshold_uncertainty_score":0.14728314},"labels":[],"label_agreement":null},{"id":"W21337280","doi":"","title":"Adaptive Language and Translation Models for Interactive Machine Translation.","year":2004,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":66,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Perplexity; Machine translation; Translation (biology); Natural language processing; Transfer-based machine translation; Artificial intelligence; Machine translation software usability; Example-based machine translation; Computer-assisted translation; Context (archaeology); Language model; Dynamic and formal equivalence","score_opus":0.024078534189354704,"score_gpt":0.2839559028019245,"score_spread":0.2598773686125698,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W21337280","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007284907,0.00044902824,0.9836986,0.00047532812,0.000090422946,0.00015564109,0.00018592642,0.0040715192,0.0035886324],"genre_scores_gemma":[0.23842435,0.0006937689,0.75217986,0.00033478558,0.00012527383,0.0009323514,0.0013056697,0.0006905171,0.0053134174],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99706846,0.0018615912,0.00011280262,0.0003859546,0.00048779856,0.00008346356],"domain_scores_gemma":[0.9940851,0.004158995,0.00021258554,0.0010762934,0.00036077766,0.00010627955],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030130523,0.00085174455,0.00063643046,0.00053055375,0.0006767155,0.0013956797,0.0019064577,0.0017296795,0.008321467],"category_scores_gemma":[0.012144727,0.0005953835,0.0010472065,0.00081833627,0.0009666774,0.0037677074,0.001216015,0.002428911,0.0035894131],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000981641,0.0005758069,0.0018851256,0.0007009982,0.0004643261,0.00050595915,0.0011322447,0.4149265,0.024312377,0.1493404,0.019929778,0.38524485],"study_design_scores_gemma":[0.00005037675,0.0001153902,0.00024860902,0.000026245287,0.000054500913,0.00021110584,0.000054815755,0.9281447,0.005016057,0.051685452,0.014354301,0.000038502065],"about_ca_topic_score_codex":0.0037090005,"about_ca_topic_score_gemma":0.0046838205,"teacher_disagreement_score":0.008321467,"about_ca_system_score_codex":0.0010868504,"about_ca_system_score_gemma":0.001061403,"threshold_uncertainty_score":0.027838051},"labels":[],"label_agreement":null},{"id":"W2133831949","doi":"10.7202/017542ar","title":"Des déclencheurs des énumérations d’entités nommées sur le Web","year":2008,"lang":"fr","type":"article","venue":"Revue québécoise de linguistique","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Philosophy","score_opus":0.04307813863473451,"score_gpt":0.2859660356106398,"score_spread":0.24288789697590532,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2133831949","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.42031696,0.0016689092,0.54246056,0.0010212418,0.0001473572,0.00020846757,0.003672129,0.0063376464,0.024166679],"genre_scores_gemma":[0.66362184,0.00120018,0.30308026,0.00018961284,0.00007715087,0.00022252162,0.0051606814,0.0011177149,0.025330136],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99842525,0.00038913428,0.00017733246,0.00040613685,0.00048513513,0.000116980016],"domain_scores_gemma":[0.98687845,0.0074180346,0.0012214076,0.0014307728,0.0027813439,0.00026999196],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018517905,0.0004821934,0.0005522778,0.0032465959,0.0013342567,0.0030327316,0.0005948716,0.00084270455,0.0068379566],"category_scores_gemma":[0.01253424,0.00069558463,0.00064359175,0.0028282339,0.0011260401,0.005676808,0.0012236848,0.001011831,0.0032033618],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008151942,0.00016601018,0.11360291,0.0010986084,0.00011584318,0.0017277834,0.0183897,0.00589184,0.08958809,0.06274866,0.010097384,0.69575787],"study_design_scores_gemma":[0.00007943741,0.00048913603,0.22282332,0.001091899,0.00034809223,0.008572971,0.015284446,0.17451967,0.1808388,0.12498464,0.27050018,0.0004674414],"about_ca_topic_score_codex":0.00911006,"about_ca_topic_score_gemma":0.012878451,"teacher_disagreement_score":0.00911006,"about_ca_system_score_codex":0.0010385449,"about_ca_system_score_gemma":0.0017198274,"threshold_uncertainty_score":0.02287525},"labels":[],"label_agreement":null},{"id":"W2133932086","doi":"10.3115/1687878.1687897","title":"A ranking approach to stress prediction for letter-to-phoneme conversion","year":2009,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Stress (linguistics); Substring; Computer science; Speech recognition; Ranking (information retrieval); Pronunciation; Artificial intelligence; Word error rate; Support vector machine; Rank (graph theory); Sequence (biology); Speech synthesis; Natural language processing; Word (group theory); Pattern recognition (psychology); Mathematics","score_opus":0.013207481643641751,"score_gpt":0.2530259903314433,"score_spread":0.23981850868780155,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2133932086","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.121190615,0.00051607465,0.8582392,0.00043082776,0.0003157181,0.00017705114,0.0015531922,0.0127511965,0.004826129],"genre_scores_gemma":[0.59903586,0.00030809952,0.39124388,0.0001696571,0.00030172194,0.00017654912,0.0022352962,0.0003677642,0.0061610825],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992981,0.00017688594,0.00007744248,0.0002037658,0.00017214134,0.0000716797],"domain_scores_gemma":[0.9977043,0.00073039514,0.000224287,0.00028418258,0.0009643102,0.00009247305],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007880436,0.001104018,0.00075218646,0.0015986413,0.0006707515,0.0011790262,0.0009484735,0.0007875027,0.0032183898],"category_scores_gemma":[0.0043134177,0.00026034153,0.0004671971,0.0012789614,0.00027194867,0.0013805565,0.00054079195,0.0008789142,0.00400222],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00034796595,0.00020591031,0.0076678772,0.0001443361,0.000052362637,0.00024007121,0.00017010469,0.01565099,0.03977207,0.0017347104,0.007837459,0.92617613],"study_design_scores_gemma":[0.000038188722,0.00035716264,0.007644788,0.00003210242,0.000058753187,0.0004581412,0.00015004326,0.93991625,0.03660746,0.0097651305,0.0048842905,0.000087629865],"about_ca_topic_score_codex":0.0029859047,"about_ca_topic_score_gemma":0.005762888,"teacher_disagreement_score":0.0032183898,"about_ca_system_score_codex":0.00032630135,"about_ca_system_score_gemma":0.0006823419,"threshold_uncertainty_score":0.010766566},"labels":[],"label_agreement":null},{"id":"W2134068523","doi":"10.18653/v1/w14-0142","title":"plWordNet as the Cornerstone of a Toolkit of Lexico-semantic Resources","year":2014,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"WordNet; Lexical database; Computer science; Natural language processing; Lexicon; Artificial intelligence","score_opus":0.008617559582282101,"score_gpt":0.24907087585452317,"score_spread":0.24045331627224106,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2134068523","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004266001,0.0015603907,0.905634,0.0077021318,0.0010159069,0.000493715,0.006673829,0.02185714,0.05079678],"genre_scores_gemma":[0.041343406,0.002476594,0.8839496,0.0032289138,0.00041446413,0.0008314757,0.020245066,0.0070285588,0.04048195],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9974337,0.0008090465,0.00030405485,0.0006068868,0.00071203767,0.00013434766],"domain_scores_gemma":[0.9968947,0.0008868775,0.00017208538,0.0009176552,0.000790489,0.00033807976],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0042971615,0.0011657309,0.00091536343,0.00502298,0.0021467307,0.007362121,0.0024744633,0.0015820453,0.02472066],"category_scores_gemma":[0.007814011,0.0012127946,0.00098356,0.005449166,0.0022136096,0.023741867,0.006330022,0.0043997536,0.017201252],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018244219,0.0001066442,0.0010389278,0.00089114124,0.000068320536,0.0004759929,0.0014286808,0.003084053,0.0055674394,0.5291283,0.13726774,0.32076034],"study_design_scores_gemma":[0.000018267427,0.000026546044,0.00028358612,0.00031967595,0.00002132347,0.00024138117,0.00039252941,0.00849113,0.0038663864,0.22760181,0.75868547,0.00005194499],"about_ca_topic_score_codex":0.004716193,"about_ca_topic_score_gemma":0.009782888,"teacher_disagreement_score":0.02472066,"about_ca_system_score_codex":0.0019802717,"about_ca_system_score_gemma":0.0043606553,"threshold_uncertainty_score":0.08269888},"labels":[],"label_agreement":null},{"id":"W2134426638","doi":"10.3115/1654690.1654695","title":"A tree adjoining grammar analysis of the syntax and semantics of<i>it</i>-clefts","year":2006,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Syntax; Grammar; Semantics (computer science); Natural language processing; Tree (set theory); Linguistics; Affix grammar; Abstract syntax tree; Word grammar; Component (thermodynamics); Artificial intelligence; Programming language; Relational grammar; Mathematics; Emergent grammar; Generative grammar; Philosophy; Combinatorics","score_opus":0.006614922308287334,"score_gpt":0.23137771761823223,"score_spread":0.2247627953099449,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2134426638","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.058892358,0.00009395274,0.92557925,0.0005376794,0.000051713618,0.0000741227,0.0001406045,0.0006145307,0.014015869],"genre_scores_gemma":[0.7565073,0.00016970516,0.2382636,0.00020943722,0.000045576577,0.000081189704,0.00020603962,0.00036351805,0.0041537145],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9995239,0.00014636469,0.000040261042,0.00011967067,0.000115823656,0.000053955595],"domain_scores_gemma":[0.9991517,0.00029823522,0.00009053997,0.00016417704,0.00024247091,0.000052883774],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00082960783,0.00033383846,0.0003441274,0.0007931692,0.00096557965,0.0013838691,0.0007775579,0.00061912223,0.0038556273],"category_scores_gemma":[0.0015606215,0.00028023557,0.001042313,0.0008256237,0.0025050403,0.0037423559,0.00097451254,0.001047935,0.00038547957],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000008332011,0.0000066315274,0.00027324393,0.000022754446,0.0000041945236,0.00019569846,0.0008982195,0.0018142129,0.0025135647,0.98888165,0.00036556317,0.0050160005],"study_design_scores_gemma":[0.000012779368,0.00002647631,0.00041865045,0.0000241251,0.000030943756,0.00029989274,0.00046551376,0.04861893,0.003635001,0.9364798,0.009969924,0.000017970893],"about_ca_topic_score_codex":0.0028245489,"about_ca_topic_score_gemma":0.002085739,"teacher_disagreement_score":0.0038556273,"about_ca_system_score_codex":0.00085915683,"about_ca_system_score_gemma":0.0010140355,"threshold_uncertainty_score":0.012898326},"labels":[],"label_agreement":null},{"id":"W2134830997","doi":"10.7202/029804ar","title":"Constructing a Large-Scale English-Persian Parallel Corpus","year":2009,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"Payame Noor University; University of Warwick","keywords":"Persian; Computer science; Natural language processing; Machine translation; Task (project management); Artificial intelligence; The Internet; String (physics); Scale (ratio); Software; Corpus linguistics; Construct (python library); Linguistics; World Wide Web; Information retrieval; Programming language; Engineering; Mathematics","score_opus":0.018131837229962852,"score_gpt":0.2598321190240259,"score_spread":0.24170028179406305,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2134830997","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.34903318,0.0023269053,0.5380201,0.0016560673,0.001044882,0.0056257253,0.031964883,0.00889987,0.061428476],"genre_scores_gemma":[0.24801669,0.0008025315,0.67563796,0.00024628572,0.00022120272,0.0034217234,0.057460554,0.0014309877,0.012762053],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9981345,0.00076691114,0.00023313018,0.0004588102,0.0003239508,0.00008264035],"domain_scores_gemma":[0.99250364,0.0033310715,0.00029513304,0.001192849,0.0024633233,0.00021406692],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003318689,0.00082497176,0.00071809103,0.0044584046,0.0022833375,0.0015114025,0.0014538794,0.00070800324,0.012549971],"category_scores_gemma":[0.009396739,0.0006920755,0.00053501735,0.005708892,0.0012452186,0.0026265124,0.0022210449,0.0013495994,0.004244595],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015105989,0.0013052371,0.008328275,0.0033448504,0.0001818849,0.008514601,0.014080043,0.018837253,0.071003355,0.065730155,0.08707112,0.7200926],"study_design_scores_gemma":[0.0007388224,0.0009924687,0.023332559,0.00053444144,0.00037737444,0.0049339733,0.010756817,0.070779406,0.11364127,0.0450353,0.72860634,0.00027119095],"about_ca_topic_score_codex":0.0025692228,"about_ca_topic_score_gemma":0.0034459368,"teacher_disagreement_score":0.012549971,"about_ca_system_score_codex":0.0008809482,"about_ca_system_score_gemma":0.0023635218,"threshold_uncertainty_score":0.041983843},"labels":[],"label_agreement":null},{"id":"W2134855652","doi":"10.63317/2bwsvfeyt3ye","title":"Developing a TT-MCTAG for German with an RCG-based Parser","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Atomic Energy of Canada Limited; University of Pennsylvania","keywords":"Computer science; Natural language processing; Parsing; Attribute grammar; Head-driven phrase structure grammar; Grammar; L-attributed grammar; Operator-precedence grammar; Tree-adjoining grammar; Artificial intelligence; Programming language; Parse tree; Rule-based machine translation; German; Linguistics; Context-free grammar; Generative grammar","score_opus":0.03785148131464473,"score_gpt":0.30993173819051245,"score_spread":0.27208025687586773,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2134855652","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0133158155,0.0002298285,0.9573692,0.00041710972,0.00020931252,0.000316848,0.0022858898,0.017483262,0.008372757],"genre_scores_gemma":[0.12912765,0.0003634916,0.8570684,0.0002710666,0.0000619751,0.00019626049,0.0042278306,0.0032876083,0.005395694],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99943966,0.00012768064,0.000085673644,0.00016425521,0.00012945036,0.000053251813],"domain_scores_gemma":[0.99943906,0.00021448763,0.00004655124,0.00012075659,0.00015712393,0.000021930755],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008395687,0.0007909502,0.00056430156,0.0010818606,0.00054749387,0.0016218015,0.0010101796,0.0010490178,0.00610974],"category_scores_gemma":[0.0014828122,0.00076889095,0.0010578587,0.00093954144,0.000815101,0.0017962167,0.001287498,0.0012225128,0.0030322573],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00030457706,0.00014190155,0.0043249764,0.0014781547,0.00016573288,0.003928183,0.0037672746,0.030210685,0.11505703,0.4995076,0.043590073,0.29752377],"study_design_scores_gemma":[0.00014039851,0.0002938923,0.00388985,0.00041750286,0.00038107717,0.0051411823,0.0009198199,0.26160616,0.17506026,0.15123668,0.40063715,0.00027599727],"about_ca_topic_score_codex":0.002592545,"about_ca_topic_score_gemma":0.0035794063,"teacher_disagreement_score":0.00610974,"about_ca_system_score_codex":0.0007896827,"about_ca_system_score_gemma":0.0014689399,"threshold_uncertainty_score":0.020439148},"labels":[],"label_agreement":null},{"id":"W2135007042","doi":"10.63317/3nmcrj4muygh","title":"Corpus-based Semantic Relatedness for the Construction of Polish WordNet","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"WordNet; Computer science; Natural language processing; Artificial intelligence; Adjective; Semantic similarity; Verb; Lexicography; Quality (philosophy); Information retrieval; Noun; Linguistics","score_opus":0.017346370211470112,"score_gpt":0.25486235868474905,"score_spread":0.23751598847327893,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2135007042","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05848061,0.00043155873,0.9286794,0.0002550795,0.0000786757,0.00081550237,0.002455614,0.0023647863,0.0064388006],"genre_scores_gemma":[0.1247553,0.00033487033,0.8651132,0.000029953784,0.000030762825,0.0012243191,0.006873607,0.00035871175,0.00127935],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99691355,0.0014351795,0.00048385805,0.00048342542,0.00059611554,0.00008793284],"domain_scores_gemma":[0.9957088,0.0018321913,0.00045832066,0.0008749256,0.00097149774,0.00015425676],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037641174,0.0006685016,0.00067930983,0.0064309933,0.001977232,0.0027145296,0.00091309124,0.0006823936,0.006572118],"category_scores_gemma":[0.01342372,0.0006941428,0.00058683695,0.0053720586,0.0014806724,0.0044777566,0.00323927,0.0011338758,0.0028262334],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00050085655,0.00029904788,0.01268897,0.0021769407,0.0001581875,0.0009780234,0.006906016,0.012181911,0.06782864,0.22894591,0.013385931,0.65394956],"study_design_scores_gemma":[0.00027129558,0.00080879126,0.04351795,0.0013164323,0.00035123463,0.0033415915,0.008276391,0.30033427,0.082973294,0.27387834,0.2845281,0.00040225722],"about_ca_topic_score_codex":0.0024651168,"about_ca_topic_score_gemma":0.0048402664,"teacher_disagreement_score":0.006572118,"about_ca_system_score_codex":0.0012084874,"about_ca_system_score_gemma":0.0023573101,"threshold_uncertainty_score":0.021985948},"labels":[],"label_agreement":null},{"id":"W2135336649","doi":"","title":"Text-level Discourse Parsing with Rich Linguistic Features","year":2012,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":150,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Parsing; Computer science; Natural language processing; Artificial intelligence; Sentence; Top-down parsing; Parser combinator; Bottom-up parsing; Style (visual arts); Linguistics; Tree (set theory); S-attributed grammar; History; Mathematics","score_opus":0.020050209894535688,"score_gpt":0.30130477174709275,"score_spread":0.28125456185255704,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2135336649","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016477428,0.00038962052,0.9502815,0.0006958416,0.00018339753,0.00022747503,0.0020723024,0.025297048,0.004375337],"genre_scores_gemma":[0.101086326,0.0002603088,0.8852627,0.00033489784,0.00019281001,0.00020668813,0.004938338,0.0023789033,0.005338982],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9979336,0.0008662943,0.00019970273,0.00044517036,0.0004622669,0.00009283807],"domain_scores_gemma":[0.98906016,0.006393722,0.00056427484,0.0016106407,0.0021795924,0.00019153343],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027808442,0.0015802732,0.0010390214,0.0015548903,0.00081154995,0.0027275498,0.0020141294,0.001476343,0.010064433],"category_scores_gemma":[0.013227863,0.000944467,0.0009928675,0.0013090614,0.0006831624,0.0058205226,0.0022818076,0.0027175986,0.0066504055],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00058504974,0.00049013767,0.0039130957,0.0023270915,0.0002024382,0.0012110985,0.0028997478,0.026848674,0.23937534,0.048060667,0.04671107,0.62737566],"study_design_scores_gemma":[0.00011735376,0.00026160185,0.0029768276,0.00032907532,0.00030578024,0.0010171103,0.000642607,0.4753493,0.32407278,0.057349138,0.1373064,0.0002720723],"about_ca_topic_score_codex":0.0012053112,"about_ca_topic_score_gemma":0.0022634722,"teacher_disagreement_score":0.010064433,"about_ca_system_score_codex":0.00068737933,"about_ca_system_score_gemma":0.0013825782,"threshold_uncertainty_score":0.033668876},"labels":[],"label_agreement":null},{"id":"W2135922901","doi":"10.1109/icassp.1985.1168119","title":"Multi-speaker computer recognition of ten connectedly spoken letters","year":2005,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Speech recognition; Computer science; Set (abstract data type); Speaker recognition; Natural language processing; Artificial intelligence; Linguistics; Programming language","score_opus":0.01921137036099671,"score_gpt":0.25677111874282443,"score_spread":0.23755974838182772,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2135922901","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.89577895,0.00016937952,0.097084984,0.00006689574,0.00008351726,0.00017410942,0.00068218226,0.0022348515,0.0037250705],"genre_scores_gemma":[0.91827565,0.00007907089,0.07536309,0.0000400672,0.00002690059,0.00011055768,0.0011736553,0.0001123512,0.0048184656],"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99964976,0.00007491372,0.0000150860615,0.00014253866,0.000082016064,0.00003561268],"domain_scores_gemma":[0.9993231,0.00032617102,0.000036678077,0.00007942391,0.00015741179,0.00007717972],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00035495742,0.00040279518,0.00047198776,0.00032719882,0.0003249816,0.0003895563,0.0004658423,0.0004405412,0.0051908577],"category_scores_gemma":[0.0011992325,0.00015507948,0.00020494415,0.0002421945,0.00021313349,0.0003558129,0.0003958164,0.0003642689,0.0014344539],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014894301,0.00036316327,0.0074977363,0.00033372373,0.000117962976,0.00065720465,0.0010442577,0.0035893023,0.4918813,0.0005168265,0.001956277,0.49055275],"study_design_scores_gemma":[0.00014729923,0.0022422771,0.1116573,0.00003096896,0.00026925717,0.0030108534,0.001310232,0.20246352,0.66925573,0.0012223704,0.0082271695,0.00016297246],"about_ca_topic_score_codex":0.0016548613,"about_ca_topic_score_gemma":0.0031386826,"teacher_disagreement_score":0.0051908577,"about_ca_system_score_codex":0.00018617605,"about_ca_system_score_gemma":0.00024765305,"threshold_uncertainty_score":0.017365098},"labels":[],"label_agreement":null},{"id":"W2136161594","doi":"10.7939/r3-e9za-6468","title":"Learning structured classifiers for statistical dependency parsing","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Dependency grammar; Artificial intelligence; Parsing; Trigram; Machine learning; Discriminative model; Supervised learning; Natural language processing; Semi-supervised learning; Margin (machine learning); Classifier (UML); Bottom-up parsing; Top-down parsing; Artificial neural network","score_opus":0.022743954310777414,"score_gpt":0.2868017065484175,"score_spread":0.2640577522376401,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2136161594","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0031163583,0.00025793776,0.9940217,0.0001878145,0.000055489116,0.0000684913,0.0002536866,0.0012760708,0.00076238846],"genre_scores_gemma":[0.1230681,0.00054956943,0.8681011,0.0004574496,0.00030265044,0.0007438071,0.003620086,0.00030199264,0.0028551205],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99751604,0.0010835674,0.00016819096,0.00064509304,0.00043841376,0.00014878869],"domain_scores_gemma":[0.98970455,0.007349712,0.00043431527,0.0011814112,0.0011275727,0.0002024932],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0041959076,0.0012457821,0.0017884005,0.0026262826,0.000986338,0.0018601613,0.0027120356,0.002369147,0.005640464],"category_scores_gemma":[0.017255541,0.0010955203,0.0017350609,0.0020720689,0.00090470014,0.004880496,0.0020983347,0.0039553996,0.004509532],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019568815,0.00027348078,0.0029800627,0.00029414505,0.00018173289,0.00018007135,0.00023216654,0.2043981,0.003857564,0.09139022,0.026502887,0.66951394],"study_design_scores_gemma":[0.000017048897,0.000039930546,0.00019766875,0.000034946916,0.000019030378,0.00003183146,0.00002065994,0.89466435,0.0012404397,0.10140088,0.0023207073,0.000012517164],"about_ca_topic_score_codex":0.0012465543,"about_ca_topic_score_gemma":0.0024519593,"teacher_disagreement_score":0.005640464,"about_ca_system_score_codex":0.0012086233,"about_ca_system_score_gemma":0.0014941358,"threshold_uncertainty_score":0.022190392},"labels":[],"label_agreement":null},{"id":"W2136180489","doi":"","title":"Phrase Clustering for Smoothing TM Probabilities - or, How to Extract Paraphrases from Phrase Tables","year":2010,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"University of Waterloo; National Research Council Canada","funders":"","keywords":"Phrase; Smoothing; Computer science; Cluster analysis; Artificial intelligence; Natural language processing; Sentence; Machine translation; Language model; Feature (linguistics); Translation (biology); Linguistics","score_opus":0.022167129957115835,"score_gpt":0.28100102371870755,"score_spread":0.2588338937615917,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2136180489","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0011011385,0.000079690595,0.99444026,0.000059215374,0.00004641476,0.000050137623,0.00020018169,0.0035368558,0.00048605507],"genre_scores_gemma":[0.023815045,0.00017422976,0.9718365,0.00008040073,0.00007503147,0.00018538687,0.00092770293,0.0013033312,0.0016023063],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99843997,0.000450959,0.00015630735,0.00043466597,0.0004317468,0.00008639128],"domain_scores_gemma":[0.9963558,0.0014590371,0.00026425568,0.0011508273,0.00071042415,0.000059768594],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002817456,0.00142932,0.0007188195,0.0026384047,0.0010835386,0.0015333483,0.0016470491,0.0013839139,0.016483715],"category_scores_gemma":[0.014767123,0.0008316443,0.0012883929,0.0030071936,0.000711066,0.0036544236,0.0015840137,0.0025741877,0.012742357],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001733062,0.000075672455,0.0010051519,0.000339131,0.00013125224,0.00012475731,0.0003759452,0.01680701,0.031863045,0.020945042,0.022976201,0.9051835],"study_design_scores_gemma":[0.00011057756,0.00014482511,0.0035826946,0.00013526401,0.00017971356,0.0008581545,0.0002696395,0.722709,0.10165908,0.08762341,0.08253083,0.00019674569],"about_ca_topic_score_codex":0.0030920194,"about_ca_topic_score_gemma":0.00562531,"teacher_disagreement_score":0.016483715,"about_ca_system_score_codex":0.0005847249,"about_ca_system_score_gemma":0.0012019406,"threshold_uncertainty_score":0.055143535},"labels":[],"label_agreement":null},{"id":"W2136212246","doi":"10.1007/s00012-004-1902-0","title":"Automated discovery of single axioms for ortholattices","year":2005,"lang":"en","type":"article","venue":"Algebra Universalis","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba","funders":"","keywords":"Mathematical proof; Axiom; Mathematics; Algebra over a field; Discrete mathematics; Theoretical computer science; Computer science; Pure mathematics","score_opus":0.012195829551426604,"score_gpt":0.2549839530227904,"score_spread":0.2427881234713638,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2136212246","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07325939,0.0005202938,0.9023525,0.0015852764,0.00025688176,0.00033814358,0.0022673998,0.009054535,0.010365568],"genre_scores_gemma":[0.39030486,0.00039054593,0.59790874,0.00036752815,0.00013826994,0.0001546348,0.0044765403,0.0011938978,0.0050650486],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9952518,0.0007522501,0.0005120749,0.001457303,0.0015579195,0.0004686412],"domain_scores_gemma":[0.9838411,0.008904009,0.00076507556,0.003646918,0.002413495,0.0004294389],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028281033,0.00094331766,0.001662887,0.0033306265,0.002552347,0.004701281,0.0034266505,0.0017171524,0.008833667],"category_scores_gemma":[0.017561419,0.0015654356,0.003694585,0.0020456898,0.002442045,0.012627791,0.005933302,0.0040268362,0.002660251],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036501998,0.00037443367,0.008910662,0.0011663031,0.0003189608,0.0012191377,0.0015396677,0.009536612,0.0184929,0.619989,0.019529926,0.31855753],"study_design_scores_gemma":[0.00007473326,0.00005822375,0.0014815672,0.00017741627,0.00024661116,0.0007252541,0.00084049144,0.12268841,0.038528677,0.80785596,0.027225459,0.00009715677],"about_ca_topic_score_codex":0.0025911643,"about_ca_topic_score_gemma":0.008372904,"teacher_disagreement_score":0.008833667,"about_ca_system_score_codex":0.0016038541,"about_ca_system_score_gemma":0.0033259557,"threshold_uncertainty_score":0.029551566},"labels":[],"label_agreement":null},{"id":"W2136433925","doi":"10.1017/s1351324903003231","title":"Surface-marker-based dialog modelling: A progress report on the MAREDI project","year":2003,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa; Université Laval; Université du Québec à Trois-Rivières","funders":"","keywords":"Computer science; Dialog box; Natural language processing; Connectionism; Semantics (computer science); Conversation; Spoken language; Artificial intelligence; Natural language; Natural language understanding; Natural language generation; Programming language; Linguistics; Artificial neural network; World Wide Web","score_opus":0.01414064130185601,"score_gpt":0.25566720439021334,"score_spread":0.24152656308835732,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2136433925","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03500116,0.0035100328,0.9349026,0.0010096027,0.00017948388,0.0006575201,0.001243248,0.014041256,0.009455197],"genre_scores_gemma":[0.13023801,0.0026641057,0.85166967,0.00018959084,0.0001872902,0.00043092112,0.0034290235,0.0014295374,0.0097617535],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9970733,0.0011696123,0.00015383204,0.0007974995,0.00068869075,0.000117154086],"domain_scores_gemma":[0.9958973,0.0016654773,0.00019415256,0.001286248,0.00072965934,0.00022729753],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0052035097,0.0015665464,0.0014762776,0.0017120636,0.00060506247,0.0036957436,0.0039170315,0.0017609124,0.006767148],"category_scores_gemma":[0.009961942,0.0011777737,0.0013175085,0.0008961253,0.0012373362,0.0069303336,0.0020828112,0.0022872994,0.0030205073],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00090643734,0.0009860353,0.003115988,0.001627226,0.00022613545,0.00038540852,0.002563362,0.108362615,0.055373468,0.08220652,0.017300418,0.7269464],"study_design_scores_gemma":[0.00014389287,0.0003798252,0.0023206018,0.00021521507,0.00014015628,0.0004373496,0.0003914485,0.8023509,0.04921948,0.018922288,0.12527758,0.00020135191],"about_ca_topic_score_codex":0.008152458,"about_ca_topic_score_gemma":0.0032547617,"teacher_disagreement_score":0.008152458,"about_ca_system_score_codex":0.0014875556,"about_ca_system_score_gemma":0.002075558,"threshold_uncertainty_score":0.027519166},"labels":[],"label_agreement":null},{"id":"W2136930489","doi":"10.1162/coli.2006.32.1.13","title":"Evaluating WordNet-based Measures of Lexical Semantic Relatedness","year":2006,"lang":"en","type":"article","venue":"Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1414,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"University of Toronto; University of Pennsylvania","keywords":"WordNet; Computer science; Semantic similarity; Natural language processing; Spelling; Artificial intelligence; Lexical database; Proxy (statistics); Similarity (geometry); Information retrieval; Measure (data warehouse); Linguistics; Machine learning; Data mining","score_opus":0.058731127994592104,"score_gpt":0.34696039051580496,"score_spread":0.28822926252121284,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2136930489","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9230577,0.0030598964,0.06271193,0.00023718434,0.0001257142,0.00025165593,0.001829505,0.0007901886,0.007936302],"genre_scores_gemma":[0.9472552,0.0006117703,0.04735991,0.000035935234,0.00008944846,0.00015330692,0.0036365245,0.00009555333,0.0007623129],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9917888,0.0035219316,0.0010666893,0.0009157127,0.002435959,0.00027095395],"domain_scores_gemma":[0.9540678,0.032822773,0.004065515,0.0026619502,0.005328716,0.0010531364],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0080470685,0.001054583,0.0012850834,0.016808625,0.00081500824,0.0018349051,0.0011535381,0.0015792699,0.0015628039],"category_scores_gemma":[0.039030224,0.000245351,0.00060670596,0.009133915,0.0011374116,0.00484967,0.0019232524,0.00065861485,0.00063508534],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002853965,0.0019656005,0.29271787,0.001979236,0.0019126602,0.00062227773,0.0020963394,0.09737049,0.031086773,0.014920579,0.0054962793,0.54697794],"study_design_scores_gemma":[0.00038422164,0.004384023,0.26307544,0.00020064371,0.0007751023,0.0017712394,0.0020961978,0.6443692,0.045145057,0.031381596,0.0060375365,0.0003797344],"about_ca_topic_score_codex":0.0018475588,"about_ca_topic_score_gemma":0.0029943145,"teacher_disagreement_score":0.016808625,"about_ca_system_score_codex":0.0008571837,"about_ca_system_score_gemma":0.00075455196,"threshold_uncertainty_score":0.042557478},"labels":[],"label_agreement":null},{"id":"W2137442651","doi":"","title":"No Sentence Is Too Confusing To Ignore","year":2010,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Noun phrase; Linguistics; Adjective; Sentence; Phrase; Verb phrase; Verb; Determiner phrase; Noun; Meaning (existential); Context (archaeology); Natural language processing; Computer science; Semantic role labeling; Artificial intelligence; Psychology; Philosophy; History","score_opus":0.01016808889002838,"score_gpt":0.2739456178297179,"score_spread":0.2637775289396895,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2137442651","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05479122,0.0068345997,0.14294042,0.18191613,0.02325533,0.00041055554,0.0037799233,0.0044376,0.5816343],"genre_scores_gemma":[0.6360956,0.0054351734,0.07004184,0.100762896,0.016463842,0.00045489267,0.004636667,0.00584633,0.16026281],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99515283,0.0022192441,0.00026492617,0.000634614,0.0015109825,0.00021739863],"domain_scores_gemma":[0.98695576,0.0060696746,0.0010411218,0.0017698836,0.0036066596,0.0005568601],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031684027,0.0010802607,0.00096740114,0.001306073,0.0029937057,0.0044882824,0.0013086605,0.002373184,0.015173514],"category_scores_gemma":[0.029487992,0.00056855043,0.00046446867,0.00096693897,0.0037135843,0.011801614,0.0039879405,0.003333119,0.012440992],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00046107796,0.000105607025,0.003205407,0.0011447612,0.00016050262,0.0014578803,0.009482312,0.0005109546,0.0067573227,0.45056686,0.37907368,0.14707366],"study_design_scores_gemma":[0.000026033418,0.00007461165,0.0012351082,0.00035368072,0.00007925707,0.0016349345,0.0029181305,0.0017352137,0.0035096118,0.23555489,0.7527994,0.0000791939],"about_ca_topic_score_codex":0.0013631722,"about_ca_topic_score_gemma":0.0025021192,"teacher_disagreement_score":0.015173514,"about_ca_system_score_codex":0.0012296753,"about_ca_system_score_gemma":0.0012958624,"threshold_uncertainty_score":0.050760508},"labels":[],"label_agreement":null},{"id":"W213769128","doi":"","title":"Classifying pluricentric languages: Extending the monolingual model","year":2012,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Natural language processing; Portuguese; Artificial intelligence; Identification (biology); Class (philosophy); Language model; Character (mathematics); Word (group theory); Linguistics; Mathematics","score_opus":0.02628952084296893,"score_gpt":0.30868423107883797,"score_spread":0.28239471023586904,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W213769128","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3898648,0.00083804707,0.58724165,0.00069650257,0.000112861126,0.0002477429,0.00086136605,0.0017736633,0.018363446],"genre_scores_gemma":[0.88479596,0.0003207993,0.1061046,0.00014856948,0.000087328415,0.00009951442,0.0012731245,0.00021521348,0.006954873],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985905,0.00056297786,0.0000867333,0.0003942923,0.00024716195,0.00011824055],"domain_scores_gemma":[0.99819404,0.0006438792,0.00016182533,0.00029096188,0.00060170813,0.00010754354],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017271816,0.0009569834,0.0005247321,0.0019317376,0.0010070312,0.0020586122,0.0008516321,0.0006164514,0.002681112],"category_scores_gemma":[0.003863182,0.00019025563,0.0008232049,0.0010781649,0.00059321267,0.0027611377,0.0013541114,0.0009239061,0.0019020195],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007754526,0.0005824406,0.08072588,0.00031875086,0.0003051615,0.00094238156,0.0030197997,0.06895101,0.02056146,0.018565986,0.0046205646,0.80063117],"study_design_scores_gemma":[0.000029850431,0.00027035456,0.013684678,0.000072368464,0.0001448476,0.00072083273,0.0013955408,0.94461125,0.008322323,0.01644826,0.014215585,0.000084031955],"about_ca_topic_score_codex":0.0130571425,"about_ca_topic_score_gemma":0.01974038,"teacher_disagreement_score":0.0130571425,"about_ca_system_score_codex":0.0008260643,"about_ca_system_score_gemma":0.0012606118,"threshold_uncertainty_score":0.025962293},"labels":[],"label_agreement":null},{"id":"W2137766642","doi":"","title":"Solving Substitution Ciphers with Combined Language Models","year":2014,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Benchmark (surveying); Suite; Decipherment; Cryptography; Tree (set theory); Substitution (logic); Cipher; Artificial intelligence; Theoretical computer science; Encryption; Algorithm; Programming language; Computer security; Mathematics","score_opus":0.008981215167887253,"score_gpt":0.23263036424876365,"score_spread":0.2236491490808764,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2137766642","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.037146665,0.00025530806,0.95941997,0.00032269055,0.000036462865,0.000045332832,0.00012377072,0.0009707542,0.0016790698],"genre_scores_gemma":[0.42167053,0.0003241786,0.5708201,0.0002818806,0.00010390453,0.0002015939,0.0006143468,0.00034160507,0.0056419475],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9991911,0.00030216694,0.00006889162,0.00015992016,0.00017614587,0.00010175077],"domain_scores_gemma":[0.9972783,0.002014272,0.00017931535,0.00027161386,0.00019819553,0.000058348287],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012126046,0.00092413457,0.0011543796,0.0005840657,0.0004513418,0.0012865872,0.0011027656,0.0018290317,0.0033914205],"category_scores_gemma":[0.004886623,0.0006213712,0.0013327142,0.00068083877,0.0011137261,0.0033897478,0.0017705044,0.002013242,0.0013009772],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029350055,0.0001465583,0.0010494263,0.00027254317,0.00009916376,0.00033015662,0.00020910127,0.8158017,0.007740674,0.06098318,0.0028443276,0.11022964],"study_design_scores_gemma":[0.000024644605,0.000037293492,0.000026027681,0.0000056483523,0.000008934508,0.00004058201,0.000020527752,0.96998674,0.0023487753,0.026904972,0.0005904515,0.0000054823554],"about_ca_topic_score_codex":0.0018026043,"about_ca_topic_score_gemma":0.0032856665,"teacher_disagreement_score":0.0033914205,"about_ca_system_score_codex":0.00063752226,"about_ca_system_score_gemma":0.001597319,"threshold_uncertainty_score":0.0113453865},"labels":[],"label_agreement":null},{"id":"W2137926008","doi":"10.1017/s1351324905004043","title":"Can syllabification improve pronunciation by analogy of English?","year":2006,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":38,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada; National Research Council Institute for Biodiagnostics","funders":"","keywords":"Syllabification; Pronunciation; Computer science; Syllable; Natural language processing; Analogy; Artificial intelligence; Speech recognition; Word (group theory); Linguistics","score_opus":0.0016253391825071673,"score_gpt":0.19236527812346438,"score_spread":0.1907399389409572,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2137926008","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3975843,0.0032571263,0.5695746,0.0025987276,0.00069905655,0.00014657396,0.00033631135,0.00718846,0.018614918],"genre_scores_gemma":[0.8140182,0.0005376015,0.18198249,0.000415823,0.000103775674,0.000049748818,0.00036084594,0.00024891165,0.0022826544],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99850684,0.00068921404,0.00007182569,0.00046341968,0.00017553741,0.00009319487],"domain_scores_gemma":[0.9940608,0.00391124,0.00027649713,0.0010301949,0.0005707787,0.00015044548],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002164605,0.0007957246,0.0008982863,0.00085210905,0.00061301305,0.0015083201,0.0012912552,0.001068291,0.0037043004],"category_scores_gemma":[0.018504255,0.00040683014,0.0008585754,0.00074695697,0.00056700374,0.0051240255,0.0014417521,0.0014874849,0.0020234354],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007694302,0.00028388857,0.019534815,0.00029020727,0.00013305894,0.0002461776,0.0009810155,0.021254595,0.032567434,0.016729238,0.002895321,0.9043148],"study_design_scores_gemma":[0.00015623667,0.0010513826,0.023171922,0.0001496012,0.00028511524,0.0011123723,0.000998642,0.7651165,0.062378626,0.1202098,0.02514736,0.00022246914],"about_ca_topic_score_codex":0.0034749762,"about_ca_topic_score_gemma":0.0040370957,"teacher_disagreement_score":0.0037043004,"about_ca_system_score_codex":0.00060613925,"about_ca_system_score_gemma":0.0010311778,"threshold_uncertainty_score":0.012392163},"labels":[],"label_agreement":null},{"id":"W2138265376","doi":"","title":"Text Deixis in Narrative Sequences","year":2007,"lang":"en","type":"article","venue":"International journal of english studies, Vol","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":31,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Deixis; Demonstrative; Linguistics; Utterance; Noun; Cohesion (chemistry); Narrative; Text linguistics; Computer science; Proper noun; Pronoun; Natural language processing; Artificial intelligence; Philosophy; Physics","score_opus":0.020700465076302455,"score_gpt":0.34173245599026936,"score_spread":0.3210319909139669,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2138265376","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6859741,0.0041112755,0.21266651,0.0013155116,0.00034286597,0.0005877443,0.002389958,0.0011920425,0.09142006],"genre_scores_gemma":[0.93047667,0.0006562513,0.049159132,0.0000725905,0.000059249745,0.00018851504,0.0017300211,0.00020145325,0.01745618],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9984498,0.00084251136,0.000101284866,0.00029657967,0.0002357638,0.000074052215],"domain_scores_gemma":[0.99649066,0.002125308,0.00054061064,0.0003108563,0.00041180884,0.000120660705],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009920454,0.0004119349,0.00024185322,0.0012578316,0.0011430308,0.0015212847,0.00047134937,0.0004008699,0.0057441513],"category_scores_gemma":[0.005598737,0.00025169383,0.00018186666,0.0012040439,0.0011658008,0.0028644851,0.0014923704,0.00053782866,0.0007530137],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008871128,0.00014059205,0.01042459,0.0021606104,0.00006543608,0.0040775556,0.15646681,0.003347234,0.05986835,0.40370217,0.008567422,0.35029215],"study_design_scores_gemma":[0.00013428558,0.00045920236,0.032706946,0.0009593714,0.00011777775,0.004709055,0.053618513,0.023337357,0.066703826,0.10361824,0.713493,0.00014237022],"about_ca_topic_score_codex":0.0015642041,"about_ca_topic_score_gemma":0.0021949343,"teacher_disagreement_score":0.0057441513,"about_ca_system_score_codex":0.0012658245,"about_ca_system_score_gemma":0.00053864985,"threshold_uncertainty_score":0.01921612},"labels":[],"label_agreement":null},{"id":"W2138859780","doi":"10.1145/2207676.2207682","title":"\"Then click ok!\"","year":2012,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Documentation; Computer science; Classifier (UML); Artificial intelligence; Natural language processing; Matching (statistics); Information retrieval; Programming language","score_opus":0.015738212009716153,"score_gpt":0.27964300926815905,"score_spread":0.2639047972584429,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2138859780","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06498517,0.0016233111,0.33362964,0.011957873,0.005918671,0.0010653165,0.0121491775,0.1734677,0.39520305],"genre_scores_gemma":[0.2665909,0.0015290206,0.20666577,0.0073446096,0.0009423428,0.0007934902,0.013653351,0.017638717,0.48484188],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99935085,0.00016945125,0.00004087964,0.00013635165,0.00021622256,0.000086223874],"domain_scores_gemma":[0.9980038,0.0008410615,0.00014252262,0.0002837192,0.00053081947,0.0001980475],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008442517,0.0011760029,0.000515733,0.00091779476,0.0010523134,0.0015773636,0.0009841247,0.0014308923,0.19114651],"category_scores_gemma":[0.0068857693,0.00033772932,0.00037898007,0.00036787838,0.00037105245,0.0029558784,0.0013452651,0.0011350979,0.13852198],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006628247,0.00013282843,0.004077566,0.00033922194,0.000027547752,0.0005801043,0.0011624536,0.00027544022,0.008655796,0.0055513144,0.600427,0.37810788],"study_design_scores_gemma":[0.000044654,0.00020752448,0.0060656713,0.00026031237,0.000049348673,0.0011234342,0.0009203962,0.00982115,0.015862161,0.005167257,0.9603293,0.00014889863],"about_ca_topic_score_codex":0.0018637851,"about_ca_topic_score_gemma":0.0034832123,"teacher_disagreement_score":0.19114651,"about_ca_system_score_codex":0.00029463728,"about_ca_system_score_gemma":0.00034027878,"threshold_uncertainty_score":0.63944876},"labels":[],"label_agreement":null},{"id":"W2139635790","doi":"","title":"Improved Reordering for Shallow-n Grammar based Hierarchical Phrase-based Translation","year":2012,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Translation (biology); Concatenation (mathematics); Phrase; Rule-based machine translation; Machine translation; Decoding methods; Algorithm; Artificial intelligence; Synchronous context-free grammar; Natural language processing; Example-based machine translation; Mathematics; Arithmetic","score_opus":0.024630894033012055,"score_gpt":0.28391671734499613,"score_spread":0.2592858233119841,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2139635790","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0078085293,0.00009199239,0.9871896,0.00012006299,0.000034824625,0.0000432843,0.00019160376,0.0025254397,0.0019946273],"genre_scores_gemma":[0.29982588,0.00028604487,0.6904081,0.0002834357,0.00004815078,0.00016466869,0.0010154556,0.0016073415,0.006360831],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99934393,0.00023233023,0.00007246347,0.00013417177,0.00017606499,0.000041083134],"domain_scores_gemma":[0.99883014,0.00046356994,0.000081891005,0.0003664057,0.00022304611,0.000034892444],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010570901,0.0007838095,0.00079248054,0.000539022,0.00045395293,0.0008098774,0.000986362,0.00068907975,0.0048250323],"category_scores_gemma":[0.002701234,0.00051597314,0.0007893152,0.0008009891,0.0008637366,0.0019671847,0.0013232575,0.0013876052,0.0034820624],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00031282002,0.0001831105,0.0017035108,0.0005329539,0.0000792799,0.00057022984,0.00084391923,0.23797931,0.10932252,0.14510906,0.013592514,0.48977086],"study_design_scores_gemma":[0.000031614858,0.000063065585,0.0002673712,0.000026785112,0.000027889,0.00019188251,0.0000346094,0.9041969,0.024646604,0.06306954,0.0074022673,0.000041501025],"about_ca_topic_score_codex":0.003325823,"about_ca_topic_score_gemma":0.007120138,"teacher_disagreement_score":0.0048250323,"about_ca_system_score_codex":0.0007582436,"about_ca_system_score_gemma":0.0013942204,"threshold_uncertainty_score":0.016141295},"labels":[],"label_agreement":null},{"id":"W2140091071","doi":"10.7202/004626ar","title":"Studying Style in Simultaneous Interpretation","year":2002,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":43,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Interpreter; Interpretation (philosophy); Computer science; Style (visual arts); Divergence (linguistics); Consistency (knowledge bases); Representation (politics); Linguistics; Fluency; Language interpretation; Natural language processing; Artificial intelligence; Programming language; Philosophy; History","score_opus":0.03301356602808644,"score_gpt":0.27610930353182994,"score_spread":0.2430957375037435,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2140091071","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.37467018,0.0010097394,0.5349571,0.00064126885,0.00013308383,0.00029100824,0.00024243566,0.00055813056,0.08749697],"genre_scores_gemma":[0.89697856,0.00027946616,0.09660154,0.00006784153,0.00010754527,0.0002350396,0.00014964168,0.000232611,0.005347681],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"qualitative","domain_scores_codex":[0.99032,0.0050244005,0.0005291918,0.0012543004,0.002401374,0.00047062492],"domain_scores_gemma":[0.9610463,0.025911437,0.004673977,0.0042150784,0.003191404,0.0009618204],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00724603,0.00087526004,0.00060918275,0.005743261,0.0014210457,0.0064264764,0.0012521102,0.0012933316,0.0050625843],"category_scores_gemma":[0.04138557,0.0006081443,0.0009887434,0.004023359,0.0061848955,0.010747323,0.0050273975,0.0022545068,0.0007535456],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007344855,0.00029319813,0.059146088,0.0008962696,0.00015002613,0.0011676484,0.23127373,0.00696887,0.028497048,0.4169041,0.0014098355,0.25255883],"study_design_scores_gemma":[0.0001245926,0.0008274252,0.053468604,0.00034396988,0.00015896627,0.0019386457,0.04094328,0.04637315,0.017472064,0.8080535,0.030075423,0.00022031377],"about_ca_topic_score_codex":0.00058390567,"about_ca_topic_score_gemma":0.0007178762,"teacher_disagreement_score":0.00724603,"about_ca_system_score_codex":0.0013415369,"about_ca_system_score_gemma":0.00067639153,"threshold_uncertainty_score":0.038321137},"labels":[],"label_agreement":null},{"id":"W2140266967","doi":"10.1016/j.tcs.2014.02.006","title":"Graph transformation for incremental natural language analysis","year":2014,"lang":"en","type":"article","venue":"Theoretical Computer Science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Transformation (genetics); Graph rewriting; Computer science; Graph; Mathematics; Programming language; Theoretical computer science","score_opus":0.0046900244990736,"score_gpt":0.26367150925723376,"score_spread":0.2589814847581602,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2140266967","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009243397,0.000141282,0.9734177,0.00022362429,0.000061426814,0.00019441993,0.0007900748,0.0127627505,0.0031652928],"genre_scores_gemma":[0.19768795,0.0002644692,0.78662074,0.00027984552,0.0000834832,0.00033624005,0.00447604,0.0027891353,0.00746206],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99901915,0.00023569226,0.00007002715,0.00028109926,0.00030306203,0.000090901354],"domain_scores_gemma":[0.9974038,0.0014027882,0.00008605769,0.0006482926,0.00040583397,0.000053239364],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00054798357,0.00061382743,0.0005871332,0.0020044444,0.00076408446,0.0010933616,0.0014902764,0.0005914919,0.009058533],"category_scores_gemma":[0.0037336817,0.00049699814,0.0018175971,0.001645788,0.0009105392,0.0029170092,0.001772162,0.0017866895,0.00299534],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035730161,0.00035090727,0.0012921677,0.00057060627,0.000120740726,0.00045232935,0.0005825804,0.025523303,0.025431763,0.19276308,0.028995533,0.72355974],"study_design_scores_gemma":[0.000080449536,0.00008355167,0.0007587498,0.00006528872,0.00017324348,0.0002708934,0.00020449438,0.3974074,0.037190747,0.5271233,0.036586076,0.00005585836],"about_ca_topic_score_codex":0.005538134,"about_ca_topic_score_gemma":0.008598479,"teacher_disagreement_score":0.009058533,"about_ca_system_score_codex":0.0007815898,"about_ca_system_score_gemma":0.0012667659,"threshold_uncertainty_score":0.030303836},"labels":[],"label_agreement":null},{"id":"W2140330079","doi":"10.7202/011001ar","title":"A New Computational Tool for Analyzing Translation Processes: The TransCorrect Project","year":2005,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Translation (biology); Context (archaeology); Computational linguistics; Natural language processing; Machine translation; Translation studies; Artificial intelligence; Linguistics; Philosophy","score_opus":0.04488124861161734,"score_gpt":0.3118918767739635,"score_spread":0.26701062816234616,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2140330079","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0074949516,0.00020517148,0.9727183,0.00011509571,0.000056450037,0.00015034895,0.0012380198,0.016062636,0.001958893],"genre_scores_gemma":[0.04647301,0.00019523255,0.9452679,0.00006162065,0.00004708588,0.00045342656,0.003219421,0.0023670108,0.0019152794],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9961389,0.0012449152,0.0003710144,0.0012013366,0.00091962086,0.00012433968],"domain_scores_gemma":[0.9936724,0.0032359418,0.00057347026,0.0015587067,0.00081393775,0.00014558002],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00325686,0.002239146,0.0012407115,0.0052230433,0.0013389501,0.0032666388,0.0016855194,0.0012340677,0.0121507],"category_scores_gemma":[0.012341295,0.00086092524,0.0019870582,0.0042008515,0.0015818863,0.0045138495,0.0034444944,0.0019707652,0.0034814517],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007697782,0.00030292457,0.006547893,0.0013789252,0.00059930433,0.0010149324,0.0029930803,0.033828657,0.027942058,0.07000886,0.03395671,0.8206569],"study_design_scores_gemma":[0.0002189308,0.0003016647,0.005629749,0.00025346837,0.00031191032,0.0024708342,0.00088983367,0.7001275,0.043226458,0.11849761,0.12783958,0.00023255196],"about_ca_topic_score_codex":0.0023850389,"about_ca_topic_score_gemma":0.0019420506,"teacher_disagreement_score":0.0121507,"about_ca_system_score_codex":0.00058352173,"about_ca_system_score_gemma":0.0015220628,"threshold_uncertainty_score":0.040648162},"labels":[],"label_agreement":null},{"id":"W2140566821","doi":"10.7202/004620ar","title":"MT Project at University of Innsbruck","year":2002,"lang":"fr","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Philosophy","score_opus":0.06732869272139835,"score_gpt":0.26566548871585016,"score_spread":0.1983367959944518,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2140566821","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.027122272,0.016356023,0.24087462,0.017839747,0.008254673,0.0020498782,0.11099238,0.13697603,0.43953446],"genre_scores_gemma":[0.097368486,0.005426641,0.18022537,0.0012513872,0.0015088796,0.0015667459,0.1726753,0.03075161,0.5092256],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9945011,0.0013323112,0.00047438263,0.0016600041,0.0015299873,0.0005022569],"domain_scores_gemma":[0.9929762,0.0021349804,0.00039525988,0.0017563998,0.0019571905,0.00078003755],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0044863885,0.0017285938,0.0029582728,0.00510682,0.0020042202,0.006906601,0.0027322203,0.0023108094,0.24066882],"category_scores_gemma":[0.014635723,0.0013022635,0.00077807705,0.00346278,0.0011632278,0.008072573,0.004469757,0.0018608068,0.15854576],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009039037,0.00016784458,0.0014094518,0.0010888809,0.000077276236,0.00053067005,0.0010350101,0.00079187256,0.010268635,0.024833776,0.42151454,0.53737813],"study_design_scores_gemma":[0.00024135651,0.000107335974,0.0031951698,0.00034904448,0.000049529495,0.0007146352,0.00041269895,0.0045683202,0.006686275,0.018114528,0.9654903,0.00007076717],"about_ca_topic_score_codex":0.008880147,"about_ca_topic_score_gemma":0.0066449773,"teacher_disagreement_score":0.24066882,"about_ca_system_score_codex":0.0015769377,"about_ca_system_score_gemma":0.0027738265,"threshold_uncertainty_score":0.80511737},"labels":[],"label_agreement":null},{"id":"W2140608987","doi":"10.1109/nlpke.2009.5313823","title":"Real-word spelling correction using Google Web 1T n-gram with backoff","year":2009,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":29,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Spelling; n-gram; Computer science; Word (group theory); Set (abstract data type); Edit distance; String (physics); Fraction (chemistry); Error detection and correction; Natural language processing; Matching (statistics); Precision and recall; Subsequence; Artificial intelligence; Information retrieval; Speech recognition; Algorithm; Language model; Mathematics; Statistics; Programming language","score_opus":0.014863048478481522,"score_gpt":0.27663143863838324,"score_spread":0.2617683901599017,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2140608987","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20510842,0.002208085,0.692383,0.00050974323,0.0011421698,0.0006699381,0.00886381,0.08491266,0.0042021405],"genre_scores_gemma":[0.23249064,0.0004947743,0.7486522,0.0001902242,0.00022858883,0.00037104942,0.010669579,0.0022858777,0.0046170517],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99592257,0.0006672519,0.00052722613,0.00095993414,0.0016959145,0.00022707492],"domain_scores_gemma":[0.98986804,0.0018211335,0.0015701272,0.0024003868,0.0041209348,0.00021938089],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001650102,0.002332276,0.0016466752,0.005470092,0.0011094782,0.0011534749,0.0015727474,0.0012047923,0.0032992817],"category_scores_gemma":[0.011238868,0.0004348829,0.0010800671,0.0039681518,0.00059447484,0.001533647,0.0011745738,0.0011873834,0.00565073],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010169706,0.00028846087,0.011389086,0.0008648154,0.00028953823,0.0007229122,0.00036949248,0.008069435,0.0956709,0.00087831233,0.016359497,0.8640806],"study_design_scores_gemma":[0.00018031517,0.0007508652,0.026812444,0.00019604046,0.00033192802,0.003002951,0.000521426,0.38022628,0.5363652,0.003889533,0.047227632,0.0004953336],"about_ca_topic_score_codex":0.008104238,"about_ca_topic_score_gemma":0.01735411,"teacher_disagreement_score":0.008104238,"about_ca_system_score_codex":0.00060223445,"about_ca_system_score_gemma":0.0019653058,"threshold_uncertainty_score":0.016114116},"labels":[],"label_agreement":null},{"id":"W2140679639","doi":"","title":"A Neural Probabilistic Language Model","year":2000,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1162,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; Computer Research Institute of Montréal","funders":"","keywords":"Trigram; Computer science; Artificial intelligence; Probabilistic logic; Word (group theory); Generalization; Natural language processing; Similarity (geometry); Language model; Curse of dimensionality; Sequence (biology); Representation (politics); Function (biology); Statistical model; Artificial neural network; Sentence; Probability distribution; Mathematics","score_opus":0.010705649963880414,"score_gpt":0.2618203588921802,"score_spread":0.2511147089282998,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2140679639","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013847325,0.0020493045,0.9620589,0.0036742182,0.000254395,0.000103291815,0.002334688,0.0017743779,0.013903492],"genre_scores_gemma":[0.6438712,0.003539429,0.30052125,0.0018372458,0.00063828344,0.000757179,0.0048118695,0.00043967104,0.043583937],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992675,0.00024099149,0.000041401952,0.0002326042,0.00015652226,0.000060954215],"domain_scores_gemma":[0.99884284,0.0007757118,0.00007729698,0.00009279602,0.00015993854,0.000051437593],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011242834,0.0006516622,0.0010703396,0.0011088812,0.0005933407,0.0022270172,0.0024995736,0.0020602166,0.008781947],"category_scores_gemma":[0.004360057,0.0005967916,0.0012293755,0.0014231537,0.0008396971,0.0040892777,0.0013134937,0.0023604555,0.0034486481],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023273946,0.0001820386,0.0018261494,0.0003432871,0.00025514816,0.00041571667,0.00030794015,0.46582782,0.0027858044,0.30955973,0.021620322,0.19664334],"study_design_scores_gemma":[0.000015239836,0.000018741526,0.00015758089,0.000017559454,0.00002077967,0.00009327352,0.00001367649,0.88617295,0.00017547944,0.1094494,0.0038516268,0.000013728617],"about_ca_topic_score_codex":0.0070323097,"about_ca_topic_score_gemma":0.0064375657,"teacher_disagreement_score":0.008781947,"about_ca_system_score_codex":0.0011221878,"about_ca_system_score_gemma":0.0012306308,"threshold_uncertainty_score":0.029378533},"labels":[],"label_agreement":null},{"id":"W2140682244","doi":"10.32920/27997979.v1","title":"Assessing the Performance of Automatic Speech Recognition Systems When Used by Native and Non-Native Speakers of Three Major Languages in Dictation Workflows","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"European Commission; Copenhagen Business School","keywords":"Dictation; Speech recognition; Computer science; Workflow; Natural language processing; First language; Linguistics; Database","score_opus":0.01732166725866894,"score_gpt":0.2933360615419495,"score_spread":0.27601439428328056,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2140682244","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.994474,0.00007031882,0.0037240349,0.000047363785,0.000032667533,0.00016463634,0.00015253898,0.000144968,0.0011894491],"genre_scores_gemma":[0.9777483,0.00010106628,0.01823526,0.00013966433,0.000048599217,0.00035979273,0.0008074681,0.00008456331,0.0024752873],"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99261594,0.0038098437,0.00095252536,0.0014058214,0.00083907176,0.0003768132],"domain_scores_gemma":[0.96248543,0.02730573,0.0014792404,0.0027546713,0.005015309,0.00095968373],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007819247,0.000911745,0.0008110374,0.0006436031,0.0006225262,0.0017066882,0.00076819304,0.0011994169,0.002946713],"category_scores_gemma":[0.03240842,0.000453176,0.0006372456,0.00034871645,0.00085885066,0.0015884171,0.0015226847,0.00070802076,0.002613284],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.017442394,0.004475288,0.16322097,0.0018701511,0.0008054796,0.0013301236,0.035115216,0.010742766,0.42382538,0.0007537169,0.0025663034,0.3378523],"study_design_scores_gemma":[0.0014730534,0.044645846,0.5340939,0.00019923881,0.0009155987,0.003463444,0.02442698,0.060300313,0.31758863,0.0011824887,0.010930111,0.00078040047],"about_ca_topic_score_codex":0.0022604268,"about_ca_topic_score_gemma":0.0031645151,"teacher_disagreement_score":0.007819247,"about_ca_system_score_codex":0.0004476697,"about_ca_system_score_gemma":0.00065934815,"threshold_uncertainty_score":0.04135263},"labels":[],"label_agreement":null},{"id":"W2140943754","doi":"10.3115/1699705.1699712","title":"DirecTL","year":2009,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":38,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Discriminative model; Sequence (biology); Artificial intelligence; Character (mathematics); Segmentation; Natural language processing; Machine learning","score_opus":0.007816252420145926,"score_gpt":0.2630345991925536,"score_spread":0.2552183467724077,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2140943754","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007069135,0.00072858826,0.7764979,0.00073800597,0.00034412695,0.00023777853,0.010006961,0.17624459,0.02813284],"genre_scores_gemma":[0.17498921,0.0009517675,0.6873763,0.0018759366,0.00027617457,0.0007001582,0.0443686,0.01729722,0.07216458],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9988305,0.00019306665,0.000059407637,0.0004186334,0.0003935404,0.000104859864],"domain_scores_gemma":[0.9980952,0.00054184993,0.000087592285,0.0007988877,0.00038671642,0.000089808724],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00096397114,0.0012203221,0.0008350987,0.0013367814,0.00069573085,0.0022984224,0.003748281,0.0014629234,0.04785498],"category_scores_gemma":[0.004702662,0.0009287631,0.00094218634,0.0010052621,0.00064051204,0.0043905145,0.00347549,0.0018639038,0.03234784],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00043885503,0.0002865546,0.0024343973,0.0006578947,0.000090081,0.00033776398,0.00017875222,0.02263452,0.008680782,0.034857277,0.23110284,0.6983003],"study_design_scores_gemma":[0.000120459896,0.00019699403,0.000630998,0.00010164396,0.000053698393,0.00067118224,0.00008927454,0.5649581,0.027284669,0.071437396,0.33435863,0.00009697732],"about_ca_topic_score_codex":0.0035935822,"about_ca_topic_score_gemma":0.007240331,"teacher_disagreement_score":0.04785498,"about_ca_system_score_codex":0.0010422439,"about_ca_system_score_gemma":0.0013276411,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W2141232148","doi":"10.3115/v1/w15-0915","title":"Building a Lexicon of Formulaic Language for Language Learners","year":2015,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Guelph; University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada; Mitacs","keywords":"Lexicon; Computer science; Natural language processing; Identification (biology); Artificial intelligence; Recall; Computational linguistics; Linguistics","score_opus":0.024499551044419653,"score_gpt":0.3261592496904346,"score_spread":0.30165969864601494,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2141232148","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05609623,0.00015951403,0.922922,0.00042114742,0.00010531497,0.00050613756,0.0019160216,0.010702727,0.0071709687],"genre_scores_gemma":[0.22057329,0.00023941214,0.7704062,0.00016516862,0.000045957615,0.0005919983,0.0044519594,0.0011066531,0.0024194324],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9986327,0.00037267857,0.00021617189,0.00038976417,0.000310665,0.00007788551],"domain_scores_gemma":[0.9932446,0.0026171212,0.00039555057,0.001278693,0.0021622605,0.0003016889],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015465904,0.0010763904,0.0011142003,0.0028120566,0.00077999756,0.0041286163,0.00182323,0.0012681856,0.010327163],"category_scores_gemma":[0.014541577,0.0010726219,0.001002542,0.0020536869,0.00077451585,0.008593748,0.002956082,0.0017892477,0.0078354115],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022036694,0.0003981025,0.009972867,0.00084221654,0.00010411939,0.0005955945,0.0031140223,0.008615481,0.0545265,0.049487323,0.016888054,0.8552354],"study_design_scores_gemma":[0.00025208766,0.001025431,0.009798798,0.0006331835,0.00042156433,0.0028752591,0.0069172718,0.46969363,0.120066516,0.21332179,0.1744795,0.00051499624],"about_ca_topic_score_codex":0.0017762833,"about_ca_topic_score_gemma":0.0027474458,"teacher_disagreement_score":0.010327163,"about_ca_system_score_codex":0.0008984473,"about_ca_system_score_gemma":0.0029054086,"threshold_uncertainty_score":0.034547806},"labels":[],"label_agreement":null},{"id":"W2141325213","doi":"10.1109/icsmc.2001.969854","title":"Filtering noisy parallel corpora of web pages","year":2002,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Parallel corpora; Translation (biology); Machine translation; Natural language processing; Artificial intelligence; Information retrieval; Web page; World Wide Web","score_opus":0.0297197218768101,"score_gpt":0.24320612826456398,"score_spread":0.21348640638775387,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2141325213","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2642702,0.0052378415,0.7026573,0.0015482964,0.0015816612,0.0008195179,0.0053952695,0.012399385,0.0060905935],"genre_scores_gemma":[0.4100993,0.003146835,0.53484184,0.0009041772,0.0014514616,0.0015772604,0.030898856,0.003072769,0.01400747],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9906963,0.0025976088,0.0011533609,0.0017585345,0.003423085,0.00037101127],"domain_scores_gemma":[0.9749708,0.01245857,0.0016759858,0.0045619844,0.0061028195,0.00022987813],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006383431,0.0017911412,0.0035394903,0.0064305793,0.002476809,0.002442493,0.0014727707,0.0017598907,0.003907894],"category_scores_gemma":[0.0328483,0.0016532463,0.001789039,0.009302962,0.0013680409,0.0037929073,0.0024875258,0.0022641334,0.0044663716],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.003759225,0.0009343771,0.013844973,0.004160136,0.0009299317,0.006187995,0.0030239653,0.052562717,0.15146239,0.010834365,0.048510488,0.7037895],"study_design_scores_gemma":[0.0005018083,0.0008922363,0.030917173,0.0004011026,0.0011806748,0.00604619,0.0021585748,0.53252196,0.25061294,0.038665626,0.13573559,0.0003660854],"about_ca_topic_score_codex":0.0030299774,"about_ca_topic_score_gemma":0.005008447,"teacher_disagreement_score":0.0064305793,"about_ca_system_score_codex":0.0008713155,"about_ca_system_score_gemma":0.001667468,"threshold_uncertainty_score":0.033759236},"labels":[],"label_agreement":null},{"id":"W2141332150","doi":"","title":"Sub-sentential exploitation of translation memories","year":2001,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":28,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Natural language processing; Translation (biology); Machine translation; Artificial intelligence; Example-based machine translation; Linguistics","score_opus":0.0232393079536819,"score_gpt":0.2740315550433252,"score_spread":0.2507922470896433,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2141332150","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6012691,0.0025719723,0.3380545,0.0016877672,0.0005352216,0.00009049992,0.0006299551,0.0068023214,0.048358615],"genre_scores_gemma":[0.9646773,0.0004163526,0.02415757,0.00012770649,0.00011666906,0.00002673615,0.00035661738,0.00040935734,0.009711626],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9995666,0.00013438954,0.000034793713,0.000100331206,0.000084394735,0.00007944311],"domain_scores_gemma":[0.99625945,0.0015760503,0.00019561732,0.0014429579,0.0004235105,0.00010249454],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005968954,0.00047786295,0.00046691194,0.0006190204,0.0005781388,0.0016716293,0.0009018952,0.00054953835,0.009592114],"category_scores_gemma":[0.0039782757,0.00041281874,0.00034768193,0.001068402,0.0006395424,0.003712519,0.0013639792,0.0006101051,0.0022018796],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0023193369,0.00021326922,0.0029548723,0.0006451493,0.00016374906,0.0021620905,0.001974556,0.014436416,0.20234121,0.10809294,0.009789678,0.6549067],"study_design_scores_gemma":[0.00018365501,0.0006212978,0.0035603875,0.00009712241,0.00047228014,0.002304221,0.0013545844,0.28564087,0.41580185,0.2445913,0.045251388,0.00012095878],"about_ca_topic_score_codex":0.00043207424,"about_ca_topic_score_gemma":0.0008753927,"teacher_disagreement_score":0.009592114,"about_ca_system_score_codex":0.00035896944,"about_ca_system_score_gemma":0.00047920726,"threshold_uncertainty_score":0.032088816},"labels":[],"label_agreement":null},{"id":"W2141510042","doi":"","title":"Resolving this-issue anaphora","year":2012,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Anaphora (linguistics); Computer science; Sentence; Natural language processing; Ranking (information retrieval); Resolution (logic); Artificial intelligence; Space (punctuation); Linguistics; Philosophy","score_opus":0.013143845405356513,"score_gpt":0.2773581684669683,"score_spread":0.2642143230616118,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2141510042","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15744066,0.0017277839,0.79030436,0.0027870794,0.0008104864,0.00044392908,0.0021488955,0.0039035128,0.0404334],"genre_scores_gemma":[0.6501284,0.0005662233,0.33528888,0.00038751893,0.0003137419,0.00010382265,0.0030054145,0.00051609403,0.009689868],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99576557,0.001401297,0.00032134014,0.0009770296,0.0013056195,0.00022915941],"domain_scores_gemma":[0.99301094,0.0032190904,0.00080253085,0.001455398,0.0013258742,0.00018610466],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0044492055,0.00086451357,0.00095962116,0.002256755,0.0021884588,0.0027138623,0.0015916638,0.0029005003,0.008298008],"category_scores_gemma":[0.015649142,0.0006684949,0.0013749301,0.0018876359,0.0008099986,0.0060038688,0.0030886552,0.0022310014,0.0025430855],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00075220695,0.0008251358,0.026649026,0.0026168078,0.00040430113,0.0021500746,0.009069266,0.023409044,0.061431095,0.18131043,0.0622966,0.62908596],"study_design_scores_gemma":[0.00019016567,0.00028930424,0.014240719,0.00039146558,0.00041082673,0.004494124,0.0038994816,0.3959957,0.09904632,0.20944944,0.27132478,0.00026770306],"about_ca_topic_score_codex":0.0023236056,"about_ca_topic_score_gemma":0.004465282,"teacher_disagreement_score":0.008298008,"about_ca_system_score_codex":0.000935474,"about_ca_system_score_gemma":0.0013347135,"threshold_uncertainty_score":0.027759612},"labels":[],"label_agreement":null},{"id":"W2141656477","doi":"10.5815/ijieeb.2012.01.01","title":"Sentence Clustering Using Parts-of-Speech","year":2012,"lang":"en","type":"article","venue":"International Journal of Information Engineering and Electronic Business","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Lakehead University","funders":"","keywords":"Computer science; Cluster analysis; Syntax; Artificial intelligence; Natural language processing; Sentence; Hierarchical clustering; Brown clustering; Index (typography); Metric (unit); Similarity (geometry); Speech recognition; Fuzzy clustering; Canopy clustering algorithm","score_opus":0.0071209194034512965,"score_gpt":0.24040784880244032,"score_spread":0.233286929398989,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2141656477","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029398581,0.0007132349,0.9639266,0.00017335244,0.00016254776,0.000434116,0.0006500804,0.002975774,0.001565746],"genre_scores_gemma":[0.16839461,0.00045584538,0.8229108,0.00012694587,0.00020559184,0.000420746,0.0038345782,0.00059457443,0.0030562477],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9980234,0.00045198132,0.00018471673,0.00064259226,0.0006041555,0.00009315472],"domain_scores_gemma":[0.9970739,0.0009710865,0.00038768997,0.00035298397,0.0011266476,0.000087642205],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012800705,0.0014495633,0.0014191787,0.0073355,0.0015892439,0.001602484,0.0017601966,0.0013502365,0.00211748],"category_scores_gemma":[0.0052462765,0.00047354458,0.0015764657,0.0048837187,0.0010174371,0.001786761,0.0010383843,0.00083424564,0.0024506487],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006256081,0.00026722075,0.0051947646,0.00080963754,0.00045563834,0.000618815,0.0018620614,0.046932288,0.09985859,0.012632557,0.011283956,0.8194588],"study_design_scores_gemma":[0.00007452475,0.0003128156,0.010205867,0.0001316604,0.00043161903,0.00094437593,0.001040835,0.8432192,0.07893032,0.041336756,0.023154799,0.00021725296],"about_ca_topic_score_codex":0.0031933435,"about_ca_topic_score_gemma":0.0032587582,"teacher_disagreement_score":0.0073355,"about_ca_system_score_codex":0.00081037957,"about_ca_system_score_gemma":0.001233164,"threshold_uncertainty_score":0.007083714},"labels":[],"label_agreement":null},{"id":"W2141821393","doi":"10.1111/j.1551-6709.2011.01174.x","title":"Rules Versus Statistics: Insights From a Highly Inflected Language","year":2011,"lang":"en","type":"article","venue":"Cognitive Science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":99,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"National Institute of Mental Health; National Heart, Lung, and Blood Institute; Canadian Institutes of Health Research","keywords":"Lemma (botany); Noun; Serbian; Linguistics; Representation (politics); Consistency (knowledge bases); Meaning (existential); Computer science; Natural language processing; Connectionism; Generalization; Word (group theory); Artificial intelligence; Mathematics; Psychology; Artificial neural network","score_opus":0.028324695201785526,"score_gpt":0.2914675035013571,"score_spread":0.26314280829957154,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2141821393","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.75425756,0.0010873479,0.2146315,0.0038372849,0.000027486994,0.000052792722,0.00033623938,0.00042211564,0.025347728],"genre_scores_gemma":[0.97361153,0.0005614117,0.024278907,0.0001658296,0.00002125414,0.000019150768,0.0001407356,0.000056968165,0.0011441795],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9998524,0.000056238416,0.0000076075517,0.00003844787,0.00003240944,0.00001279186],"domain_scores_gemma":[0.99866927,0.0008278227,0.00012280139,0.00019897951,0.00009232067,0.00008886837],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005035556,0.00033634607,0.0003232725,0.00076362625,0.000327392,0.0018354772,0.00085008284,0.0006829886,0.002104926],"category_scores_gemma":[0.0038183683,0.00034184064,0.00035558638,0.0005587802,0.002166804,0.005767324,0.0008599205,0.00096620957,0.00021181181],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00043928804,0.00029807512,0.052806813,0.0005481071,0.00015442667,0.0034778444,0.021015475,0.07049612,0.07492862,0.6042387,0.002847146,0.16874945],"study_design_scores_gemma":[0.000017479282,0.00009224292,0.01772503,0.000041451873,0.000022883343,0.0006110211,0.0014471272,0.14807843,0.0028112524,0.82400286,0.0051171766,0.000033119908],"about_ca_topic_score_codex":0.0028142277,"about_ca_topic_score_gemma":0.0033121724,"teacher_disagreement_score":0.0028142277,"about_ca_system_score_codex":0.0006665016,"about_ca_system_score_gemma":0.00030728278,"threshold_uncertainty_score":0.0070416927},"labels":[],"label_agreement":null},{"id":"W2141971526","doi":"","title":"Statistical Phrase-Based Post-Editing","year":2007,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":168,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"Human Resources and Skills Development Canada","keywords":"Machine translation; Computer science; Phrase; Natural language processing; BLEU; Task (project management); Artificial intelligence; Translation (biology); Mode (computer interface); Human–computer interaction; Engineering","score_opus":0.010851235828080551,"score_gpt":0.28337099586631076,"score_spread":0.2725197600382302,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2141971526","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012864206,0.00022967748,0.97345114,0.00024585705,0.00014573649,0.00015303947,0.00045007622,0.009286262,0.0031740307],"genre_scores_gemma":[0.16747236,0.00028436142,0.8239572,0.00039503534,0.0002221165,0.0002110203,0.0019983642,0.0009769907,0.0044825273],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99543977,0.0015071922,0.00037540638,0.0008916147,0.0016529069,0.00013307274],"domain_scores_gemma":[0.978893,0.007799916,0.0013839024,0.006140734,0.0055600055,0.00022248189],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033241196,0.00097550615,0.0012032933,0.0016493936,0.00054611405,0.0015983672,0.0027435096,0.0009925753,0.005537262],"category_scores_gemma":[0.016894737,0.00051446207,0.00079282955,0.0024196012,0.00082612416,0.0022800788,0.0010841195,0.0016190453,0.0065854187],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025456576,0.00041038918,0.0031405988,0.0006236419,0.00022751199,0.00058539934,0.00038962736,0.018282948,0.12124702,0.012861892,0.011159179,0.83081716],"study_design_scores_gemma":[0.00014550278,0.0011596881,0.006673494,0.000079250596,0.0002722153,0.0028184939,0.00025107537,0.58342415,0.30159456,0.0474063,0.05590469,0.00027063114],"about_ca_topic_score_codex":0.000898324,"about_ca_topic_score_gemma":0.0016240499,"teacher_disagreement_score":0.005537262,"about_ca_system_score_codex":0.00028175965,"about_ca_system_score_gemma":0.0013414972,"threshold_uncertainty_score":0.018523991},"labels":[],"label_agreement":null},{"id":"W2142233105","doi":"10.1093/llc/fql049","title":"Developing Web Databases for Aboriginal Language Preservation","year":2006,"lang":"en","type":"article","venue":"Literary and Linguistic Computing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Context (archaeology); Citizen journalism; Participatory action research; World Wide Web; Participatory design; Focus (optics); Database; Sociology; Engineering; History","score_opus":0.015520884457934501,"score_gpt":0.3168664813097029,"score_spread":0.3013455968517684,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2142233105","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13696201,0.0021868937,0.82054025,0.005501611,0.00019000506,0.001743454,0.0009826523,0.0059157624,0.025977291],"genre_scores_gemma":[0.15802,0.0015460498,0.83056563,0.00033901082,0.000051514944,0.00064254954,0.0020941398,0.0008102577,0.005930779],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9913455,0.0044446965,0.0010278346,0.0009358719,0.0019898587,0.0002563245],"domain_scores_gemma":[0.97298807,0.013299111,0.0014301811,0.005934537,0.005405617,0.0009426171],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.021735469,0.00045888117,0.00061244314,0.004039368,0.0033811277,0.00857839,0.0038669363,0.0015117797,0.0031007477],"category_scores_gemma":[0.03218155,0.00078738836,0.00070876913,0.0036264427,0.0018195957,0.012980207,0.006144507,0.0017067776,0.0012918586],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00031705285,0.00088134053,0.013412463,0.0014954943,0.00015954404,0.0016160238,0.056329217,0.009908739,0.02906841,0.0896087,0.01717637,0.7800267],"study_design_scores_gemma":[0.00020471263,0.00050903694,0.009040968,0.0012552206,0.00029987647,0.0022825783,0.06983203,0.07045993,0.10262791,0.07232613,0.67086893,0.00029272068],"about_ca_topic_score_codex":0.006097932,"about_ca_topic_score_gemma":0.007893202,"teacher_disagreement_score":0.021735469,"about_ca_system_score_codex":0.0018068599,"about_ca_system_score_gemma":0.00422984,"threshold_uncertainty_score":0.114949524},"labels":[],"label_agreement":null},{"id":"W2142435644","doi":"10.3115/1699648.1699651","title":"A compact forest for scalable inference over entailment and paraphrase rules","year":2009,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"FP7 Information and Communication Technologies; Azrieli Foundation; Israel Science Foundation; Bar-Ilan University","keywords":"Paraphrase; Logical consequence; Inference; Computer science; Scalability; Formalism (music); Natural language processing; Artificial intelligence; Rule of inference; Textual entailment; Knowledge acquisition; Theoretical computer science; Database","score_opus":0.015575313473020685,"score_gpt":0.306192395107473,"score_spread":0.2906170816344523,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2142435644","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0043546855,0.00014440395,0.9920397,0.00012817515,0.00002693362,0.00008540423,0.00033621208,0.0021895333,0.00069491874],"genre_scores_gemma":[0.05811952,0.00015910172,0.938166,0.000098862765,0.000054409873,0.00015665269,0.0015683798,0.0002529796,0.0014240496],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985619,0.00025765566,0.00014427118,0.00038959272,0.00050624297,0.00014034289],"domain_scores_gemma":[0.9962759,0.0020954686,0.00018576386,0.0007200429,0.0005984054,0.00012439025],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001766569,0.0006690922,0.0011295673,0.0018763446,0.0010393526,0.0017898355,0.002316251,0.0012897932,0.0064828],"category_scores_gemma":[0.010462983,0.00084731827,0.0013402955,0.0027699512,0.00094964815,0.005203443,0.0026856305,0.002271277,0.0024505835],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00031889265,0.0002843836,0.0010472016,0.0003996573,0.000093146635,0.00023958075,0.00025896903,0.08703157,0.012399328,0.10084558,0.014462531,0.7826191],"study_design_scores_gemma":[0.00007757371,0.00005455229,0.00026284452,0.00004666688,0.00004360456,0.00016311627,0.00004078949,0.8034438,0.004260667,0.18558078,0.006004273,0.000021358848],"about_ca_topic_score_codex":0.0091367895,"about_ca_topic_score_gemma":0.01817603,"teacher_disagreement_score":0.0091367895,"about_ca_system_score_codex":0.0010335747,"about_ca_system_score_gemma":0.0024276644,"threshold_uncertainty_score":0.02168715},"labels":[],"label_agreement":null},{"id":"W2142459460","doi":"","title":"Evaluating Roget's Thesauri","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Thesaurus; WordNet; Computer science; Sentence; Information retrieval; Natural language processing; Artificial intelligence; Synonym (taxonomy); Biology","score_opus":0.06413836483233207,"score_gpt":0.35234287561063277,"score_spread":0.28820451077830067,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2142459460","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.90222204,0.0033104762,0.0459558,0.0009299754,0.00039410812,0.000638848,0.002725853,0.0070521557,0.036770854],"genre_scores_gemma":[0.8007747,0.0009847061,0.17116806,0.00038285024,0.00011224198,0.0003583675,0.0155673055,0.0014736542,0.009178081],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9865579,0.006450311,0.0014975589,0.0016329,0.0035694484,0.0002918041],"domain_scores_gemma":[0.95873445,0.025082406,0.0017464145,0.0046467483,0.009083466,0.00070650136],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012997868,0.0008947111,0.0013170531,0.008075723,0.0015461132,0.0047897315,0.0020069191,0.0020399797,0.0038712735],"category_scores_gemma":[0.072869375,0.000740609,0.0011480389,0.0048869657,0.0014747018,0.008467758,0.0053288178,0.0013155844,0.0020183579],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0026214893,0.0011136549,0.04238504,0.003271661,0.0012183312,0.0015526894,0.017817162,0.02323725,0.039157536,0.021878522,0.043094862,0.8026518],"study_design_scores_gemma":[0.0012776725,0.006264612,0.16059479,0.0015411482,0.0021510879,0.0066651325,0.013487998,0.3459896,0.063109785,0.018825583,0.37916133,0.0009312807],"about_ca_topic_score_codex":0.010234218,"about_ca_topic_score_gemma":0.012691761,"teacher_disagreement_score":0.012997868,"about_ca_system_score_codex":0.0022298894,"about_ca_system_score_gemma":0.0014997856,"threshold_uncertainty_score":0.06874013},"labels":[],"label_agreement":null},{"id":"W2142669279","doi":"10.3115/v1/w14-3363","title":"Linear Mixture Models for Robust Machine Translation","year":2014,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Domain adaptation; Computer science; Machine translation; Leverage (statistics); Artificial intelligence; Adaptation (eye); Translation (biology); Machine learning; Domain (mathematical analysis); Repurposing; Robustness (evolution); Quality (philosophy); Linear model; Homogeneous; Mathematics","score_opus":0.026229986052772263,"score_gpt":0.2692085829693952,"score_spread":0.24297859691662294,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2142669279","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0013230353,0.0018394267,0.99447256,0.00033491038,0.00008042062,0.000023361432,0.0001269631,0.00087162456,0.00092774257],"genre_scores_gemma":[0.2299516,0.00611984,0.74277216,0.00066578324,0.001108479,0.00070229574,0.0026268153,0.0014667478,0.01458628],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99729115,0.0013575508,0.00015857979,0.0005618561,0.00049819006,0.00013263263],"domain_scores_gemma":[0.99448794,0.0038293197,0.0004334272,0.00066626974,0.00048653703,0.00009655208],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004174308,0.0019989442,0.002136527,0.0023419857,0.00083859253,0.0021571603,0.0021987339,0.0028652875,0.0046614376],"category_scores_gemma":[0.013648521,0.0017416965,0.0022293408,0.0028157777,0.0019706797,0.0026540274,0.0029522984,0.0044510243,0.0047526346],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019038531,0.00008147934,0.00063740386,0.00041121693,0.00042765468,0.00016776595,0.00018901489,0.6656458,0.0040680035,0.11284657,0.009066972,0.20626774],"study_design_scores_gemma":[0.000011120522,0.00002090073,0.00015072842,0.000026226942,0.000024819119,0.000033801898,0.000009674049,0.9123765,0.00080521003,0.082975425,0.0035352102,0.000030336867],"about_ca_topic_score_codex":0.006509618,"about_ca_topic_score_gemma":0.0051399576,"teacher_disagreement_score":0.006509618,"about_ca_system_score_codex":0.0014720297,"about_ca_system_score_gemma":0.0011229926,"threshold_uncertainty_score":0.02207613},"labels":[],"label_agreement":null},{"id":"W2142831503","doi":"10.1017/s0305000904006099","title":"Commentary on <i>Learnability and linguistic performance</i> by Kenneth F. Drozd","year":2004,"lang":"en","type":"article","venue":"Journal of Child Language","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Learnability; Modularity (biology); Linguistics; Psychology; Matching (statistics); Cognitive science; Artificial intelligence; Philosophy; Computer science; Mathematics","score_opus":0.0029959378526653287,"score_gpt":0.2321506281381869,"score_spread":0.22915469028552157,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2142831503","genre_codex":"commentary","genre_gemma":"commentary","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"commentary","genre_consensus":"commentary","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0000416485,0.0024033585,0.000053394262,0.98013043,0.016833335,0.0000015247397,0.000026613281,0.0000064009623,0.0005032592],"genre_scores_gemma":[0.0025292868,0.0035321002,0.00017068617,0.9460265,0.044565074,0.00002675647,0.000025396112,0.000044715936,0.003079419],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9941089,0.001866981,0.00041651542,0.0012669835,0.0018865598,0.00045404254],"domain_scores_gemma":[0.95776176,0.030318996,0.0013211137,0.0008190938,0.007855953,0.0019230762],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009796075,0.0013439205,0.0019779888,0.0017253722,0.0067960187,0.005326432,0.007676038,0.036227964,0.0066398825],"category_scores_gemma":[0.044343952,0.0008413847,0.0015132185,0.0018427015,0.01158584,0.008344474,0.003206483,0.05594682,0.004689261],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000016290647,0.0000033050533,0.00003396004,0.00004356212,0.000004847252,0.000049361315,0.00019641832,0.000018587265,0.000021778995,0.0031035184,0.99454486,0.0019634725],"study_design_scores_gemma":[0.000052321822,0.000033409888,0.00075192685,0.0011031495,0.00003736766,0.00029709775,0.0013601457,0.00021720304,0.00020760673,0.028448708,0.9674077,0.00008334208],"about_ca_topic_score_codex":0.038382705,"about_ca_topic_score_gemma":0.041580927,"teacher_disagreement_score":0.038382705,"about_ca_system_score_codex":0.006738811,"about_ca_system_score_gemma":0.0063135475,"threshold_uncertainty_score":0.07631862},"labels":[],"label_agreement":null},{"id":"W2142986430","doi":"10.1109/icassp.2009.4960494","title":"Effect of pronounciations on OOV queries in spoken term detection","year":2009,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":53,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"IBM (Canada)","funders":"Türkiye Bilimsel ve Teknolojik Araştırma Kurumu","keywords":"Term (time); Computer science; Artificial intelligence; Speech recognition; Natural language processing","score_opus":0.004233618475100973,"score_gpt":0.26278438582417,"score_spread":0.25855076734906907,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2142986430","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9687576,0.0011732714,0.023598215,0.0002848841,0.00011639533,0.00007050125,0.00039264097,0.0031871388,0.0024192897],"genre_scores_gemma":[0.9806713,0.0002575237,0.015900044,0.0001695956,0.000053459196,0.00004106058,0.00092913175,0.00041687148,0.0015610803],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9929819,0.0027267705,0.0007960861,0.0012065793,0.0017173924,0.0005712872],"domain_scores_gemma":[0.9234747,0.06825917,0.0016359894,0.0028842883,0.0025042328,0.0012415354],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005929253,0.001187488,0.0016439713,0.0009718256,0.000957949,0.0025785668,0.0010333419,0.001517945,0.0018328957],"category_scores_gemma":[0.062974066,0.0005457563,0.00058066176,0.0011051897,0.0010135286,0.0045767743,0.0019732784,0.0017196515,0.0010503868],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.015296922,0.0014203017,0.060765173,0.0012865011,0.0005447536,0.0018679578,0.0026261057,0.12351413,0.2085526,0.0020114686,0.0052073067,0.57690674],"study_design_scores_gemma":[0.00023382445,0.0029653064,0.036300823,0.00006909538,0.00041432562,0.0013510684,0.0013005638,0.79461294,0.15657634,0.0031553393,0.0027947014,0.00022572235],"about_ca_topic_score_codex":0.012552364,"about_ca_topic_score_gemma":0.009922801,"teacher_disagreement_score":0.012552364,"about_ca_system_score_codex":0.0009462387,"about_ca_system_score_gemma":0.0012829682,"threshold_uncertainty_score":0.03135723},"labels":[],"label_agreement":null},{"id":"W2142989929","doi":"10.1007/978-3-642-21043-3_10","title":"Automatic Semantic Web Annotation of Named Entities","year":2011,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Information retrieval; Annotation; Entity linking; Semantic similarity; Semantic Web Stack; Semantic Web; RDF; Social Semantic Web; Set (abstract data type); Context (archaeology); Semantic annotation; Natural language processing; Artificial intelligence; Knowledge base; Programming language","score_opus":0.013889539820225304,"score_gpt":0.24719479246439624,"score_spread":0.23330525264417093,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2142989929","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022015452,0.0021299787,0.92025274,0.000825908,0.0006145754,0.00024666087,0.008783657,0.02261597,0.022515045],"genre_scores_gemma":[0.1159197,0.002662494,0.8111941,0.0003429864,0.00020404912,0.00029430017,0.049459647,0.0030579432,0.01686471],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99882597,0.00024075799,0.00014940866,0.00022924876,0.00048374114,0.000070829316],"domain_scores_gemma":[0.9975526,0.0009642223,0.00015470728,0.0005927454,0.0006669884,0.00006869848],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014596928,0.00079698925,0.0008613491,0.005505829,0.0014601877,0.002875522,0.0015216868,0.0011498925,0.008370196],"category_scores_gemma":[0.0038568585,0.00066570326,0.0011862054,0.0054523847,0.0006887659,0.005477999,0.002521056,0.0014605355,0.006449016],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00027550038,0.00029886293,0.0022032494,0.0014386597,0.0001135221,0.0009595525,0.00090975396,0.0061858967,0.033225153,0.10451805,0.09161276,0.758259],"study_design_scores_gemma":[0.00006836054,0.00006689661,0.004658185,0.0009051444,0.0002612844,0.0015117569,0.0008562155,0.24179785,0.08927034,0.22578786,0.43467993,0.00013620553],"about_ca_topic_score_codex":0.002988437,"about_ca_topic_score_gemma":0.005347576,"teacher_disagreement_score":0.008370196,"about_ca_system_score_codex":0.00076828484,"about_ca_system_score_gemma":0.0015884935,"threshold_uncertainty_score":0.02800107},"labels":[],"label_agreement":null},{"id":"W2143177362","doi":"10.3115/v1/w14-3346","title":"A Systematic Comparison of Smoothing Techniques for Sentence-Level BLEU","year":2014,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":188,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta; National Research Council Canada","funders":"","keywords":"BLEU; Computer science; Smoothing; Sentence; Artificial intelligence; Natural language processing; Speech recognition; Machine translation; Computer vision","score_opus":0.03508668186407472,"score_gpt":0.3199661980098258,"score_spread":0.2848795161457511,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2143177362","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11419979,0.041137047,0.75458187,0.0010540262,0.0020712782,0.0021746175,0.013792443,0.04271761,0.028271401],"genre_scores_gemma":[0.29811308,0.010191444,0.64036137,0.00061087764,0.0003488898,0.002932439,0.029253598,0.008201726,0.009986575],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9888859,0.0048478264,0.0011447467,0.0019117551,0.0028518802,0.00035795313],"domain_scores_gemma":[0.9572591,0.020658202,0.0016980738,0.0097220335,0.010311894,0.00035073227],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01347895,0.0017991378,0.0012563542,0.007004101,0.0013774597,0.0016539595,0.001843076,0.001250185,0.0044306093],"category_scores_gemma":[0.056526106,0.00058600557,0.0015557677,0.0072635375,0.00053867546,0.0030601434,0.0016748738,0.0019555069,0.004111981],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0022050117,0.00040869467,0.007816457,0.002633109,0.0015188166,0.00006367067,0.00066671206,0.01969358,0.023761485,0.008643453,0.031744856,0.9008442],"study_design_scores_gemma":[0.00091118633,0.005497408,0.103474416,0.0025028654,0.002995435,0.0017500148,0.0013058608,0.3461384,0.21405178,0.042627584,0.2772306,0.0015143931],"about_ca_topic_score_codex":0.004306984,"about_ca_topic_score_gemma":0.008722151,"teacher_disagreement_score":0.01347895,"about_ca_system_score_codex":0.0013284544,"about_ca_system_score_gemma":0.0018418473,"threshold_uncertainty_score":0.071284354},"labels":[],"label_agreement":null},{"id":"W2143231262","doi":"10.7202/013261ar","title":"Target Text Contraction in English-into-Korean Translations: A Contradiction of Presumed Translation Universals?*","year":2006,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Problem of universals; Spelling; Dynamic and formal equivalence; Contradiction; Linguistics; Computer science; Translation (biology); Contraction (grammar); Natural language processing; Artificial intelligence; Machine translation; Philosophy","score_opus":0.01647929274778142,"score_gpt":0.24887069368151205,"score_spread":0.23239140093373062,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2143231262","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8948233,0.00095222594,0.07448406,0.0012174348,0.00006466595,0.00010111796,0.00019990867,0.0002982377,0.027859109],"genre_scores_gemma":[0.9914731,0.0001529174,0.007620443,0.00008489903,0.0000130691515,0.000047206995,0.00008974543,0.00006849365,0.00045012461],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.995231,0.002413903,0.0005443197,0.0009731294,0.000618705,0.00021890996],"domain_scores_gemma":[0.9752814,0.01454812,0.0031749255,0.004052246,0.0026890514,0.00025429906],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008515412,0.00024544928,0.00050706783,0.0012077586,0.0011823154,0.0018407956,0.000636505,0.00037787255,0.0020613002],"category_scores_gemma":[0.03069231,0.00042166913,0.0003295769,0.0012498444,0.0033932533,0.0052669747,0.002108311,0.0010482745,0.00031632802],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016085386,0.00024815573,0.06912651,0.0018315221,0.00016946768,0.0018663938,0.104469,0.0039238115,0.038705777,0.42875615,0.0019250151,0.34736964],"study_design_scores_gemma":[0.00035799318,0.0013584034,0.23864217,0.0010806437,0.00056613423,0.0093066525,0.098256744,0.079568736,0.12929976,0.3728748,0.0684072,0.0002808285],"about_ca_topic_score_codex":0.0011794334,"about_ca_topic_score_gemma":0.0012779527,"teacher_disagreement_score":0.008515412,"about_ca_system_score_codex":0.00092400564,"about_ca_system_score_gemma":0.001190428,"threshold_uncertainty_score":0.04503435},"labels":[],"label_agreement":null},{"id":"W2143564602","doi":"10.3115/1220175.1220271","title":"An end-to-end discriminative approach to machine translation","year":2006,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":265,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Fonds Québécois de la Recherche sur la Nature et les Technologies; Microsoft Research","keywords":"Discriminative model; Computer science; Artificial intelligence; Machine translation; Phrase; Heuristic; Feature (linguistics); Machine learning; Translation (biology); Process (computing); Decoding methods; Perceptron; Natural language processing; Pattern recognition (psychology); Artificial neural network; Algorithm","score_opus":0.018024623253586287,"score_gpt":0.28280607116973405,"score_spread":0.26478144791614777,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2143564602","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0045812055,0.00010612261,0.98720956,0.00018992227,0.000051454066,0.000045983947,0.0001732613,0.0050917068,0.0025508197],"genre_scores_gemma":[0.20714036,0.00023340569,0.7765785,0.0005928919,0.0001819965,0.0001559364,0.0019978494,0.00085714494,0.0122618815],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9981565,0.0006205077,0.000118699376,0.00047671396,0.0005096754,0.00011791479],"domain_scores_gemma":[0.9969982,0.00071814115,0.00020687614,0.0012178664,0.000743375,0.000115451905],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018062509,0.0010104034,0.0011339735,0.0010560438,0.0010013636,0.00137644,0.0020344579,0.0013393295,0.005596932],"category_scores_gemma":[0.0058761504,0.00060562533,0.0005107394,0.002023786,0.0007910636,0.0029018123,0.002084572,0.0022605818,0.005932039],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002947923,0.0005040477,0.0017353158,0.00026623785,0.00008628405,0.00026772788,0.00022270986,0.04578188,0.045951605,0.025621155,0.020862794,0.8584054],"study_design_scores_gemma":[0.000048067337,0.00028305541,0.00097194174,0.000025341256,0.00004562788,0.0008818382,0.0000601411,0.88657784,0.037909918,0.047959737,0.025166169,0.000070265814],"about_ca_topic_score_codex":0.0012360694,"about_ca_topic_score_gemma":0.00381234,"teacher_disagreement_score":0.005596932,"about_ca_system_score_codex":0.00049838383,"about_ca_system_score_gemma":0.0010420489,"threshold_uncertainty_score":0.018723607},"labels":[],"label_agreement":null},{"id":"W2143929029","doi":"10.1109/icassp.1996.540305","title":"Using a transcription graph for large vocabulary continuous speech recognition","year":2002,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Institut National de la Recherche Scientifique","funders":"","keywords":"Computer science; Transcription (linguistics); Vocabulary; Natural language processing; Graph; Lexicon; Artificial intelligence; Speech recognition; Theoretical computer science; Linguistics","score_opus":0.05265805189005797,"score_gpt":0.2831527656018131,"score_spread":0.23049471371175512,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2143929029","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0017256789,0.00010712165,0.9904017,0.00006839896,0.00004359265,0.000043818018,0.00030353662,0.0066302544,0.00067596254],"genre_scores_gemma":[0.061381828,0.00027735878,0.9297847,0.00012883359,0.00006240379,0.00017717187,0.0042922786,0.0011322608,0.0027631582],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991179,0.00020640102,0.000061360326,0.00030673319,0.00025929505,0.00004832938],"domain_scores_gemma":[0.9982895,0.0008664425,0.00010875327,0.00031371808,0.0003636447,0.000057840392],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005730745,0.001147041,0.00075754954,0.0017521514,0.00069690146,0.0013453856,0.0015140445,0.0010790157,0.007199641],"category_scores_gemma":[0.0035273468,0.0005247861,0.0007672036,0.0020518613,0.0008536978,0.0026867113,0.0010299466,0.0012325653,0.0043517],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022659541,0.00011357936,0.000626594,0.0003049494,0.00006738377,0.00040445113,0.00021038672,0.057985526,0.036498975,0.028866544,0.015539176,0.85915595],"study_design_scores_gemma":[0.00011484888,0.00018932958,0.00064179976,0.00008457207,0.000087456785,0.0005767289,0.00018384786,0.80063605,0.0394081,0.11729497,0.040664114,0.00011826156],"about_ca_topic_score_codex":0.008262852,"about_ca_topic_score_gemma":0.008944153,"teacher_disagreement_score":0.008262852,"about_ca_system_score_codex":0.0007145413,"about_ca_system_score_gemma":0.0011127897,"threshold_uncertainty_score":0.024085224},"labels":[],"label_agreement":null},{"id":"W2144243803","doi":"10.4230/dagsemproc.10291.12","title":"On the Notion of Genre in Digital Preservation","year":2010,"lang":"en","type":"article","venue":"DROPS (Schloss Dagstuhl – Leibniz Center for Informatics)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Context (archaeology); Key (lock); Representation (politics); Open research; Digital preservation; Automation; World Wide Web; Data science; Information retrieval; Engineering; Computer security; History; Political science; Politics","score_opus":0.010652695654290995,"score_gpt":0.2503479191156682,"score_spread":0.23969522346137723,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2144243803","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.040782854,0.037292425,0.60661495,0.021297924,0.003197183,0.00027299635,0.0006101218,0.0004035213,0.2895281],"genre_scores_gemma":[0.674415,0.022773622,0.26273906,0.004218465,0.0057703564,0.00048074583,0.00096760783,0.00036671618,0.028268345],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99430686,0.0030375032,0.00054327,0.0007630502,0.0010493008,0.00030002533],"domain_scores_gemma":[0.99141055,0.0052080043,0.0007012635,0.0013490783,0.0008191134,0.00051194563],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0062579466,0.0007566841,0.0008878629,0.007884935,0.004791266,0.009556505,0.0013854166,0.0022642305,0.0062364163],"category_scores_gemma":[0.011997003,0.00051847193,0.0012805064,0.007850832,0.018407578,0.027772363,0.0060571083,0.003919607,0.0012150974],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000021231295,0.000008590613,0.0002872223,0.00012115925,0.0000070924793,0.00007291118,0.0019066527,0.0003940552,0.00027898923,0.97211623,0.0015046078,0.023281228],"study_design_scores_gemma":[0.000012021534,0.000029207691,0.000601761,0.00028614135,0.000019895324,0.00033515587,0.0015185922,0.0027662744,0.0003513308,0.9256966,0.06835114,0.000031961266],"about_ca_topic_score_codex":0.0024714402,"about_ca_topic_score_gemma":0.0016401331,"teacher_disagreement_score":0.009556505,"about_ca_system_score_codex":0.003226064,"about_ca_system_score_gemma":0.0010170436,"threshold_uncertainty_score":0.0330956},"labels":[],"label_agreement":null},{"id":"W2144746247","doi":"10.3115/1220355.1220401","title":"Confidence estimation for machine translation","year":2004,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":356,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"NIST; Correctness; Computer science; Machine translation; Natural language processing; Context (archaeology); Artificial intelligence; Translation (biology); Estimation; Evaluation of machine translation; Machine learning; Machine translation software usability; Example-based machine translation; Algorithm; Engineering","score_opus":0.019423825306592715,"score_gpt":0.3005074643097555,"score_spread":0.2810836390031628,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2144746247","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008382616,0.0025729765,0.9865919,0.0003575102,0.00007392007,0.000037806203,0.00016313094,0.00053734705,0.0012827828],"genre_scores_gemma":[0.5637931,0.0024503462,0.4279362,0.00033520476,0.0010465508,0.00027093233,0.0019362116,0.00061218976,0.0016192518],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.97660404,0.0119916415,0.0017329927,0.002758735,0.0061276373,0.0007848838],"domain_scores_gemma":[0.7571788,0.21053526,0.009310734,0.011168131,0.0108111575,0.0009960175],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.017064702,0.0015713598,0.0027110996,0.006036328,0.001360917,0.004188172,0.0029406978,0.00282462,0.004034267],"category_scores_gemma":[0.20468698,0.0011257967,0.001831429,0.0056215157,0.0030843313,0.007656869,0.003694483,0.004639934,0.0014653228],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007985792,0.00012500698,0.013887125,0.0011952852,0.000549535,0.0004198067,0.00058017747,0.3181071,0.0064823,0.20204164,0.0075132907,0.4483002],"study_design_scores_gemma":[0.000036666595,0.00011290984,0.0021332987,0.00012892752,0.000075214164,0.00036204097,0.000043790522,0.8140833,0.005487905,0.17450163,0.0029453926,0.0000889362],"about_ca_topic_score_codex":0.002006905,"about_ca_topic_score_gemma":0.0007858269,"teacher_disagreement_score":0.017064702,"about_ca_system_score_codex":0.0017134397,"about_ca_system_score_gemma":0.0011806649,"threshold_uncertainty_score":0.09024787},"labels":[],"label_agreement":null},{"id":"W2144806294","doi":"10.1109/icmla.2010.76","title":"Building a Biomedical Tokenizer Using the Token Lattice Design Pattern and the Adapted Viterbi Algorithm","year":2010,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"","keywords":"Security token; Computer science; Lexical analysis; Lattice (music); Computer network; Artificial intelligence; Physics","score_opus":0.01827251379567915,"score_gpt":0.2824310508638777,"score_spread":0.2641585370681986,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2144806294","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.002409088,0.000045839217,0.9915759,0.00010497365,0.000051292318,0.00011364655,0.00020921348,0.0049326685,0.0005574219],"genre_scores_gemma":[0.027488554,0.000045291243,0.9689956,0.00009781005,0.000024668707,0.0002558144,0.00053972617,0.0006346015,0.0019180138],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99741274,0.0008916188,0.00039403987,0.000659589,0.0004971767,0.00014492543],"domain_scores_gemma":[0.99559253,0.0024528152,0.00026735623,0.0005451701,0.0010287325,0.000113273134],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0045429277,0.0010015044,0.0011007095,0.0022601073,0.00083808124,0.0016758586,0.0024592655,0.0012945166,0.010756497],"category_scores_gemma":[0.012473989,0.0008699814,0.00084158074,0.0018232466,0.0013726347,0.0030130006,0.0017613246,0.0019151205,0.006253358],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013473477,0.00012830876,0.0017988364,0.0008509951,0.00015695358,0.0004456496,0.0005872922,0.05001055,0.052435946,0.07679613,0.023109028,0.7923329],"study_design_scores_gemma":[0.00024197536,0.00023930454,0.00040206968,0.000072227784,0.00011074583,0.0006150648,0.0001649419,0.77426255,0.129942,0.052932587,0.040900722,0.00011590689],"about_ca_topic_score_codex":0.0024579207,"about_ca_topic_score_gemma":0.0038351952,"teacher_disagreement_score":0.010756497,"about_ca_system_score_codex":0.001261029,"about_ca_system_score_gemma":0.003270861,"threshold_uncertainty_score":0.03598404},"labels":[],"label_agreement":null},{"id":"W2145016481","doi":"","title":"How do you pronounce your name? Improving G2P with transliterations","year":2011,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Grapheme; Ranking (information retrieval); Leverage (statistics); Pronunciation; Artificial intelligence; Natural language processing; Speech recognition; Linguistics; Engineering","score_opus":0.01992059269736257,"score_gpt":0.23265244763037235,"score_spread":0.21273185493300978,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2145016481","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3075009,0.002024834,0.6006176,0.0026155864,0.00073705113,0.00032726524,0.0033513189,0.06088955,0.021935852],"genre_scores_gemma":[0.5777633,0.0009242977,0.39628536,0.0008593083,0.00021390282,0.0001281088,0.0048929024,0.003503235,0.015429539],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9983754,0.0007215293,0.00007383451,0.00040534697,0.00029025285,0.00013365535],"domain_scores_gemma":[0.99538064,0.0028295158,0.00027687164,0.0005067523,0.00090462685,0.00010164287],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015609969,0.0020306672,0.0012370993,0.0011609818,0.0009305226,0.0017453184,0.0012958457,0.0013419605,0.0064996155],"category_scores_gemma":[0.010819637,0.0003070614,0.00041256775,0.0014866324,0.00052703894,0.003114326,0.0010100083,0.001359441,0.012587917],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009899433,0.00027052526,0.0088177305,0.00077947654,0.000106616666,0.0010254696,0.0016263421,0.016826838,0.065843776,0.0028937794,0.051779065,0.84904045],"study_design_scores_gemma":[0.0002731319,0.0006062905,0.009837922,0.000114596856,0.00027333933,0.0023757955,0.0030073852,0.6446966,0.25356996,0.02071447,0.064252615,0.00027789877],"about_ca_topic_score_codex":0.006476409,"about_ca_topic_score_gemma":0.010034088,"teacher_disagreement_score":0.0064996155,"about_ca_system_score_codex":0.00050189544,"about_ca_system_score_gemma":0.0010450099,"threshold_uncertainty_score":0.021743417},"labels":[],"label_agreement":null},{"id":"W2145084016","doi":"10.1023/a:1012262211784","title":"Unit Completion for a Computer-aided Translation Typing System","year":2000,"lang":"en","type":"article","venue":"Machine Translation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":28,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Computational linguistics; Natural language processing; Context (archaeology); Word (group theory); Artificial intelligence; Machine translation; Linguistics; History","score_opus":0.0330547707272114,"score_gpt":0.2821371979595065,"score_spread":0.24908242723229507,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2145084016","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025714153,0.00017821645,0.93669754,0.00030263298,0.00025491763,0.00021041751,0.0008450493,0.032138433,0.0036585643],"genre_scores_gemma":[0.18375333,0.00015627609,0.7893629,0.00015979697,0.0001434401,0.0003278669,0.0029105714,0.0038430735,0.01934265],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9981798,0.0005128578,0.0002596083,0.00042924457,0.00044462288,0.000173761],"domain_scores_gemma":[0.9963876,0.0015094843,0.00014609928,0.00064630865,0.001167184,0.00014325036],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016232351,0.0011194891,0.0018710457,0.0010957938,0.001753035,0.0022064124,0.0015761092,0.0015205721,0.025341207],"category_scores_gemma":[0.0055716024,0.00087451556,0.0011396566,0.0014864277,0.0006753654,0.0020985953,0.0016310076,0.0017345845,0.012730131],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0033536009,0.0005277111,0.0011661856,0.0011839494,0.0001330747,0.0016128046,0.0013424839,0.025775403,0.086574286,0.06262113,0.07692749,0.73878187],"study_design_scores_gemma":[0.00051864574,0.0008543474,0.0010634944,0.00016967216,0.00028163704,0.0011976317,0.0005434793,0.67248464,0.18659513,0.065351136,0.07068934,0.00025096256],"about_ca_topic_score_codex":0.003402582,"about_ca_topic_score_gemma":0.004037228,"teacher_disagreement_score":0.025341207,"about_ca_system_score_codex":0.0007660081,"about_ca_system_score_gemma":0.002220989,"threshold_uncertainty_score":0.08477479},"labels":[],"label_agreement":null},{"id":"W2145604837","doi":"","title":"Bilingual Sense Similarity for Statistical Machine Translation","year":2010,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Machine translation; Similarity (geometry); Example-based machine translation; Computer science; Translation (biology); Phrase; Artificial intelligence; Transfer-based machine translation; Natural language processing; Synchronous context-free grammar; Rule-based machine translation; Support vector machine; Machine translation software usability; Computer-assisted translation; Machine learning","score_opus":0.01755323051402968,"score_gpt":0.3061031573730617,"score_spread":0.288549926859032,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2145604837","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0045950864,0.0011194268,0.99039584,0.00019024654,0.00016609812,0.000059123096,0.00014681724,0.0007603892,0.0025669588],"genre_scores_gemma":[0.19579537,0.0013224605,0.79461616,0.0002867495,0.0006496755,0.0004152276,0.0018326845,0.0004969728,0.0045846817],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99599326,0.0018104597,0.00030332667,0.0007647013,0.001016988,0.00011121961],"domain_scores_gemma":[0.99581534,0.0019484892,0.0003172139,0.0009795526,0.0008317265,0.00010779669],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033158076,0.0007914877,0.0012591047,0.0032561712,0.001158029,0.0020821155,0.0012787008,0.0011003573,0.0049122632],"category_scores_gemma":[0.012128679,0.00039337698,0.0011709153,0.003633205,0.001403511,0.0041821543,0.0022537303,0.0013642001,0.0036583461],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00031706944,0.00014197096,0.0017556212,0.0004472287,0.00020519175,0.00020119266,0.0003553794,0.03721094,0.009027938,0.31968498,0.011998564,0.61865395],"study_design_scores_gemma":[0.00006330344,0.00019500748,0.0012606693,0.00008688715,0.00006977737,0.0006001529,0.00016464462,0.4402482,0.008066153,0.51651603,0.032641616,0.00008752837],"about_ca_topic_score_codex":0.0011132122,"about_ca_topic_score_gemma":0.001312332,"teacher_disagreement_score":0.0049122632,"about_ca_system_score_codex":0.0010475168,"about_ca_system_score_gemma":0.0012746565,"threshold_uncertainty_score":0.017535865},"labels":[],"label_agreement":null},{"id":"W2145752793","doi":"10.3115/1073416.1073421","title":"Indexing methods for efficient parsing","year":2003,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"University of Pennsylvania","keywords":"Search engine indexing; Parsing; Computer science; Grammar; Rule-based machine translation; Focus (optics); Natural language processing; Feature (linguistics); Artificial intelligence; Programming language; Information retrieval; Linguistics","score_opus":0.02912238448955648,"score_gpt":0.3811663802908275,"score_spread":0.352043995801271,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2145752793","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00058036787,0.00078860606,0.9938806,0.00010944723,0.00008827565,0.000050565504,0.0001236114,0.0026170271,0.0017615499],"genre_scores_gemma":[0.018101957,0.0015302168,0.9741305,0.00013189639,0.00034123604,0.00026283137,0.00089930644,0.0017229369,0.0028791367],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9951082,0.001317066,0.0004829484,0.001013106,0.0018433462,0.00023539417],"domain_scores_gemma":[0.9902705,0.0052584386,0.00044789532,0.0027518477,0.0011297637,0.00014162582],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005296976,0.0014550152,0.0014900728,0.003841834,0.001623331,0.003397694,0.0031614827,0.0014056353,0.014883303],"category_scores_gemma":[0.015612091,0.001049887,0.0016397552,0.006293363,0.0023062057,0.008080247,0.0027730153,0.0029785365,0.00790286],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009318697,0.00007582914,0.00032032,0.0005412429,0.0000552263,0.000097240154,0.00037836874,0.011146892,0.008277049,0.3173426,0.017822376,0.6438496],"study_design_scores_gemma":[0.00009107165,0.00009061067,0.0004997918,0.00019589551,0.000085413354,0.00050019665,0.00013201716,0.18070298,0.021621715,0.6398691,0.15608923,0.00012191418],"about_ca_topic_score_codex":0.0014856276,"about_ca_topic_score_gemma":0.001314647,"teacher_disagreement_score":0.014883303,"about_ca_system_score_codex":0.0013626829,"about_ca_system_score_gemma":0.0021249177,"threshold_uncertainty_score":0.049789608},"labels":[],"label_agreement":null},{"id":"W2145761652","doi":"","title":"Thomson Legal and Regulatory at NTCIR-5: Japanese and Korean Experiments","year":2005,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Thomson Reuters (Canada)","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Task (project management); Lexical analysis; Information retrieval; Relevance (law); Political science","score_opus":0.009487089254997538,"score_gpt":0.26461994563180835,"score_spread":0.2551328563768108,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2145761652","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8667736,0.0021616868,0.012846477,0.0025305673,0.0008593022,0.004409931,0.02294992,0.0031812792,0.084287286],"genre_scores_gemma":[0.816274,0.0008728587,0.04954209,0.0037267138,0.00036497583,0.0058888383,0.07299425,0.0013287527,0.04900748],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9958615,0.0020874795,0.0004827762,0.00048025238,0.00065291836,0.00043509525],"domain_scores_gemma":[0.99009305,0.005063598,0.00028503296,0.0015995587,0.0016757214,0.0012829879],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0065058228,0.0011579351,0.0012303342,0.0009238556,0.0028782797,0.0016056456,0.0014142463,0.0020861044,0.014681255],"category_scores_gemma":[0.014061955,0.0006110936,0.00058819685,0.0014529134,0.0009766495,0.0020836112,0.001975556,0.0015302349,0.0052752276],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.025426624,0.01681328,0.014945094,0.005830109,0.0007301776,0.0051459465,0.013373249,0.015113971,0.17831793,0.01831929,0.52759236,0.17839193],"study_design_scores_gemma":[0.021533612,0.013787864,0.10633818,0.00051530526,0.0012269578,0.004862672,0.015201419,0.07435716,0.15331896,0.017056804,0.59020436,0.0015967026],"about_ca_topic_score_codex":0.025568753,"about_ca_topic_score_gemma":0.037643936,"teacher_disagreement_score":0.025568753,"about_ca_system_score_codex":0.001335617,"about_ca_system_score_gemma":0.0024905282,"threshold_uncertainty_score":0.05083984},"labels":[],"label_agreement":null},{"id":"W2145765191","doi":"10.3115/1626355.1626379","title":"NRC's PORTAGE system for WMT 2007","year":2007,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":37,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"National Institute of Standards and Technology","keywords":"Machine translation; Computer science; Phrase; Pruning; Focus (optics); Natural language processing; Artificial intelligence; Feature (linguistics); Task (project management); Translation (biology); Table (database); Speech recognition; Machine translation software usability; Example-based machine translation; Data mining; Engineering; Linguistics","score_opus":0.010621329665503845,"score_gpt":0.2801154078517293,"score_spread":0.26949407818622545,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2145765191","genre_codex":"software","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0026663465,0.0008733559,0.19975623,0.0010873465,0.0016490706,0.00041679447,0.08971962,0.6609702,0.04286101],"genre_scores_gemma":[0.024127841,0.0012107141,0.35633758,0.0012017553,0.0008037005,0.0016293945,0.4405756,0.10956197,0.06455149],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99825925,0.00041518157,0.00019449445,0.00039304473,0.0006142988,0.00012374863],"domain_scores_gemma":[0.9971316,0.00051130744,0.00018029494,0.001365277,0.00062458223,0.00018697142],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029086454,0.0028738286,0.0022001842,0.004243386,0.0015596887,0.004695938,0.0045425375,0.0027856524,0.20223536],"category_scores_gemma":[0.011452823,0.0017136756,0.0018174207,0.0041458635,0.00062751636,0.0059908833,0.004783748,0.0031779394,0.19376656],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029425393,0.00007814462,0.00026635756,0.0004271478,0.0000810611,0.00032936392,0.00009768693,0.001572856,0.004118667,0.0059695668,0.8822293,0.10453549],"study_design_scores_gemma":[0.00031348065,0.00008018926,0.0005813591,0.00013014016,0.00007578277,0.000924104,0.000065165055,0.02169175,0.013233482,0.020847829,0.9418791,0.0001775837],"about_ca_topic_score_codex":0.0059355386,"about_ca_topic_score_gemma":0.0053832503,"teacher_disagreement_score":0.20223536,"about_ca_system_score_codex":0.0010994053,"about_ca_system_score_gemma":0.0022090643,"threshold_uncertainty_score":0.67654467},"labels":[],"label_agreement":null},{"id":"W2145779983","doi":"10.7202/019664ar","title":"Quantifying Phraseological Style in Two Modern Chinese Versions of Don Quijote","year":2009,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Phraseology; Stylistics; Stylometry; Style (visual arts); Linguistics; Writing style; Computer science; Literature; History; Art; Philosophy","score_opus":0.04748067431588807,"score_gpt":0.3381902227220878,"score_spread":0.2907095484061997,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2145779983","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9720354,0.00033276033,0.0015532337,0.0005091136,0.00013108997,0.000045115677,0.00011397302,0.000023192151,0.025256187],"genre_scores_gemma":[0.9929711,0.00017812932,0.0016089136,0.0000733976,0.00002907992,0.00003227519,0.000118196076,0.000025711468,0.0049630986],"study_design_codex":"qualitative","study_design_gemma":"not_applicable","domain_scores_codex":[0.9991831,0.00031543415,0.000071036164,0.00010680544,0.0002435799,0.00008008655],"domain_scores_gemma":[0.9981596,0.00092501886,0.00020220257,0.00024227728,0.0003646501,0.00010627439],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014089067,0.00043225632,0.00035114688,0.001923874,0.002039878,0.0017209911,0.00032552303,0.0004556533,0.0016675529],"category_scores_gemma":[0.005318649,0.0001750098,0.00017983172,0.0019042912,0.0037845639,0.0009364452,0.0012018639,0.0008722923,0.00016772661],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010327174,0.00016088445,0.037749477,0.0007440208,0.000071123344,0.0056209303,0.6682632,0.0006938523,0.019784283,0.14042991,0.0051766285,0.12027295],"study_design_scores_gemma":[0.00014966929,0.0011226566,0.44091624,0.0006797844,0.00018099348,0.006793807,0.24863808,0.0085746655,0.020269405,0.023246137,0.24912088,0.00030767807],"about_ca_topic_score_codex":0.010415472,"about_ca_topic_score_gemma":0.026958978,"teacher_disagreement_score":0.010415472,"about_ca_system_score_codex":0.0029428715,"about_ca_system_score_gemma":0.0011167421,"threshold_uncertainty_score":0.021352172},"labels":[],"label_agreement":null},{"id":"W2145944407","doi":"","title":"Evaluating Productivity Gains of Hybrid ASR-MT Systems for Translation Dictation","year":2008,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Dictation; Computer science; Productivity; Speech recognition; Word error rate; Vocabulary; Natural language processing; Machine translation; Word (group theory); Artificial intelligence; Linguistics","score_opus":0.09129635800624226,"score_gpt":0.35474404261066383,"score_spread":0.2634476846044216,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2145944407","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9788028,0.00049849163,0.01606475,0.00022328523,0.00006536069,0.00030070933,0.00018829417,0.00064047804,0.0032159132],"genre_scores_gemma":[0.97783864,0.0001558028,0.019478852,0.000062470324,0.00006012063,0.00021904432,0.00038025645,0.0000802408,0.0017244696],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99228376,0.0045919414,0.00069278345,0.00083036226,0.0012644582,0.00033658437],"domain_scores_gemma":[0.9478856,0.04129405,0.0017167785,0.003308409,0.0047696684,0.0010254808],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0086053675,0.0015028722,0.000996819,0.001023256,0.0008388594,0.0015350131,0.0014469809,0.001714817,0.0030276508],"category_scores_gemma":[0.03780491,0.00064266013,0.00046795586,0.0010095042,0.00089691154,0.0025432936,0.002099756,0.0009125151,0.0015183787],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.052065853,0.00797686,0.031018384,0.0024358388,0.0008855893,0.0011296935,0.004102225,0.17630816,0.10627076,0.0019035137,0.0030974196,0.6128057],"study_design_scores_gemma":[0.0064278306,0.0825172,0.08582473,0.00013711149,0.0012947066,0.0012773104,0.0036683278,0.67588645,0.1303549,0.0041244337,0.008074779,0.00041225363],"about_ca_topic_score_codex":0.0033542328,"about_ca_topic_score_gemma":0.0028625913,"teacher_disagreement_score":0.0086053675,"about_ca_system_score_codex":0.0012808728,"about_ca_system_score_gemma":0.0006332425,"threshold_uncertainty_score":0.045510054},"labels":[],"label_agreement":null},{"id":"W2146304057","doi":"10.18653/v1/w14-0146","title":"Registers in the System of Semantic Relations in plWordNet","year":2014,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Markedness; Register (sociolinguistics); Computer science; WordNet; Natural language processing; Consistency (knowledge bases); Artificial intelligence; Word (group theory); Position (finance); Tree (set theory); Linguistics; Mathematics","score_opus":0.009665228805945478,"score_gpt":0.2442048669206152,"score_spread":0.2345396381146697,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2146304057","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.054814413,0.00028678228,0.91529405,0.00089793385,0.00013135221,0.00031046534,0.0031262655,0.012104693,0.013033959],"genre_scores_gemma":[0.38080236,0.00037289556,0.6003644,0.00024730535,0.00008681001,0.00042554058,0.005978652,0.0016123413,0.010109716],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99839336,0.00045681727,0.0002287536,0.0005720673,0.00023914328,0.00010993286],"domain_scores_gemma":[0.9988238,0.00054541044,0.00012166856,0.00025197887,0.00020778178,0.00004936417],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021963103,0.0005545235,0.00066466775,0.002108608,0.0015902898,0.0034736537,0.0011827479,0.00088977686,0.0064367726],"category_scores_gemma":[0.004752398,0.00087586994,0.0008796494,0.0018006475,0.0021342034,0.0075777574,0.0017109009,0.0011908477,0.0025723272],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00052461895,0.00008932894,0.0036051772,0.0006711863,0.00008700528,0.0005947972,0.004072428,0.026875004,0.011965076,0.7426523,0.014517363,0.19434571],"study_design_scores_gemma":[0.00006682283,0.00009649653,0.0017141863,0.00021028693,0.00018585297,0.00044561666,0.00091041916,0.20268142,0.026517605,0.6021267,0.1649335,0.000111144065],"about_ca_topic_score_codex":0.005846506,"about_ca_topic_score_gemma":0.0074701374,"teacher_disagreement_score":0.0064367726,"about_ca_system_score_codex":0.0017581376,"about_ca_system_score_gemma":0.0015690454,"threshold_uncertainty_score":0.021533132},"labels":[],"label_agreement":null},{"id":"W2147565428","doi":"10.1023/a:1025720518994","title":"Extending Dublin Core Metadata to Support the Description and Discovery of Language Resources","year":2003,"lang":"en","type":"article","venue":"Computers and the Humanities","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":39,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Atomic Energy of Canada Limited; National Science Foundation","keywords":"Metadata; Computer science; World Wide Web; Reuse; Computational linguistics; Information retrieval; Natural language processing; Engineering","score_opus":0.039808824074941296,"score_gpt":0.26059815286093374,"score_spread":0.22078932878599244,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2147565428","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005074587,0.00031754444,0.98126894,0.0010573806,0.00014760204,0.0003105339,0.0021222073,0.004647688,0.005053642],"genre_scores_gemma":[0.0563843,0.00066857133,0.92405903,0.00062649784,0.000115696974,0.00039257426,0.010527727,0.0011096847,0.0061158924],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9943605,0.0017482876,0.0011064829,0.00073145557,0.0016528643,0.00040047546],"domain_scores_gemma":[0.9788615,0.008428821,0.0012676368,0.006549256,0.0040843715,0.00080849923],"candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.013139128,0.00069176365,0.0011173019,0.010554732,0.0024510731,0.007454319,0.0032989192,0.001559183,0.003715116],"category_scores_gemma":[0.025464054,0.0013284588,0.0019160734,0.0081276875,0.0023618473,0.01819097,0.0078585325,0.003280865,0.0024903978],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002302436,0.00030126178,0.0031510966,0.00095327216,0.00014203516,0.00057296205,0.0036121898,0.004795642,0.010922,0.5883151,0.029837111,0.35716692],"study_design_scores_gemma":[0.000098728815,0.00007446332,0.0010272005,0.00062968716,0.00020653282,0.0008429271,0.0012381762,0.09235163,0.027805198,0.550624,0.3248862,0.00021528592],"about_ca_topic_score_codex":0.018277463,"about_ca_topic_score_gemma":0.022152914,"teacher_disagreement_score":0.99254566,"about_ca_system_score_codex":0.0025100089,"about_ca_system_score_gemma":0.007820893,"threshold_uncertainty_score":0.069487214},"labels":[],"label_agreement":null},{"id":"W2147994465","doi":"10.21248/hpsg.2003.9","title":"A constraint-based approach to information structure and prosody correspondence","year":2003,"lang":"en","type":"article","venue":"Proceedings of the International Conference on Head-Driven Phrase Structure Grammar","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Constraint (computer-aided design); Grammar; Head-driven phrase structure grammar; Phonology; Syntax; Natural language processing; Information structure; Linguistics; Prosody; Point (geometry); Domain (mathematical analysis); Artificial intelligence; Architecture; Generative grammar; Programming language; Mathematics; Speech recognition; Geography","score_opus":0.016918505088204295,"score_gpt":0.26223912125932575,"score_spread":0.24532061617112144,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2147994465","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00804972,0.00078538357,0.9114325,0.003337767,0.00013104001,0.00007429592,0.00022256032,0.00025105238,0.07571569],"genre_scores_gemma":[0.60187674,0.0014432578,0.36981648,0.0014979215,0.00081244437,0.0005069641,0.00044089428,0.00049536576,0.023109946],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9973454,0.0008350573,0.00013852339,0.00051197316,0.0009510701,0.0002180202],"domain_scores_gemma":[0.99765223,0.0012582454,0.00018250395,0.00049009867,0.0003208477,0.0000959913],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024476873,0.0009292995,0.00081135664,0.00257214,0.001906445,0.004899676,0.0039420426,0.002962281,0.010436332],"category_scores_gemma":[0.005342937,0.0008599656,0.001522099,0.0028444782,0.0073625837,0.010764755,0.0036988452,0.0038801245,0.0011911023],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000055574474,0.0000050527988,0.000039593568,0.000013056697,0.0000056895647,0.000056498742,0.0001629166,0.0017297613,0.00026503103,0.99458534,0.0002704326,0.0028609585],"study_design_scores_gemma":[0.0000068785484,0.000011156567,0.000085639105,0.000009006551,0.00000720566,0.000076971366,0.000051150513,0.011189114,0.0004084339,0.981028,0.007114439,0.000012070444],"about_ca_topic_score_codex":0.0026287045,"about_ca_topic_score_gemma":0.0018493711,"teacher_disagreement_score":0.010436332,"about_ca_system_score_codex":0.0023251902,"about_ca_system_score_gemma":0.0017948861,"threshold_uncertainty_score":0.034913003},"labels":[],"label_agreement":null},{"id":"W2148362501","doi":"10.1017/s1351324904003560","title":"Correcting real-word spelling errors by restoring lexical cohesion","year":2005,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":180,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Spelling; Computer science; Cohesion (chemistry); Natural language processing; Lexicon; Artificial intelligence; Word (group theory); Context (archaeology); Precision and recall; Recall; Speech recognition; Linguistics","score_opus":0.005958882391220053,"score_gpt":0.2512037126170917,"score_spread":0.24524483022587162,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2148362501","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.52111965,0.0013645913,0.4521361,0.0006001048,0.0004623499,0.0003287003,0.0009890244,0.018496858,0.0045025805],"genre_scores_gemma":[0.611146,0.0004590046,0.38164377,0.00018862175,0.00011550092,0.00011125076,0.0015594782,0.0011624002,0.0036139956],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99766064,0.0003889921,0.00035226595,0.0006811218,0.00079342007,0.0001235423],"domain_scores_gemma":[0.9816126,0.0042293463,0.0035818592,0.0062554227,0.004041274,0.0002794037],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018421201,0.0010673602,0.0013143965,0.003202153,0.0008274391,0.0014600368,0.0011778628,0.0009905815,0.001872053],"category_scores_gemma":[0.020042837,0.00046553067,0.00058420375,0.0021450154,0.00083409343,0.002031855,0.0016834474,0.00093600544,0.0020656746],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00033972747,0.00030967648,0.01995013,0.0007794987,0.00015451158,0.0008543896,0.0013805039,0.008452491,0.1991465,0.0021699988,0.005659628,0.76080304],"study_design_scores_gemma":[0.00031438094,0.00095836207,0.06800993,0.00026335762,0.0007602337,0.0054619336,0.0016685543,0.16688253,0.6866637,0.018787794,0.049793266,0.0004359946],"about_ca_topic_score_codex":0.0022954163,"about_ca_topic_score_gemma":0.0038770663,"teacher_disagreement_score":0.003202153,"about_ca_system_score_codex":0.00040201415,"about_ca_system_score_gemma":0.0013486904,"threshold_uncertainty_score":0.0097422},"labels":[],"label_agreement":null},{"id":"W2148437670","doi":"10.18653/v1/2023.acl-long","title":"Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)","year":2023,"lang":"en","type":"paratext","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":354,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"Yuhan; Atomic Energy of Canada Limited; Strong; Styrelsen för Internationellt Utvecklingssamarbete","keywords":"Volume (thermodynamics); Computational linguistics; Association (psychology); Computer science; Linguistics; Natural language processing; Philosophy; Epistemology","score_opus":0.01096066040400236,"score_gpt":0.26998243307418723,"score_spread":0.2590217726701849,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2148437670","genre_codex":"editorial","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0074139684,0.16280502,0.030689044,0.18000588,0.34461412,0.0010261236,0.027876416,0.0076191314,0.23795037],"genre_scores_gemma":[0.020693708,0.091289304,0.033198502,0.039317265,0.065441884,0.001714496,0.047350664,0.008260616,0.69273365],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9941081,0.001849118,0.00089666515,0.0010953312,0.0016768503,0.00037401423],"domain_scores_gemma":[0.96318805,0.015369887,0.0015734769,0.0040252632,0.012138306,0.0037050224],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013482621,0.0012212716,0.002959837,0.0063473517,0.0030505876,0.013807651,0.0029628598,0.0037287502,0.2823546],"category_scores_gemma":[0.036319986,0.0011612738,0.0012426326,0.005593422,0.0021939091,0.011450453,0.004110031,0.0050996104,0.2549989],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000046594243,0.000036520276,0.0004200318,0.00037542035,0.000026883816,0.000053945627,0.00012036329,0.000023539336,0.00027843492,0.0009880605,0.94810003,0.0495302],"study_design_scores_gemma":[0.000014791443,0.000015448275,0.00092377746,0.00055855745,0.000018692735,0.00012120548,0.00021105293,0.00011981796,0.00009661205,0.0017565972,0.9961444,0.000018992192],"about_ca_topic_score_codex":0.0039730356,"about_ca_topic_score_gemma":0.0056934142,"teacher_disagreement_score":0.2823546,"about_ca_system_score_codex":0.002949964,"about_ca_system_score_gemma":0.0058260188,"threshold_uncertainty_score":0.9445702},"labels":[],"label_agreement":null},{"id":"W2148708444","doi":"10.1145/1149982.1149988","title":"A new top-down parsing algorithm to accommodate ambiguity and left recursion in polynomial time","year":2006,"lang":"en","type":"article","venue":"ACM SIGPLAN Notices","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Windsor","funders":"","keywords":"Computer science; Memoization; Recursion (computer science); Programming language; Backtracking; Time complexity; Parsing; Theoretical computer science; Algorithm; Modular design; Top-down parsing","score_opus":0.007901481672039504,"score_gpt":0.2541193857042048,"score_spread":0.24621790403216529,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2148708444","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0021068437,0.00013595218,0.9867801,0.00016252843,0.00010770027,0.00008543844,0.00014234873,0.007531522,0.0029475656],"genre_scores_gemma":[0.02730739,0.00015280633,0.963415,0.0002509107,0.00007123015,0.00015281652,0.0006431607,0.0014425454,0.0065641603],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9989532,0.00014214974,0.000101845544,0.0003121283,0.0003592222,0.000131361],"domain_scores_gemma":[0.9984877,0.0005155569,0.00006765609,0.000454973,0.00039581096,0.000078350655],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001001327,0.0013352425,0.001141701,0.001419201,0.0013895815,0.0023042185,0.0028415557,0.0014374885,0.010823564],"category_scores_gemma":[0.0035779695,0.00096729875,0.0015833299,0.001780249,0.0010775498,0.0040012226,0.0029802916,0.0025801065,0.0057896064],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023947921,0.00025061928,0.0006134516,0.0004035156,0.0000812949,0.00035370784,0.0004961618,0.026692718,0.030273868,0.08314376,0.05172893,0.8057225],"study_design_scores_gemma":[0.00020441676,0.00016475866,0.00038518966,0.00009715562,0.00017183504,0.00083353254,0.00015169785,0.6944884,0.036109917,0.18217747,0.08508892,0.00012672726],"about_ca_topic_score_codex":0.0031747357,"about_ca_topic_score_gemma":0.0050626155,"teacher_disagreement_score":0.010823564,"about_ca_system_score_codex":0.0008968825,"about_ca_system_score_gemma":0.002582364,"threshold_uncertainty_score":0.03620845},"labels":[],"label_agreement":null},{"id":"W2148861208","doi":"10.3115/1626355.1626366","title":"Using word dependent transition models in HMM based word alignment for statistical machine translation","year":2007,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":54,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Hidden Markov model; Word (group theory); Word error rate; Machine translation; Phrase; Artificial intelligence; Speech recognition; Natural language processing; Translation (biology); IBM; Transition (genetics); Bayesian probability; Mathematics","score_opus":0.047580834872230804,"score_gpt":0.32768718252808765,"score_spread":0.28010634765585685,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2148861208","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005300031,0.00029751554,0.9922214,0.0001075535,0.00005125933,0.000033954842,0.0001119232,0.0013669514,0.0005094376],"genre_scores_gemma":[0.26236445,0.00095044053,0.72985035,0.0002822565,0.00013321693,0.00041488503,0.0014933703,0.00073348486,0.003777591],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9981285,0.0009915823,0.00010443081,0.000345276,0.00035215882,0.00007797852],"domain_scores_gemma":[0.9960135,0.0029899883,0.00019443092,0.0002896675,0.00044868907,0.000063617794],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022615357,0.0007409987,0.0008963157,0.0008148995,0.00061493064,0.00092318683,0.0010151273,0.0010765961,0.0025158692],"category_scores_gemma":[0.009800437,0.0007511907,0.00070367765,0.0011085216,0.00050600636,0.0023027346,0.00085850374,0.0019945,0.0028964132],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005476289,0.00023877683,0.0028488687,0.0003827314,0.00031194655,0.00020113854,0.0004601227,0.39403912,0.024802139,0.015773328,0.0045313286,0.5558629],"study_design_scores_gemma":[0.000030689673,0.000072506904,0.0006389659,0.000030199848,0.000053666805,0.00006649541,0.000027929093,0.97751826,0.0059785377,0.011738565,0.0037926792,0.000051474653],"about_ca_topic_score_codex":0.008311309,"about_ca_topic_score_gemma":0.017119706,"teacher_disagreement_score":0.008311309,"about_ca_system_score_codex":0.00074811635,"about_ca_system_score_gemma":0.0015261263,"threshold_uncertainty_score":0.016525865},"labels":[],"label_agreement":null},{"id":"W2148884131","doi":"10.7202/012246ar","title":"Corpus issus du Web : constitution et analyse informationnelle","year":2006,"lang":"fr","type":"article","venue":"Revue québécoise de linguistique","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Art","score_opus":0.010282154909177553,"score_gpt":0.2674369991668808,"score_spread":0.25715484425770324,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2148884131","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.120919526,0.0029814458,0.8304743,0.0015564733,0.00034663285,0.0007966308,0.009233178,0.0069857812,0.026706092],"genre_scores_gemma":[0.25173736,0.0020781092,0.7138466,0.00021371947,0.00016769212,0.0016634008,0.012959139,0.0025677746,0.014766177],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9953472,0.0015675045,0.000500369,0.00088841096,0.0015225832,0.00017394839],"domain_scores_gemma":[0.985636,0.009586221,0.0005183226,0.0017148597,0.0023504368,0.00019418642],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003320955,0.0006582268,0.0009203215,0.008049174,0.0017671149,0.006988488,0.0010347841,0.0010466464,0.005901715],"category_scores_gemma":[0.019977523,0.0007648293,0.0008658878,0.007957676,0.0017480346,0.0043953634,0.0019817925,0.0016122552,0.002827458],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007897967,0.00021475976,0.011472093,0.002293747,0.00026340256,0.0014120137,0.01703902,0.009106996,0.10330059,0.10049804,0.01895814,0.73465145],"study_design_scores_gemma":[0.00015238867,0.00031885566,0.03477171,0.0009635515,0.00047742995,0.003687782,0.010472963,0.16102959,0.19866157,0.09936606,0.48978338,0.00031476936],"about_ca_topic_score_codex":0.009219183,"about_ca_topic_score_gemma":0.007670515,"teacher_disagreement_score":0.009219183,"about_ca_system_score_codex":0.0014273184,"about_ca_system_score_gemma":0.0025031643,"threshold_uncertainty_score":0.019743264},"labels":[],"label_agreement":null},{"id":"W2149013336","doi":"","title":"ICML2011 Unsupervised and Transfer Learning Workshop","year":2011,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Acadia University","funders":"","keywords":"Transfer of learning; Computer science; Unsupervised learning; Task (project management); Artificial intelligence; Similarity (geometry); Feature (linguistics); Deep learning; Machine learning; Feature learning; Data science; Engineering; Image (mathematics)","score_opus":0.030118580053664343,"score_gpt":0.24268387070160657,"score_spread":0.21256529064794222,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2149013336","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012883994,0.013517745,0.7249477,0.099542774,0.05162235,0.0009907587,0.017498791,0.032122552,0.046873443],"genre_scores_gemma":[0.07404323,0.008765444,0.4874743,0.01840198,0.019503046,0.0020908408,0.11050432,0.008647777,0.2705691],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9895091,0.004040551,0.00048759827,0.0019940166,0.0030362697,0.00093243923],"domain_scores_gemma":[0.98413944,0.004958862,0.00025824673,0.0029361546,0.0050605475,0.0026467484],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.018938,0.0033350792,0.0039064856,0.0027968544,0.002445888,0.007621469,0.0072607594,0.006376591,0.03099311],"category_scores_gemma":[0.024718411,0.0009834456,0.002764088,0.0026780816,0.002603038,0.008384322,0.006793879,0.008421351,0.025754476],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002846504,0.00028566364,0.00025415982,0.00021006126,0.00007646655,0.00019638265,0.00011769252,0.005531984,0.0014951527,0.0056433366,0.7720762,0.21382825],"study_design_scores_gemma":[0.00022606543,0.00033835584,0.0014079537,0.00030071038,0.00009730179,0.00051002624,0.0005110231,0.13889447,0.0122621525,0.04773133,0.7975545,0.00016613332],"about_ca_topic_score_codex":0.016106244,"about_ca_topic_score_gemma":0.025288211,"teacher_disagreement_score":0.03099311,"about_ca_system_score_codex":0.0047769994,"about_ca_system_score_gemma":0.007211416,"threshold_uncertainty_score":0.10368228},"labels":[],"label_agreement":null},{"id":"W2149459752","doi":"","title":"Probabilistic relaxed unification formalism and its application in question answering","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Unification; Correctness; Computer science; Theoretical computer science; Formalism (music); Probabilistic logic; Algorithm; Artificial intelligence; Programming language","score_opus":0.013909213755517623,"score_gpt":0.2582510057561088,"score_spread":0.2443417920005912,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2149459752","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0030900214,0.00024960868,0.99439865,0.0003552396,0.000037710513,0.000100951096,0.000099428835,0.00057208043,0.0010962602],"genre_scores_gemma":[0.15434515,0.0005567091,0.8405237,0.0004402372,0.00021705613,0.00047941192,0.0008863229,0.00036286924,0.0021884649],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9808091,0.009356048,0.0015448718,0.0037234658,0.003809225,0.000757384],"domain_scores_gemma":[0.9660678,0.023944644,0.0020230575,0.005123036,0.002424504,0.00041687916],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.017172989,0.0013272922,0.0018237602,0.006167805,0.0024377098,0.0046248906,0.004649806,0.0029263676,0.0046697324],"category_scores_gemma":[0.050293576,0.0015324316,0.0052194744,0.00613294,0.005721606,0.013445081,0.0069624046,0.0055236174,0.0011745179],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015585015,0.00010467986,0.0012250933,0.00044976373,0.00015615008,0.00044611594,0.0019224004,0.11719913,0.004047401,0.76070577,0.0025089143,0.11107893],"study_design_scores_gemma":[0.000022535647,0.00004238728,0.0002452265,0.000077907665,0.000051213785,0.00016297003,0.0001191923,0.36954668,0.0025986128,0.6196545,0.0074292747,0.00004948293],"about_ca_topic_score_codex":0.0077909864,"about_ca_topic_score_gemma":0.004469874,"teacher_disagreement_score":0.017172989,"about_ca_system_score_codex":0.0037643586,"about_ca_system_score_gemma":0.0027889567,"threshold_uncertainty_score":0.09082055},"labels":[],"label_agreement":null},{"id":"W2150180307","doi":"10.3115/v1/d14-1028","title":"Two Improvements to Left-to-Right Decoding for Hierarchical Phrase-based Machine Translation","year":2014,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Decoding methods; Phrase; Machine translation; Computer science; Translation (biology); Speech recognition; Order (exchange); Algorithm; Natural language processing; Artificial intelligence","score_opus":0.01471255471474986,"score_gpt":0.3031092586100726,"score_spread":0.28839670389532274,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2150180307","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006463361,0.0005635436,0.9788617,0.00038695917,0.0003101094,0.00019077955,0.00038000089,0.0096192965,0.0032241263],"genre_scores_gemma":[0.054826315,0.0004395462,0.93301904,0.00035639282,0.00022519457,0.00022447028,0.0018115017,0.0018566155,0.00724095],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9954308,0.0011634917,0.00045939945,0.0009605625,0.0016128145,0.000373018],"domain_scores_gemma":[0.9936568,0.0014130725,0.00028390074,0.0025030016,0.0019342105,0.00020898262],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022258086,0.0022931232,0.0018904022,0.0017099808,0.0013549842,0.0020651477,0.0019881818,0.0023205657,0.0096317],"category_scores_gemma":[0.009928438,0.0010071237,0.0016972293,0.0019444264,0.0012824684,0.0040863636,0.0035794808,0.0057366565,0.011675505],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005789038,0.00048805246,0.0011060355,0.00056959107,0.000128471,0.00035484874,0.000560173,0.028376112,0.09623145,0.02093277,0.023920517,0.8267531],"study_design_scores_gemma":[0.00028763866,0.0006085551,0.0019201539,0.000104307546,0.0002360563,0.0015889029,0.00022558014,0.7277398,0.16945085,0.0189548,0.07850813,0.00037516764],"about_ca_topic_score_codex":0.007027131,"about_ca_topic_score_gemma":0.012448206,"teacher_disagreement_score":0.0096317,"about_ca_system_score_codex":0.0009921598,"about_ca_system_score_gemma":0.002902628,"threshold_uncertainty_score":0.032221198},"labels":[],"label_agreement":null},{"id":"W2150417504","doi":"","title":"Creating Robust Supervised Classifiers via Web-Scale N-Gram Data","year":2010,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"n-gram; Bracketing (phenomenology); Computer science; Artificial intelligence; Scale (ratio); Natural language processing; Gram; Noun; Verb; Speech recognition; Language model","score_opus":0.03026958532982212,"score_gpt":0.28012856700393785,"score_spread":0.24985898167411572,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2150417504","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.24100105,0.0016641995,0.7258127,0.00082525867,0.0006202235,0.00060630956,0.0035772298,0.017407969,0.008485139],"genre_scores_gemma":[0.5343362,0.00040296532,0.44897696,0.00056773424,0.0003758576,0.0007261612,0.010817608,0.00058846886,0.003207994],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99561054,0.001707724,0.00028742888,0.00094868854,0.0012224064,0.00022325777],"domain_scores_gemma":[0.9827632,0.010216678,0.0010913254,0.0029857121,0.0026313034,0.00031188986],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0048139677,0.001312297,0.0017130368,0.001995189,0.0010507406,0.001668281,0.0015357074,0.0020094162,0.0023069617],"category_scores_gemma":[0.020990266,0.00037738797,0.00076614413,0.0017072433,0.00069343887,0.0051185205,0.0018992136,0.0020281002,0.0048062433],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010516035,0.0010189429,0.012814496,0.000509056,0.00027515827,0.00032083504,0.0001575016,0.09088639,0.022873662,0.0030182581,0.018635223,0.8484389],"study_design_scores_gemma":[0.000066526154,0.00019876199,0.0021557312,0.00005308983,0.00005281804,0.00017720152,0.00017014852,0.9505898,0.032384597,0.007523115,0.006582401,0.00004581542],"about_ca_topic_score_codex":0.0012908188,"about_ca_topic_score_gemma":0.00284807,"teacher_disagreement_score":0.0048139677,"about_ca_system_score_codex":0.0005969555,"about_ca_system_score_gemma":0.0013063449,"threshold_uncertainty_score":0.025458992},"labels":[],"label_agreement":null},{"id":"W2150452164","doi":"10.3115/1218955.1218984","title":"Optimizing typed feature structure grammar parsing through non-statistical indexing","year":2004,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Parsing; Search engine indexing; Feature (linguistics); Grammar; Artificial intelligence; Unification; Natural language processing; Compiler; Rule-based machine translation; Programming language","score_opus":0.01041816187345583,"score_gpt":0.2756423069808692,"score_spread":0.2652241451074134,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2150452164","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0429425,0.00016893166,0.9181837,0.00016799405,0.000046924095,0.00008749595,0.00045491196,0.03605638,0.0018911953],"genre_scores_gemma":[0.299618,0.00015857133,0.6892504,0.00015051164,0.00007758027,0.00016422867,0.0028391406,0.005433415,0.0023081424],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9975915,0.0005561109,0.00022784754,0.00058634824,0.00079199113,0.00024619472],"domain_scores_gemma":[0.9934596,0.003574829,0.00044463674,0.001724608,0.00068415434,0.00011220156],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020158219,0.0010902559,0.00154651,0.0012921497,0.00070386915,0.0016625165,0.0019887753,0.0006427042,0.0030452355],"category_scores_gemma":[0.0075769937,0.00088760536,0.0012281517,0.0022105565,0.001123414,0.0037846575,0.0015439206,0.0014053889,0.0016425021],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00048408646,0.00034529367,0.0066511724,0.00031525665,0.00012339317,0.00027721454,0.00048414664,0.11501251,0.06480815,0.026217377,0.008980432,0.77630097],"study_design_scores_gemma":[0.00008177357,0.0001276084,0.0023539313,0.000018139603,0.00009358082,0.00019194875,0.00015294268,0.8800368,0.06363421,0.04629414,0.0069386223,0.00007639513],"about_ca_topic_score_codex":0.004803441,"about_ca_topic_score_gemma":0.0070237448,"teacher_disagreement_score":0.004803441,"about_ca_system_score_codex":0.0012664432,"about_ca_system_score_gemma":0.00356933,"threshold_uncertainty_score":0.010660827},"labels":[],"label_agreement":null},{"id":"W2150533253","doi":"10.7202/002578ar","title":"Traduire l’annuaire","year":2002,"lang":"fr","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Humanities; Art; Philosophy","score_opus":0.06662537081594293,"score_gpt":0.28203535285127046,"score_spread":0.21540998203532752,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2150533253","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14412272,0.0018770244,0.19401672,0.0023958038,0.002430276,0.00019149284,0.0016066943,0.0029885236,0.6503707],"genre_scores_gemma":[0.5114469,0.0015145339,0.098694965,0.00051130855,0.00036628346,0.000063910615,0.0010772765,0.002241352,0.38408345],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9992028,0.00011014292,0.000042383126,0.00019343059,0.0003487839,0.00010253426],"domain_scores_gemma":[0.99813193,0.0003640809,0.00008784087,0.0007320767,0.0006203404,0.00006371044],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009398933,0.00048964587,0.00025510808,0.00095123326,0.0026784572,0.0028816017,0.0010962146,0.0007386253,0.032651424],"category_scores_gemma":[0.002855908,0.0003229832,0.00031103988,0.0012633529,0.0020816044,0.0020295542,0.0011977782,0.0013028218,0.011790842],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029694982,0.000057335026,0.006396612,0.0003943756,0.000019269923,0.00162984,0.020441657,0.0028955415,0.05228753,0.41051766,0.039221518,0.46584174],"study_design_scores_gemma":[0.000008859263,0.000051213454,0.0025818555,0.00011833335,0.00001598622,0.0006419711,0.0034787871,0.0019096679,0.020785945,0.010296656,0.960073,0.00003765828],"about_ca_topic_score_codex":0.05745795,"about_ca_topic_score_gemma":0.07928905,"teacher_disagreement_score":0.05745795,"about_ca_system_score_codex":0.0027072025,"about_ca_system_score_gemma":0.003593034,"threshold_uncertainty_score":0.114247024},"labels":[],"label_agreement":null},{"id":"W2150639765","doi":"","title":"Generating Update Summaries for DUC 2007","year":2007,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Task (project management); Coreference; Set (abstract data type); Information retrieval; Artificial intelligence; Data mining; Natural language processing; Programming language; Resolution (logic)","score_opus":0.013371397173664815,"score_gpt":0.28932689623282565,"score_spread":0.27595549905916084,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2150639765","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16323487,0.0035518033,0.5673331,0.0013924383,0.0014547936,0.0012283708,0.11813738,0.12246314,0.021204123],"genre_scores_gemma":[0.24093375,0.00072291674,0.54141057,0.00019838165,0.00023721946,0.0005289673,0.19763751,0.0046755406,0.013655136],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987954,0.00033461527,0.00012273972,0.00029255886,0.00038825863,0.00006634491],"domain_scores_gemma":[0.9963678,0.0010025244,0.0002239261,0.000681136,0.0016125463,0.00011215011],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011677302,0.0011982402,0.0006468438,0.0026334529,0.00057471445,0.0011461617,0.0008609187,0.00087754807,0.005805443],"category_scores_gemma":[0.00800126,0.00041905127,0.00047513092,0.0019059995,0.00016485655,0.001186601,0.00074017537,0.0006375347,0.0025470634],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011322239,0.00017147472,0.003371531,0.001073884,0.00029457523,0.000814809,0.0008990075,0.033349894,0.027685506,0.006871294,0.3270284,0.5973074],"study_design_scores_gemma":[0.00040688744,0.00085913856,0.009137301,0.0001331215,0.0004442825,0.001217323,0.0009422595,0.4916135,0.105784096,0.0172521,0.37199968,0.00021031559],"about_ca_topic_score_codex":0.006612956,"about_ca_topic_score_gemma":0.013883194,"teacher_disagreement_score":0.006612956,"about_ca_system_score_codex":0.00081388146,"about_ca_system_score_gemma":0.00071691285,"threshold_uncertainty_score":0.0194211},"labels":[],"label_agreement":null},{"id":"W215104783","doi":"","title":"La mise en evidence textuelle: d'ou venons nous et ou allons nous (Textual Enhancement: Where Have We Been and Where Are We Going)?.","year":2002,"lang":"fr","type":"article","venue":"Canadian Modern Language Review/ La Revue canadienne des langues vivantes","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Philosophy","score_opus":0.0322725252733055,"score_gpt":0.27180305860800297,"score_spread":0.23953053333469748,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W215104783","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024733733,0.35729343,0.14462186,0.38641027,0.014052818,0.0002628147,0.0030659968,0.0020271032,0.06753205],"genre_scores_gemma":[0.35990506,0.23994973,0.2615229,0.042595245,0.012980779,0.00034614035,0.002871058,0.0011498675,0.07867926],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.98881125,0.006048924,0.0009031481,0.0009373225,0.0029786173,0.00032076915],"domain_scores_gemma":[0.90740216,0.05790748,0.006169427,0.0044779065,0.022273073,0.0017699768],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.025317224,0.0006073803,0.00097380683,0.005040354,0.002075329,0.010284073,0.002149526,0.0034776242,0.01079523],"category_scores_gemma":[0.09868684,0.0004833185,0.0004814216,0.0031228482,0.0063112057,0.015742451,0.0022970724,0.0036238732,0.0051695504],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004532191,0.000063213774,0.0038501767,0.004873248,0.00014281995,0.00082930736,0.0060451375,0.0003003658,0.006014579,0.064750016,0.1542984,0.7583796],"study_design_scores_gemma":[0.00006122361,0.000098309254,0.0051781605,0.0054401434,0.00023302277,0.002087986,0.0099104885,0.0014516168,0.0093845865,0.06094235,0.90509415,0.00011784426],"about_ca_topic_score_codex":0.017880447,"about_ca_topic_score_gemma":0.032264367,"teacher_disagreement_score":0.025317224,"about_ca_system_score_codex":0.004093966,"about_ca_system_score_gemma":0.009200181,"threshold_uncertainty_score":0.13389188},"labels":[],"label_agreement":null},{"id":"W2151594415","doi":"","title":"Mixing Multiple Translation Models in Statistical Machine Translation","year":2012,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"National Research Council Canada; Simon Fraser University","funders":"","keywords":"Machine translation; Computer science; Decoding methods; Translation (biology); Domain adaptation; Artificial intelligence; Domain (mathematical analysis); Transfer-based machine translation; Example-based machine translation; Natural language processing; Statistical model; Adaptation (eye); Machine learning; Algorithm; Mathematics","score_opus":0.036381745943648895,"score_gpt":0.29026661610910476,"score_spread":0.25388487016545586,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2151594415","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0033127756,0.0017225062,0.9928623,0.0003890316,0.000094418276,0.00003643893,0.00007094299,0.0006523889,0.00085926196],"genre_scores_gemma":[0.20957595,0.0044304035,0.7783551,0.0005379492,0.00075359823,0.00044148145,0.0010062854,0.0006088004,0.004290349],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9951361,0.0033652126,0.00017644024,0.00060259894,0.0006138323,0.000105959945],"domain_scores_gemma":[0.99164927,0.0065606227,0.00033002018,0.0007993736,0.00055511075,0.000105537234],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0056099435,0.00165078,0.0019288696,0.0015232394,0.00094698183,0.0019516151,0.0015172939,0.0028340183,0.002075414],"category_scores_gemma":[0.016551996,0.0011971308,0.0014142003,0.0035922762,0.0015059449,0.0041468837,0.002361262,0.003814559,0.002515631],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00027507712,0.00014333159,0.0011987651,0.00037830285,0.00035252827,0.00028366368,0.00043644352,0.5936861,0.003946968,0.085918725,0.0064659934,0.30691406],"study_design_scores_gemma":[0.000021863767,0.000062534644,0.0001408951,0.000031176376,0.000042631258,0.000105251274,0.000021580934,0.901599,0.0017668761,0.091782466,0.004387453,0.000038291022],"about_ca_topic_score_codex":0.0029301278,"about_ca_topic_score_gemma":0.003406387,"teacher_disagreement_score":0.0056099435,"about_ca_system_score_codex":0.000850984,"about_ca_system_score_gemma":0.0010700189,"threshold_uncertainty_score":0.02966857},"labels":[],"label_agreement":null},{"id":"W2151839123","doi":"","title":"Translating Unknown Words by Analogical Learning","year":2007,"lang":"en","type":"article","venue":"Empirical Methods in Natural Language Processing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":63,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Lexicon; Computer science; Artificial intelligence; Natural language processing; Machine translation; Quality (philosophy); Translation (biology)","score_opus":0.030972761303358447,"score_gpt":0.432158857636875,"score_spread":0.40118609633351654,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2151839123","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.034192465,0.00054030673,0.9551979,0.0008488593,0.00016161577,0.0001424862,0.00013682236,0.0009328793,0.007846642],"genre_scores_gemma":[0.4625249,0.00074032205,0.52742946,0.00085273053,0.0003532822,0.00032793323,0.00088997855,0.0002751131,0.0066063865],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9972542,0.0012379753,0.00017742711,0.00081569003,0.00043627067,0.00007846116],"domain_scores_gemma":[0.9905122,0.0070236046,0.00048737976,0.0014352339,0.00045596415,0.00008548981],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023345472,0.0009315308,0.0008510428,0.0015251262,0.0008260175,0.0020360027,0.0018016848,0.0015343847,0.009029192],"category_scores_gemma":[0.019955331,0.0005088205,0.0011958532,0.001564461,0.0026206768,0.0051969215,0.0028493379,0.002304941,0.0021275887],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023352476,0.0003457711,0.0024854492,0.00069655303,0.00021412644,0.00052726414,0.00091829116,0.06547508,0.0072395736,0.18602836,0.007289914,0.728546],"study_design_scores_gemma":[0.000096751624,0.00012731424,0.0007754393,0.00006335229,0.000053424086,0.00042470597,0.00025733592,0.3256868,0.004840369,0.65792274,0.009701287,0.000050438233],"about_ca_topic_score_codex":0.0008260214,"about_ca_topic_score_gemma":0.0011055972,"teacher_disagreement_score":0.009029192,"about_ca_system_score_codex":0.0007952209,"about_ca_system_score_gemma":0.0007048314,"threshold_uncertainty_score":0.030205607},"labels":[],"label_agreement":null},{"id":"W2152249239","doi":"","title":"Combining Morpheme-based Machine Translation with Post-processing Morpheme Prediction","year":2011,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":35,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Morpheme; Computer science; Natural language processing; Phrase; Artificial intelligence; Machine translation; Focus (optics); Translation (biology); Task (project management); Word (group theory); Linguistics; Engineering","score_opus":0.024535854372568697,"score_gpt":0.23278380012725514,"score_spread":0.20824794575468644,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2152249239","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01604509,0.00030126644,0.97529817,0.00013831117,0.00008477245,0.000060873546,0.000108620814,0.005954803,0.0020081338],"genre_scores_gemma":[0.32511362,0.00032516132,0.66812545,0.00019098229,0.00020516412,0.00012842675,0.0009174344,0.00076199515,0.0042317268],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992906,0.00017754643,0.00006142831,0.0002343718,0.0001854527,0.000050508497],"domain_scores_gemma":[0.9983006,0.0007865503,0.00010940412,0.0003479921,0.00041703496,0.000038454447],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009819064,0.0010508347,0.00086714275,0.0009888927,0.00044005175,0.0009496925,0.0011442243,0.00077851576,0.0030330645],"category_scores_gemma":[0.0029760369,0.00047413516,0.00069388247,0.0009898536,0.0003806901,0.0014200934,0.000804404,0.0010135281,0.0033842416],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002637779,0.00019589339,0.001331953,0.0001751939,0.00016925407,0.00015345143,0.00009015801,0.049398255,0.07359796,0.002969633,0.0029527429,0.86870164],"study_design_scores_gemma":[0.00004771628,0.00021834785,0.0015687579,0.000020530699,0.00011579281,0.00026510778,0.000037475445,0.88993376,0.08931369,0.012112975,0.006310737,0.000055113556],"about_ca_topic_score_codex":0.0014401309,"about_ca_topic_score_gemma":0.002867918,"teacher_disagreement_score":0.0030330645,"about_ca_system_score_codex":0.00030084714,"about_ca_system_score_gemma":0.0006274757,"threshold_uncertainty_score":0.010146618},"labels":[],"label_agreement":null},{"id":"W2152382718","doi":"10.3115/1626431.1626478","title":"Stabilizing minimum error rate training","year":2009,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":46,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Computer science; Stability (learning theory); Word error rate; Variation (astronomy); Training set; Feature (linguistics); Test data; Component (thermodynamics); Machine translation; Translation (biology); Artificial intelligence; Test (biology); Machine learning; Training (meteorology); Algorithm; Pattern recognition (psychology)","score_opus":0.03676422810836918,"score_gpt":0.3020566031008958,"score_spread":0.26529237499252667,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2152382718","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02953872,0.00022680349,0.9653105,0.00016505609,0.000072064904,0.00010550239,0.00009633348,0.002564058,0.0019210235],"genre_scores_gemma":[0.5118444,0.00016927702,0.4814749,0.00030166603,0.00009256122,0.00061434176,0.0008324355,0.0009702002,0.0037002072],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99521756,0.002099339,0.0002696778,0.0010359797,0.001077423,0.00029997726],"domain_scores_gemma":[0.98434573,0.008664551,0.00079541176,0.002805216,0.003203086,0.00018592544],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007005925,0.001239106,0.0015740553,0.0009489725,0.0009660375,0.0011076913,0.0023842244,0.0017708475,0.0027280361],"category_scores_gemma":[0.04097695,0.0006672493,0.0007532617,0.0009818812,0.0012338596,0.0017689992,0.0019807392,0.0024713203,0.0022195734],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007561571,0.0002740864,0.0044742357,0.00024801088,0.00015342922,0.00016765772,0.000367738,0.5660141,0.018732775,0.016512215,0.0061828177,0.3861167],"study_design_scores_gemma":[0.000033555807,0.00012849708,0.00058987754,0.000022489001,0.000017777444,0.000079689984,0.000027428829,0.97847223,0.012808489,0.0066792914,0.0011243746,0.000016214755],"about_ca_topic_score_codex":0.0020517723,"about_ca_topic_score_gemma":0.0024772272,"teacher_disagreement_score":0.007005925,"about_ca_system_score_codex":0.0008007658,"about_ca_system_score_gemma":0.0014675963,"threshold_uncertainty_score":0.03705132},"labels":[],"label_agreement":null},{"id":"W2152511883","doi":"10.16995/dscn.131","title":"The eastcree.org Web Databases: Participatory Action Research with Information Technology","year":2009,"lang":"en","type":"article","venue":"Digital Studies / Le champ numérique","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Humanities; Political science; Context (archaeology); Library science; Computer science; Art; Geography","score_opus":0.11541103219043575,"score_gpt":0.3820122702572213,"score_spread":0.26660123806678554,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2152511883","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0741127,0.0036436971,0.14184727,0.11079776,0.0019303053,0.0077094343,0.013342135,0.0052289506,0.64138776],"genre_scores_gemma":[0.3936309,0.0028032227,0.1641084,0.009801493,0.00030088375,0.009925415,0.006264748,0.0032367848,0.40992823],"study_design_codex":"design_other","study_design_gemma":"qualitative","domain_scores_codex":[0.98143023,0.012435205,0.00083931786,0.0009925903,0.0036275361,0.0006750637],"domain_scores_gemma":[0.91465974,0.044404663,0.0020211136,0.017164864,0.011838414,0.00991121],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03494659,0.0002566335,0.00046803732,0.001441019,0.0021955904,0.0062937927,0.0018937016,0.0013441138,0.054114364],"category_scores_gemma":[0.0631592,0.0005042153,0.00028729063,0.0020794936,0.0019820528,0.0063650087,0.004799274,0.0014231482,0.010102053],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012118125,0.0005778167,0.0076491684,0.0017389846,0.000055367153,0.0003010184,0.021982064,0.0005510138,0.002644169,0.09566182,0.32889035,0.5387363],"study_design_scores_gemma":[0.00027004257,0.00020286007,0.005323535,0.0010302614,0.00003523512,0.00015551136,0.011689181,0.0014155643,0.003677261,0.014931313,0.96121246,0.000056690707],"about_ca_topic_score_codex":0.008522425,"about_ca_topic_score_gemma":0.022238582,"teacher_disagreement_score":0.054114364,"about_ca_system_score_codex":0.002894932,"about_ca_system_score_gemma":0.012852399,"threshold_uncertainty_score":0.18481743},"labels":[],"label_agreement":null},{"id":"W2152808281","doi":"10.1109/tnn.2007.912312","title":"Adaptive Importance Sampling to Accelerate Training of a Neural Probabilistic Language Model","year":2008,"lang":"en","type":"article","venue":"IEEE Transactions on Neural Networks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":233,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Language model; Artificial neural network; Vocabulary; Probabilistic neural network; Speedup; Artificial intelligence; Probabilistic logic; Computation; Machine learning; Feedforward neural network; Sampling (signal processing); Statistical model; Training (meteorology); Backpropagation; Importance sampling; Time delay neural network; Algorithm; Statistics; Mathematics; Monte Carlo method","score_opus":0.0677840225644307,"score_gpt":0.29191262992274414,"score_spread":0.22412860735831344,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2152808281","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01263745,0.000106726606,0.9860241,0.00009055774,0.00003229607,0.000022184575,0.000015996347,0.00069803634,0.0003725924],"genre_scores_gemma":[0.35840714,0.0001908351,0.6390409,0.00012955524,0.00010771505,0.0001352105,0.00020670905,0.00019037991,0.0015915997],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99938524,0.0002446672,0.000030644253,0.00009257408,0.00018697378,0.000059931626],"domain_scores_gemma":[0.9979785,0.0014216742,0.00007275648,0.00021424243,0.00025057848,0.00006222086],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015853784,0.0005938052,0.00072304794,0.0004931991,0.00028706284,0.00041603102,0.0014401644,0.00072441524,0.0019078827],"category_scores_gemma":[0.006871984,0.0005024433,0.00045091478,0.00061859016,0.00047970153,0.0015714638,0.0010470898,0.0018119178,0.00052977377],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002780461,0.00017043413,0.0011646148,0.000099629644,0.000053805757,0.00014577892,0.00012918474,0.6906656,0.0137467645,0.022577131,0.0028464831,0.26812258],"study_design_scores_gemma":[0.0000039422334,0.0000070909055,0.00003231261,8.3688565e-7,0.0000015576994,0.00000713231,0.0000014172009,0.997331,0.00070447277,0.0017710984,0.00013797531,0.0000012239425],"about_ca_topic_score_codex":0.004947551,"about_ca_topic_score_gemma":0.0072868317,"teacher_disagreement_score":0.004947551,"about_ca_system_score_codex":0.00054400606,"about_ca_system_score_gemma":0.00091350335,"threshold_uncertainty_score":0.009837508},"labels":[],"label_agreement":null},{"id":"W2153245512","doi":"10.1109/icci.1993.315357","title":"Pattern matching for case analysis: a computational definition of closeness","year":2002,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Closeness; Computer science; Semantic similarity; Sentence; Natural language processing; Context (archaeology); Similarity (geometry); Matching (statistics); Metric (unit); Artificial intelligence; Pattern matching; Semantics (computer science); Meaning (existential); Information retrieval; Mathematics; Programming language","score_opus":0.033902599765022697,"score_gpt":0.2845699175675961,"score_spread":0.2506673178025734,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2153245512","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0036496457,0.0004347847,0.9907148,0.0005970644,0.00004513965,0.00014464096,0.00015277058,0.00017710749,0.0040840623],"genre_scores_gemma":[0.08092772,0.00048777906,0.91563284,0.00023007239,0.00015758011,0.0006641059,0.00051813305,0.00010867808,0.0012730036],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9836007,0.0071796924,0.0016940666,0.0032452503,0.0038300254,0.00045028023],"domain_scores_gemma":[0.9819225,0.0122099165,0.0014053971,0.0029882777,0.000967919,0.0005060119],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008222776,0.0012289054,0.0018122224,0.0126871485,0.002886346,0.0091879275,0.005127924,0.003610175,0.008769039],"category_scores_gemma":[0.041213784,0.0011714804,0.0042980104,0.011335821,0.01392315,0.02145683,0.01060386,0.004330143,0.001600133],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000479974,0.000045740162,0.0012912931,0.00024305824,0.0001038733,0.00018723888,0.0007927933,0.011228797,0.0006793633,0.91662735,0.0015169021,0.06723544],"study_design_scores_gemma":[0.000014543082,0.000027562362,0.00026062256,0.00007752521,0.000030626263,0.00027616604,0.00021862434,0.043555263,0.00070362224,0.94585174,0.008948945,0.00003476588],"about_ca_topic_score_codex":0.0021623836,"about_ca_topic_score_gemma":0.0013759828,"teacher_disagreement_score":0.0126871485,"about_ca_system_score_codex":0.0027800894,"about_ca_system_score_gemma":0.0018438224,"threshold_uncertainty_score":0.043486714},"labels":[],"label_agreement":null},{"id":"W2153305265","doi":"10.1504/ijbpim.2013.059136","title":"RSenter: terms mining tool from unstructured data sources","year":2013,"lang":"en","type":"article","venue":"International Journal of Business Process Integration and Management","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Saskatchewan","funders":"","keywords":"Unstructured data; Computer science; Data mining; Data science; Big data","score_opus":0.01636727443417805,"score_gpt":0.2825990706188246,"score_spread":0.26623179618464654,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2153305265","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011437733,0.0007648942,0.6765014,0.0010245778,0.00019674237,0.0011563676,0.03498397,0.26555517,0.008379056],"genre_scores_gemma":[0.0504744,0.00069296284,0.88596284,0.0003367141,0.000083537176,0.00080255413,0.050021917,0.006164562,0.005460603],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99798167,0.00026571122,0.00038132476,0.00035098207,0.00090569595,0.000114643706],"domain_scores_gemma":[0.9951054,0.002627861,0.0004962251,0.0009034819,0.0007093932,0.00015762722],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025050568,0.0017034619,0.0011882104,0.008899423,0.00085954653,0.0031023102,0.002645427,0.001130912,0.013071782],"category_scores_gemma":[0.015533517,0.0007768121,0.002346125,0.005238474,0.00069760013,0.0061226794,0.004406016,0.0015564351,0.009420947],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00076536974,0.00027849374,0.005800484,0.00362002,0.0003492884,0.0028531032,0.002346525,0.0074582547,0.018754251,0.0535843,0.20394816,0.7002418],"study_design_scores_gemma":[0.0003321541,0.00033148227,0.004389605,0.0009903597,0.00018009527,0.005031793,0.0025770543,0.2733117,0.068138264,0.13836958,0.506022,0.00032595245],"about_ca_topic_score_codex":0.0020085261,"about_ca_topic_score_gemma":0.004323522,"teacher_disagreement_score":0.013071782,"about_ca_system_score_codex":0.0006909009,"about_ca_system_score_gemma":0.0020492724,"threshold_uncertainty_score":0.043729424},"labels":[],"label_agreement":null},{"id":"W2153579005","doi":"10.48550/arxiv.1310.4546","title":"Distributed Representations of Words and Phrases and their Compositionality","year":2013,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":18086,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Principle of compositionality; Softmax function; Word (group theory); Natural language processing; Artificial intelligence; Simple (philosophy); Semantics (computer science); Quality (philosophy); Speedup; Linguistics; Artificial neural network","score_opus":0.03203702315028123,"score_gpt":0.19886743815782462,"score_spread":0.1668304150075434,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2153579005","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02890214,0.00025340356,0.96876645,0.00018489343,0.000026749703,0.000041582833,0.00019955765,0.0005430061,0.0010822376],"genre_scores_gemma":[0.6138421,0.0006620522,0.37718466,0.00020146616,0.00014993497,0.0003088529,0.0014186363,0.00024668605,0.0059855888],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9989674,0.0003990796,0.00005032069,0.00027878428,0.00022696059,0.000077356475],"domain_scores_gemma":[0.9978423,0.0012329429,0.00021926529,0.00037998724,0.00026141084,0.00006410707],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012174537,0.0006661215,0.0007507542,0.0014155667,0.00051519694,0.0011543806,0.0010785644,0.0008912294,0.0032045906],"category_scores_gemma":[0.008046692,0.0005523507,0.00074321614,0.0015305112,0.001039984,0.0039681233,0.0014098099,0.0014013816,0.0011610659],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004450993,0.00020728038,0.004032312,0.0004273972,0.00014200994,0.00039861814,0.00081552326,0.19712146,0.028487558,0.20649232,0.004316067,0.55711436],"study_design_scores_gemma":[0.000021411233,0.000090010646,0.00075462006,0.000021853635,0.000023288998,0.00011643156,0.00007532917,0.82665884,0.0029826527,0.16694333,0.0022927239,0.000019505824],"about_ca_topic_score_codex":0.0017286778,"about_ca_topic_score_gemma":0.0029612193,"teacher_disagreement_score":0.0032045906,"about_ca_system_score_codex":0.0004961253,"about_ca_system_score_gemma":0.00066276814,"threshold_uncertainty_score":0.010720372},"labels":[],"label_agreement":null},{"id":"W2154109570","doi":"10.5539/elt.v4n1p70","title":"The Effect of Collocation on Meaning Representation of Adjectives such as Big and Large in Translation from Two Languages Used in the Article to English Language Texts","year":2011,"lang":"en","type":"article","venue":"English Language Teaching","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Linguistics; Meaning (existential); Collocation (remote sensing); Noun; Psychology; Representation (politics); Computer science; Philosophy","score_opus":0.012281956426886634,"score_gpt":0.30201818717645834,"score_spread":0.2897362307495717,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2154109570","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.92494017,0.0008462271,0.02580185,0.0016797688,0.00059665414,0.00025775508,0.00019402306,0.0008004838,0.044883035],"genre_scores_gemma":[0.98503464,0.00021980987,0.009454172,0.0004710536,0.00006380832,0.00009621537,0.0001768032,0.0006140124,0.003869508],"study_design_codex":"bench_or_experimental","study_design_gemma":"observational","domain_scores_codex":[0.9806786,0.014418711,0.0011211642,0.0017343166,0.0016535698,0.0003936856],"domain_scores_gemma":[0.7192042,0.24635476,0.009493952,0.012283186,0.010676479,0.001987351],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.014573161,0.0010224052,0.000576245,0.0014389614,0.0023881895,0.003934744,0.001011312,0.0014521484,0.014128902],"category_scores_gemma":[0.16081215,0.00085788354,0.00065966934,0.0014039115,0.0041817455,0.0063218004,0.0041868356,0.0031023547,0.0023428036],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.023918992,0.0013413634,0.101721175,0.0040751672,0.00071447855,0.0041563595,0.11236141,0.011079718,0.355565,0.03448833,0.0122148385,0.3383632],"study_design_scores_gemma":[0.0022634796,0.007159022,0.5593465,0.0014862989,0.0031542701,0.009018758,0.06198721,0.089296825,0.13976681,0.07197942,0.053523056,0.0010183726],"about_ca_topic_score_codex":0.0053764633,"about_ca_topic_score_gemma":0.0045192707,"teacher_disagreement_score":0.014573161,"about_ca_system_score_codex":0.001553119,"about_ca_system_score_gemma":0.00093308627,"threshold_uncertainty_score":0.07707113},"labels":[],"label_agreement":null},{"id":"W2154417380","doi":"","title":"Fast Consensus Hypothesis Regeneration for Machine Translation","year":2010,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"NIST; Machine translation; Computer science; Decoding methods; Regeneration (biology); Task (project management); Translation (biology); Artificial intelligence; Natural language processing; Baseline (sea); Algorithm; Engineering","score_opus":0.021684188397252988,"score_gpt":0.2625955019507148,"score_spread":0.2409113135534618,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2154417380","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007312008,0.00041361887,0.98360956,0.0001610708,0.00008980257,0.00010306919,0.00020687423,0.006831243,0.0012727194],"genre_scores_gemma":[0.20274456,0.0002881542,0.7886535,0.00026361176,0.00014318987,0.00039997682,0.0021863366,0.0010584447,0.004262255],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99653447,0.0013106988,0.00020505671,0.0007252669,0.0009986811,0.00022575435],"domain_scores_gemma":[0.990355,0.0055586677,0.00047840836,0.0015874762,0.0018313294,0.00018909374],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035187772,0.0014590264,0.0014560104,0.0024700207,0.0018152118,0.0013401015,0.002574076,0.0020170365,0.009867731],"category_scores_gemma":[0.0104666315,0.00079547556,0.0012635943,0.0021058735,0.0012105331,0.0032243242,0.0026613101,0.00236465,0.0072844764],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00078052987,0.00019383582,0.00090636435,0.00049150817,0.00015987971,0.00052323344,0.00036638664,0.054457016,0.03847216,0.015431444,0.016536776,0.8716809],"study_design_scores_gemma":[0.00015679096,0.00029875134,0.00054583524,0.00004281218,0.00008165764,0.00056828046,0.00020302976,0.84293956,0.074612826,0.0670288,0.0134204915,0.0001011916],"about_ca_topic_score_codex":0.0021400067,"about_ca_topic_score_gemma":0.0027563756,"teacher_disagreement_score":0.009867731,"about_ca_system_score_codex":0.0007987958,"about_ca_system_score_gemma":0.001982415,"threshold_uncertainty_score":0.03301084},"labels":[],"label_agreement":null},{"id":"W2154507487","doi":"10.63317/44fpr4ygd2xg","title":"New Tools for Web-Scale N-grams","year":2010,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":74,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Ambiguity; n-gram; Set (abstract data type); Scale (ratio); Natural language processing; Annotation; World Wide Web; Information retrieval; Web resource; Artificial intelligence; Resource (disambiguation); Language model","score_opus":0.013651884990419671,"score_gpt":0.27778554876215683,"score_spread":0.26413366377173714,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2154507487","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0018349688,0.00036030737,0.92542464,0.00036867618,0.00017791998,0.00022265592,0.0030644855,0.064900056,0.0036462913],"genre_scores_gemma":[0.013933009,0.0002810374,0.97093666,0.0001989576,0.00013844343,0.0004279797,0.005287899,0.005787515,0.0030084692],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99582857,0.0008465679,0.0008202886,0.00062140997,0.001756989,0.0001262206],"domain_scores_gemma":[0.9856619,0.007401186,0.0013381535,0.0030776986,0.002153077,0.00036784995],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035090698,0.0020521132,0.0016990328,0.007681786,0.001776598,0.0045513473,0.0024656022,0.0015834868,0.019200282],"category_scores_gemma":[0.03037625,0.0015367959,0.0016009948,0.008296991,0.0011576506,0.011913175,0.0045687524,0.0034696057,0.015656086],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00051180547,0.00023513756,0.0025833123,0.0010019136,0.00016727137,0.00067597406,0.0012509046,0.0060049947,0.017836988,0.0918187,0.08125754,0.7966556],"study_design_scores_gemma":[0.00017021694,0.000091339636,0.0019652836,0.00045439953,0.00008681643,0.00140476,0.00053097896,0.21684885,0.02869097,0.30793825,0.44151154,0.00030661732],"about_ca_topic_score_codex":0.0015732852,"about_ca_topic_score_gemma":0.003762955,"teacher_disagreement_score":0.019200282,"about_ca_system_score_codex":0.0010058824,"about_ca_system_score_gemma":0.0016033264,"threshold_uncertainty_score":0.064231396},"labels":[],"label_agreement":null},{"id":"W2154558620","doi":"","title":"Revisiting Context-based Projection Methods for Term-Translation Spotting in Comparable Corpora","year":2010,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":85,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Context (archaeology); Spotting; Term (time); Projection (relational algebra); Translation (biology); Task (project management); Natural language processing; Artificial intelligence; Keyword spotting; Domain (mathematical analysis); Information retrieval; Machine learning; Algorithm; Mathematics; Engineering","score_opus":0.045445463587981474,"score_gpt":0.38610266193254295,"score_spread":0.3406571983445615,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2154558620","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.036578387,0.0026806172,0.95279443,0.00035063128,0.00019061964,0.00026906302,0.00041585395,0.004670077,0.0020503083],"genre_scores_gemma":[0.18732262,0.001065635,0.8061688,0.0002223501,0.0004539019,0.00043487235,0.0018127218,0.0008009434,0.0017182011],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9930702,0.003528226,0.00057579513,0.0013377831,0.0012525853,0.00023541467],"domain_scores_gemma":[0.9825022,0.011704919,0.0006357133,0.0023746456,0.0024728416,0.00030957],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009505608,0.0018065966,0.0019299496,0.0047954014,0.0013731387,0.0030634531,0.0021183295,0.0021321431,0.0039236546],"category_scores_gemma":[0.028424613,0.00081134506,0.0011978112,0.006375071,0.001380494,0.005539589,0.0031038225,0.0029145651,0.003829335],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004686346,0.00032522416,0.0031585465,0.00044679976,0.00031581652,0.00022900669,0.00058269285,0.02599401,0.019688413,0.00699973,0.005616966,0.93617415],"study_design_scores_gemma":[0.0001655071,0.0004927351,0.004305751,0.00015144078,0.00024672813,0.0007662976,0.0006890292,0.9155457,0.021277422,0.04413774,0.012091617,0.00012999463],"about_ca_topic_score_codex":0.0027388248,"about_ca_topic_score_gemma":0.005112337,"teacher_disagreement_score":0.009505608,"about_ca_system_score_codex":0.0004325684,"about_ca_system_score_gemma":0.0022123083,"threshold_uncertainty_score":0.050271094},"labels":[],"label_agreement":null},{"id":"W2154699777","doi":"10.1186/1472-6963-14-s2-p130","title":"Identifying emerging priorities in Knowledge Translation from the perspective of trainees","year":2014,"lang":"en","type":"article","venue":"BMC Health Services Research","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; Vancouver Coastal Health; Toronto Metropolitan University; Dalhousie University; Institute for Work & Health; University of British Columbia","funders":"","keywords":"Health informatics; Nursing research; Health administration; Medicine; Knowledge translation; Perspective (graphical); Public health; Translation (biology); Health services research; Quality of Life Research; Medical education; Nursing; Knowledge management; Artificial intelligence; Computer science","score_opus":0.10355673342003928,"score_gpt":0.4588506081069041,"score_spread":0.3552938746868648,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2154699777","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4639126,0.015973981,0.087773725,0.34638208,0.0018292125,0.00057561585,0.00029818687,0.00014411134,0.083110414],"genre_scores_gemma":[0.94531614,0.005350306,0.03407727,0.009540586,0.00037059427,0.00035399466,0.00017462813,0.0000875874,0.004728884],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.87175184,0.09997126,0.0051813843,0.0037776143,0.009834532,0.009483319],"domain_scores_gemma":[0.84340185,0.1078319,0.0106812995,0.0036512627,0.019228768,0.015204993],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.109435484,0.00077033945,0.000984232,0.005877802,0.012491075,0.025529161,0.002386778,0.006771521,0.0054559903],"category_scores_gemma":[0.11473073,0.0011693019,0.0007996675,0.004867819,0.015941175,0.025919672,0.020935746,0.011948391,0.0014259279],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029662528,0.0001887568,0.017909937,0.0019760956,0.00009962259,0.0023747964,0.7461297,0.0005230138,0.0031357869,0.12596682,0.008436706,0.092962064],"study_design_scores_gemma":[0.000036817437,0.00012673176,0.005198978,0.0019088195,0.000049180606,0.0011539079,0.7827833,0.0018858706,0.0015683488,0.14733827,0.05784857,0.00010120476],"about_ca_topic_score_codex":0.0046487004,"about_ca_topic_score_gemma":0.0065413564,"teacher_disagreement_score":0.8905645,"about_ca_system_score_codex":0.013513962,"about_ca_system_score_gemma":0.02944271,"threshold_uncertainty_score":0.57875705},"labels":[],"label_agreement":null},{"id":"W2154726829","doi":"10.15398/jlm.v2i1.78","title":"Evaluation of automatic updates of Roget’s Thesaurus","year":2014,"lang":"en","type":"article","venue":"Journal of Language Modelling","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Thesaurus; Computer science; Information retrieval; Natural language processing; WordNet; Vocabulary; Artificial intelligence; Word (group theory); Linguistics","score_opus":0.029702668167642263,"score_gpt":0.3008975775526793,"score_spread":0.27119490938503704,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2154726829","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9181035,0.0029179726,0.05139967,0.0004191025,0.0005711035,0.0014560235,0.0024283926,0.014192449,0.008511791],"genre_scores_gemma":[0.81279486,0.0006760663,0.16212167,0.00028936,0.00014387324,0.00084903836,0.017255154,0.0011613688,0.0047085406],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9811257,0.00801796,0.002326293,0.0035924085,0.004434487,0.000503111],"domain_scores_gemma":[0.94427675,0.035676677,0.0024958993,0.0065814657,0.009808834,0.0011603755],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.016960854,0.0019211668,0.0020317992,0.0062639993,0.0018303267,0.002919605,0.0047781556,0.0030025297,0.0028031787],"category_scores_gemma":[0.07927291,0.0012107245,0.0013063313,0.0035034418,0.0015674984,0.006720194,0.004292651,0.0021284441,0.0017145445],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004110975,0.0028964956,0.031379476,0.0030821317,0.0014641884,0.00091920636,0.0050165844,0.055096872,0.041309565,0.0024837682,0.015895633,0.83634514],"study_design_scores_gemma":[0.001545121,0.006285837,0.11650168,0.0005635111,0.0011501209,0.0027397028,0.0032949895,0.735555,0.08297993,0.0026805375,0.046202723,0.0005008871],"about_ca_topic_score_codex":0.0130294645,"about_ca_topic_score_gemma":0.015795507,"teacher_disagreement_score":0.016960854,"about_ca_system_score_codex":0.0022233275,"about_ca_system_score_gemma":0.0021027715,"threshold_uncertainty_score":0.08969867},"labels":[],"label_agreement":null},{"id":"W2155846544","doi":"","title":"Leveraging Transliterations from Multiple Languages","year":2011,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Transliteration; Computer science; Natural language processing; Artificial intelligence; Ranking (information retrieval); Task (project management); Hebrew; Hindi; Linguistics; Engineering","score_opus":0.03175716691687523,"score_gpt":0.26008777188605625,"score_spread":0.228330604969181,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2155846544","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12653017,0.0013127909,0.8324145,0.0009181947,0.00040594838,0.00029909727,0.0010850877,0.021218136,0.015816065],"genre_scores_gemma":[0.521413,0.0005476117,0.4625206,0.00030668848,0.00019964844,0.00015613883,0.0032250967,0.0017900545,0.009841141],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9974515,0.0008909674,0.0002046083,0.00072793855,0.00056276534,0.00016214089],"domain_scores_gemma":[0.9944219,0.0021953427,0.00047109602,0.0013147433,0.001433341,0.0001635612],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019289762,0.0015452418,0.0012738378,0.0014512077,0.00081306515,0.0015232125,0.0009702186,0.00087446056,0.004938864],"category_scores_gemma":[0.010038466,0.00055188837,0.00084017013,0.001261979,0.00042393265,0.0030216977,0.0023253993,0.0014990681,0.0067613875],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00047853554,0.00024137044,0.005021818,0.00081060355,0.00021980675,0.0009806148,0.00096714287,0.01772635,0.14604193,0.002710732,0.00994523,0.8148558],"study_design_scores_gemma":[0.000108981345,0.0010593174,0.008711045,0.00029292013,0.00054670614,0.0031569747,0.0010492828,0.52823424,0.34615588,0.018273529,0.09213016,0.00028090796],"about_ca_topic_score_codex":0.0019033599,"about_ca_topic_score_gemma":0.0050515938,"teacher_disagreement_score":0.004938864,"about_ca_system_score_codex":0.00043461644,"about_ca_system_score_gemma":0.0011425578,"threshold_uncertainty_score":0.01652217},"labels":[],"label_agreement":null},{"id":"W2156809035","doi":"10.5539/ells.v1n1p50","title":"Can the Essential Lexicon of Geology be Appropriately Represented in an Intuitively Written EAP Module?","year":2011,"lang":"en","type":"article","venue":"English Language and Literature Studies","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Lexicon; Consistency (knowledge bases); Vocabulary; Computer science; Word (group theory); Section (typography); Key (lock); Software; Mathematics education; Linguistics; Natural language processing; Artificial intelligence; Programming language; Mathematics; Philosophy","score_opus":0.01936681065497135,"score_gpt":0.28082781899841447,"score_spread":0.2614610083434431,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2156809035","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8791858,0.0006829089,0.04491503,0.0012332271,0.00012009149,0.00015630308,0.00057013653,0.0005250496,0.07261141],"genre_scores_gemma":[0.978564,0.00037269422,0.015793249,0.0001637052,0.000021349178,0.000066117405,0.0006529276,0.00011637679,0.004249494],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9988167,0.0006008396,0.00011908719,0.00017533905,0.00021700807,0.0000710257],"domain_scores_gemma":[0.9958046,0.0023730218,0.0005806955,0.0003206107,0.00078518514,0.00013596776],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009915356,0.00022707108,0.00023782453,0.0011774511,0.0002949827,0.0027890124,0.00041233975,0.00035078306,0.004628378],"category_scores_gemma":[0.014306779,0.00017568693,0.00016200086,0.0011420747,0.0006656018,0.0031588057,0.00089768483,0.00044818188,0.0014777953],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00047130752,0.00048957404,0.119263746,0.002149283,0.000059837,0.0024366276,0.05340721,0.0021056368,0.08024082,0.038106155,0.01065903,0.6906107],"study_design_scores_gemma":[0.0001073535,0.0014419621,0.4221767,0.0019960925,0.00028166734,0.008118726,0.12882784,0.028326578,0.05075482,0.045695927,0.3120875,0.00018483368],"about_ca_topic_score_codex":0.0011112763,"about_ca_topic_score_gemma":0.0012720446,"teacher_disagreement_score":0.004628378,"about_ca_system_score_codex":0.00059600605,"about_ca_system_score_gemma":0.0010681729,"threshold_uncertainty_score":0.015483499},"labels":[],"label_agreement":null},{"id":"W2157014857","doi":"10.1111/j.1467-8535.2006.00614.x","title":"e‐Learning for depth in the Semantic Web","year":2006,"lang":"en","type":"article","venue":"British Journal of Educational Technology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Formative assessment; Semantic Web; Implementation; World Wide Web; Educational technology; Semantics (computer science); Deep learning; Comprehension; Meaningful learning; Mathematics education; Artificial intelligence; Software engineering; Psychology; Programming language","score_opus":0.00764132777862409,"score_gpt":0.27798339436845937,"score_spread":0.2703420665898353,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2157014857","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.21469852,0.0006897198,0.7186423,0.002948469,0.000096914744,0.0004890305,0.00031707855,0.0037010028,0.058416974],"genre_scores_gemma":[0.67329353,0.00048193036,0.31798723,0.0003283605,0.000030402198,0.00026661807,0.00031407678,0.0001440693,0.007153753],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99853873,0.00081614737,0.000063518026,0.00009826545,0.0004029425,0.00008034386],"domain_scores_gemma":[0.9949012,0.0039436575,0.00015352062,0.000576711,0.00029720107,0.00012777779],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028994142,0.00038880392,0.000264926,0.0008028839,0.00046609974,0.002301854,0.0006805274,0.001026566,0.0061964537],"category_scores_gemma":[0.0067295777,0.0001555902,0.00037348806,0.00081144174,0.0009041874,0.0077662375,0.0020540534,0.0012519235,0.0010699084],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00034484646,0.0011675716,0.003076432,0.00054972933,0.00003983263,0.000260476,0.004381519,0.010696435,0.021543592,0.17718099,0.008404444,0.77235407],"study_design_scores_gemma":[0.00026231052,0.0005720691,0.0051953485,0.00054464524,0.00009442899,0.0005942301,0.0035322877,0.14136726,0.06962583,0.55663675,0.2214859,0.000088945024],"about_ca_topic_score_codex":0.00037794153,"about_ca_topic_score_gemma":0.0005392113,"teacher_disagreement_score":0.0061964537,"about_ca_system_score_codex":0.0005005899,"about_ca_system_score_gemma":0.0008826539,"threshold_uncertainty_score":0.020729184},"labels":[],"label_agreement":null},{"id":"W2157331557","doi":"10.3115/v1/d14-1179","title":"Learning Phrase Representations using RNN Encoder–Decoder for Statistical Machine Translation","year":2014,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":24447,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Machine translation; Computer science; Phrase; Natural language processing; Artificial intelligence; Encoder; Translation (biology); Statistical learning; Speech recognition","score_opus":0.04515960229356928,"score_gpt":0.3691423632671048,"score_spread":0.3239827609735355,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2157331557","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016020175,0.0019086124,0.9611914,0.0005021311,0.0005310296,0.00015884495,0.0010758277,0.013409671,0.0052022967],"genre_scores_gemma":[0.2320935,0.0011946189,0.7424739,0.00056916557,0.00060275453,0.00048489484,0.008184108,0.0016821966,0.012714815],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987036,0.00051743345,0.00009456237,0.00035973472,0.00020507156,0.00011964159],"domain_scores_gemma":[0.99741334,0.0014045533,0.00011030169,0.00048690182,0.0004990729,0.000085825006],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001586163,0.0019182193,0.0013714717,0.0012492524,0.000685956,0.0013847742,0.0016065417,0.0018283007,0.008033043],"category_scores_gemma":[0.0060941554,0.00080817105,0.0011170794,0.0015668329,0.0005848185,0.002990984,0.0016559999,0.0025064822,0.011613985],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004962503,0.00027627483,0.00072502805,0.00033942904,0.00019012862,0.0004258547,0.00016249245,0.06922774,0.016654318,0.015491743,0.029419668,0.86659104],"study_design_scores_gemma":[0.0000915686,0.00012442183,0.00022179868,0.000036960668,0.00007210032,0.00016042966,0.00006479194,0.95072573,0.010679371,0.0314944,0.006300459,0.000027980215],"about_ca_topic_score_codex":0.0039490233,"about_ca_topic_score_gemma":0.008685879,"teacher_disagreement_score":0.008033043,"about_ca_system_score_codex":0.0007224108,"about_ca_system_score_gemma":0.0015877215,"threshold_uncertainty_score":0.026873171},"labels":[],"label_agreement":null},{"id":"W2157455456","doi":"10.7202/008019ar","title":"Utilisation des poids empiriques dans l’analyse syntaxique : une application en Traduction Automatique","year":2004,"lang":"fr","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Philosophy","score_opus":0.0330532106259275,"score_gpt":0.30313302000971126,"score_spread":0.2700798093837838,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2157455456","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010436193,0.00035947104,0.98612195,0.00032884273,0.000033730055,0.00007640315,0.0000677453,0.0006348379,0.0019408013],"genre_scores_gemma":[0.09245136,0.00035282623,0.9042769,0.00009898743,0.000024527153,0.00019331291,0.0001596447,0.00046820226,0.0019742276],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9866414,0.0071920697,0.0010377287,0.0018908081,0.0029976147,0.00024039752],"domain_scores_gemma":[0.9453532,0.04167932,0.0017514462,0.0050742836,0.0058330894,0.00030871553],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.016019644,0.0016273329,0.0020012045,0.0044740853,0.0024413734,0.006916227,0.0014675971,0.0020365869,0.004841741],"category_scores_gemma":[0.046705037,0.0018056735,0.0018062983,0.0032504683,0.0050752554,0.009899606,0.004308553,0.003670702,0.001427939],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007405248,0.00023137055,0.014711949,0.0015448102,0.000381926,0.00073746825,0.009072186,0.050534323,0.031556804,0.26967502,0.002523355,0.61829025],"study_design_scores_gemma":[0.00017780351,0.00044713978,0.006159487,0.0008374721,0.00032381123,0.0010722536,0.0029424909,0.5663325,0.06455292,0.2832693,0.07356627,0.00031853787],"about_ca_topic_score_codex":0.0053103054,"about_ca_topic_score_gemma":0.0067851003,"teacher_disagreement_score":0.016019644,"about_ca_system_score_codex":0.002642266,"about_ca_system_score_gemma":0.0036602656,"threshold_uncertainty_score":0.08472097},"labels":[],"label_agreement":null},{"id":"W2157616430","doi":"10.1017/s1351324908004737","title":"Multilingual pronunciation by analogy","year":2008,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Institute for Biodiagnostics","funders":"","keywords":"Computer science; Pronunciation; Transcription (linguistics); Natural language processing; Orthography; Spelling; Analogy; Artificial intelligence; Variation (astronomy); Phonetic transcription; Linguistics; Speech recognition","score_opus":0.004698680563171272,"score_gpt":0.2267604826131823,"score_spread":0.22206180205001103,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2157616430","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6555782,0.0014106855,0.3097917,0.00025867068,0.00022130254,0.00019827428,0.0019414026,0.005632005,0.024967747],"genre_scores_gemma":[0.93966454,0.00016413188,0.055126455,0.00004866818,0.000029924548,0.000050879684,0.0018230415,0.00024709635,0.0028451795],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99728644,0.0011520714,0.00022424416,0.0007270664,0.00046740417,0.00014272447],"domain_scores_gemma":[0.9961959,0.0019059984,0.00019533848,0.00076665846,0.0008723659,0.000063718224],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011998158,0.0006079079,0.0006839917,0.001232567,0.0004237048,0.0010534027,0.00050859747,0.00029589448,0.0069946847],"category_scores_gemma":[0.0062983003,0.0001882266,0.0003667937,0.0012029891,0.00037004918,0.0010190159,0.0017906637,0.00051546854,0.0031241502],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00090464606,0.00016098455,0.024277382,0.00065297866,0.00015599036,0.0008648525,0.0016658352,0.030837726,0.10652626,0.0039734077,0.0034569332,0.82652295],"study_design_scores_gemma":[0.00034077402,0.00244945,0.11389663,0.00027897907,0.00033396145,0.007825371,0.004455219,0.42349657,0.31346577,0.024346707,0.10881224,0.00029844747],"about_ca_topic_score_codex":0.0024631429,"about_ca_topic_score_gemma":0.0027775576,"teacher_disagreement_score":0.0069946847,"about_ca_system_score_codex":0.00034269423,"about_ca_system_score_gemma":0.00043474344,"threshold_uncertainty_score":0.023399591},"labels":[],"label_agreement":null},{"id":"W2157951638","doi":"10.1007/s10278-007-9034-7","title":"Improving the Utility of Speech Recognition Through Error Detection","year":2007,"lang":"en","type":"article","venue":"Journal of Digital Imaging","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":38,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia; Simon Fraser University","funders":"Centers for Disease Control and Prevention; Simon Fraser University","keywords":"Proofreading; Word error rate; Computer science; Speech recognition; Transcription (linguistics); Error detection and correction; Vendor; Artificial intelligence; Algorithm","score_opus":0.023348233773392832,"score_gpt":0.2892260673313842,"score_spread":0.2658778335579913,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2157951638","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17823946,0.0015561653,0.8012883,0.0007099043,0.00029131625,0.00008503759,0.00044334846,0.013447242,0.0039392877],"genre_scores_gemma":[0.6556811,0.00075481983,0.33522645,0.0003410754,0.00024427936,0.00005999471,0.00070491544,0.00077704625,0.0062103732],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9973719,0.0008135868,0.00014974749,0.00043874167,0.0010157215,0.00021031962],"domain_scores_gemma":[0.97902703,0.014469403,0.0005800086,0.001931974,0.0037680452,0.00022345803],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021878278,0.0015233154,0.0012017173,0.0016815868,0.00039065097,0.0015942658,0.0013919364,0.0018633921,0.005437256],"category_scores_gemma":[0.01893632,0.0005146995,0.00042834095,0.0008370776,0.00047320002,0.0027651016,0.0011987283,0.0014430336,0.004590561],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021406303,0.00038611816,0.005434926,0.00019271592,0.00011057048,0.0002897462,0.0001430273,0.018757127,0.11013659,0.0019973489,0.0037226847,0.8566885],"study_design_scores_gemma":[0.000077612814,0.0004321309,0.0035055776,0.00003748621,0.00015023898,0.0006425856,0.00008416519,0.74798715,0.24126819,0.0026114073,0.0031387971,0.000064649714],"about_ca_topic_score_codex":0.0026926973,"about_ca_topic_score_gemma":0.0026522332,"teacher_disagreement_score":0.005437256,"about_ca_system_score_codex":0.00031572784,"about_ca_system_score_gemma":0.0008314042,"threshold_uncertainty_score":0.01818943},"labels":[],"label_agreement":null},{"id":"W2158327947","doi":"","title":"MISTRAL: A Lattice Translation System for IWSLT 2007","year":2007,"lang":"en","type":"article","venue":"IWSLT","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Sentence; Phrase; Computer science; Natural language processing; Discriminative model; Translation (biology); Lattice (music); Artificial intelligence; Speech recognition; Machine translation; Linguistics; Physics; Philosophy","score_opus":0.023399304268822013,"score_gpt":0.3037183061385247,"score_spread":0.28031900186970266,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2158327947","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01491105,0.0012957248,0.6039855,0.0010457624,0.0012590856,0.0006602052,0.013337071,0.3384153,0.025090363],"genre_scores_gemma":[0.09942421,0.0007276986,0.77804404,0.0007961541,0.0003755857,0.00077671616,0.07761036,0.018459978,0.023785256],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99835384,0.00035136988,0.00020506093,0.00036480385,0.0006135426,0.00011138842],"domain_scores_gemma":[0.9986467,0.00025816503,0.00011556132,0.00034058842,0.0005221537,0.0001168026],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016308291,0.0012899062,0.00096639467,0.0017463603,0.0010774318,0.0017054895,0.0018385672,0.0010939565,0.024607869],"category_scores_gemma":[0.004435765,0.00080140895,0.00092573854,0.0011173099,0.0005953503,0.0025692154,0.0018171386,0.0021170327,0.023583103],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006894336,0.00028968675,0.001056088,0.00080799166,0.000117852614,0.0006567356,0.0006578794,0.006531664,0.075582884,0.012462505,0.4045207,0.49662665],"study_design_scores_gemma":[0.00067003514,0.00053872756,0.0019598147,0.0002020313,0.00014858432,0.0022317776,0.00050081307,0.17603624,0.11688618,0.01751304,0.68295926,0.000353564],"about_ca_topic_score_codex":0.0047071683,"about_ca_topic_score_gemma":0.008040821,"teacher_disagreement_score":0.024607869,"about_ca_system_score_codex":0.0011952878,"about_ca_system_score_gemma":0.002024022,"threshold_uncertainty_score":0.082321525},"labels":[],"label_agreement":null},{"id":"W2158401175","doi":"10.5539/mas.v3n1p125","title":"Natural Language Processing for Foreign Languages Learning as Computer-based Learning Tools","year":2008,"lang":"en","type":"article","venue":"Modern Applied Science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Foreign language; Natural language processing; Natural language; Language acquisition; Artificial intelligence; Mathematics education; Psychology","score_opus":0.017485713761843377,"score_gpt":0.28216843183810675,"score_spread":0.26468271807626337,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2158401175","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00524807,0.0038846384,0.97031,0.0021294518,0.0001701609,0.00023244621,0.00007177674,0.0010065718,0.016946899],"genre_scores_gemma":[0.06108502,0.00368991,0.92644906,0.0003653389,0.00031398953,0.00053533673,0.00025029675,0.00019374963,0.0071174144],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9989268,0.000565494,0.0000707954,0.0001609403,0.00023569881,0.000040207073],"domain_scores_gemma":[0.99871755,0.00089789723,0.00006022298,0.00019422427,0.00009654713,0.000033511464],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016970952,0.000546502,0.00047399863,0.0010776782,0.00087569177,0.0026602952,0.0011733789,0.0010277667,0.0076595726],"category_scores_gemma":[0.0032980635,0.00024452546,0.00065884343,0.0010307713,0.002530229,0.0053234138,0.0014873807,0.0015440601,0.0018420246],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000044430268,0.000091566326,0.00030109292,0.0006181656,0.000026063693,0.0002895857,0.0011438213,0.0035226347,0.008784447,0.64873564,0.007304709,0.32913786],"study_design_scores_gemma":[0.000032043623,0.00008732934,0.000547757,0.0003004123,0.000025972518,0.0005987819,0.00046448872,0.029455887,0.0077402648,0.7812074,0.17948471,0.000054904336],"about_ca_topic_score_codex":0.0013226698,"about_ca_topic_score_gemma":0.0013255507,"teacher_disagreement_score":0.0076595726,"about_ca_system_score_codex":0.0010130642,"about_ca_system_score_gemma":0.0011327448,"threshold_uncertainty_score":0.025623858},"labels":[],"label_agreement":null},{"id":"W2158403176","doi":"10.1109/nlpke.2009.5313839","title":"Managing the Google Web 1T 5-gram data set","year":2009,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; n-gram; Set (abstract data type); Task (project management); Word (group theory); Measure (data warehouse); Information retrieval; World Wide Web; Database; Artificial intelligence; Programming language; Language model; Engineering","score_opus":0.033150368826111554,"score_gpt":0.3142550993607125,"score_spread":0.28110473053460094,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2158403176","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.45135206,0.0026932547,0.10750637,0.0036747013,0.0012734707,0.0022432364,0.21267144,0.1939544,0.024631135],"genre_scores_gemma":[0.4597346,0.0010302981,0.20393896,0.00054550765,0.00034299574,0.001363656,0.32011905,0.007212763,0.0057121166],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9955146,0.000529173,0.000562282,0.0006962226,0.0022757573,0.00042200214],"domain_scores_gemma":[0.9903527,0.002458538,0.00055089884,0.0035044155,0.0027486598,0.00038486437],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020774335,0.0020950623,0.0020007675,0.0057729236,0.001854306,0.0033029797,0.0023927074,0.0013761446,0.0032841535],"category_scores_gemma":[0.014627418,0.00084049685,0.0012671992,0.007039246,0.0008639737,0.007941964,0.0037851555,0.002187526,0.0048609986],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0048183133,0.0009825937,0.028425777,0.0022831364,0.00047158514,0.003590857,0.0022871115,0.025280586,0.09296758,0.011203496,0.2529789,0.5747101],"study_design_scores_gemma":[0.00045940554,0.0007668393,0.04328334,0.00038663953,0.00034824482,0.0024997238,0.0033214723,0.30047274,0.24402206,0.041537385,0.3618609,0.001041274],"about_ca_topic_score_codex":0.011974506,"about_ca_topic_score_gemma":0.0112768775,"teacher_disagreement_score":0.011974506,"about_ca_system_score_codex":0.00146342,"about_ca_system_score_gemma":0.002540374,"threshold_uncertainty_score":0.023809612},"labels":[],"label_agreement":null},{"id":"W2158468759","doi":"","title":"Reversing Morphological Tokenization in English-to-Arabic SMT","year":2013,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Lexical analysis; Natural language processing; Artificial intelligence; Word (group theory); Task (project management); Machine translation; Speech recognition; Discriminative model; String (physics); Linguistics; Mathematics","score_opus":0.010425758659469524,"score_gpt":0.24236691473173086,"score_spread":0.23194115607226135,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2158468759","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11563169,0.0005596732,0.8690158,0.00029418027,0.00024226501,0.0001883822,0.00043277093,0.009359354,0.00427574],"genre_scores_gemma":[0.51891434,0.00034382092,0.4754488,0.00013807497,0.000074150914,0.000114975395,0.00097433385,0.00069400563,0.0032975234],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99887246,0.00042946602,0.00015716534,0.0002568676,0.00019777312,0.00008617734],"domain_scores_gemma":[0.9979424,0.0009421419,0.00019552778,0.0004979609,0.0003628735,0.000059104357],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011803624,0.0007075739,0.0006109754,0.00057058595,0.00058197795,0.00096371357,0.0010104877,0.0007379562,0.0037673388],"category_scores_gemma":[0.0040582945,0.00034605098,0.00042079098,0.0008117913,0.0006666327,0.0015464695,0.0010940148,0.0010375956,0.0043969806],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007438892,0.00014585862,0.001999627,0.0004014298,0.000040193714,0.00076072215,0.00050583767,0.032433808,0.14825253,0.009126607,0.0045781657,0.80101126],"study_design_scores_gemma":[0.000059081118,0.0004459892,0.0023119163,0.00006734328,0.00007065731,0.001635771,0.0004945002,0.62859875,0.32751697,0.019136334,0.01955507,0.000107593834],"about_ca_topic_score_codex":0.0012734685,"about_ca_topic_score_gemma":0.002362455,"teacher_disagreement_score":0.0037673388,"about_ca_system_score_codex":0.00031413892,"about_ca_system_score_gemma":0.0009283322,"threshold_uncertainty_score":0.012602985},"labels":[],"label_agreement":null},{"id":"W2159249756","doi":"10.1109/icdm.2001.989536","title":"Subject classification in the Oxford English Dictionary","year":2002,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Mitacs","keywords":"Computer science; Subject (documents); Artificial intelligence; Probabilistic logic; Natural language processing; Weighting; Term (time); Naive Bayes classifier; Hierarchy; Word (group theory); Information retrieval; Linguistics; Support vector machine","score_opus":0.02662954429509814,"score_gpt":0.2525118843802993,"score_spread":0.22588234008520117,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2159249756","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0073493617,0.047350746,0.053869832,0.008450951,0.006814397,0.0017959949,0.58456814,0.0071278256,0.2826728],"genre_scores_gemma":[0.03734468,0.043756504,0.12978368,0.002479603,0.0025890039,0.001996274,0.65575784,0.0037498546,0.12254256],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9922516,0.0011409156,0.003807163,0.0007672612,0.001774246,0.0002588039],"domain_scores_gemma":[0.9820809,0.005358666,0.0024855884,0.0024358141,0.0065196147,0.001119518],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0041629043,0.0010220138,0.0018639686,0.033771757,0.0019135864,0.006883719,0.0013644393,0.0012241465,0.08578789],"category_scores_gemma":[0.020042622,0.00066508976,0.00068038347,0.044290174,0.0015196617,0.006891315,0.002742603,0.001754491,0.07326682],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000107911226,0.000036488334,0.0022370848,0.0046610134,0.000033910143,0.00027738616,0.0013355466,0.00037776917,0.0015247347,0.044477783,0.722361,0.22256944],"study_design_scores_gemma":[0.000008248393,0.000010272402,0.0023747934,0.0005753734,0.000007286801,0.0001448788,0.00020768195,0.00015462516,0.00015099562,0.0034111815,0.99293643,0.000018186653],"about_ca_topic_score_codex":0.0092221005,"about_ca_topic_score_gemma":0.0085208975,"teacher_disagreement_score":0.08578789,"about_ca_system_score_codex":0.0023290087,"about_ca_system_score_gemma":0.0072365142,"threshold_uncertainty_score":0.2869891},"labels":[],"label_agreement":null},{"id":"W2159556392","doi":"","title":"Ensemble Triangulation for Statistical Machine Translation","year":2013,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Machine translation; Triangulation; Computer science; Translation (biology); Baseline (sea); Artificial intelligence; BLEU; Decoding methods; Natural language processing; Quality (philosophy); Machine translation system; Machine learning; Algorithm; Mathematics","score_opus":0.020729757615993705,"score_gpt":0.29277545419096007,"score_spread":0.27204569657496636,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2159556392","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0033939097,0.00041090246,0.9932961,0.00007348038,0.000051359697,0.000024506979,0.000090987895,0.0016366423,0.0010222525],"genre_scores_gemma":[0.19427797,0.000996619,0.7964104,0.00014682491,0.00024836854,0.00032631648,0.0016618754,0.0011291526,0.004802506],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99782157,0.0010054254,0.00009890861,0.0004101257,0.0005592855,0.00010460289],"domain_scores_gemma":[0.9965754,0.0015112633,0.00026711865,0.0009676361,0.0005695808,0.00010903294],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00203958,0.0010166202,0.0015262498,0.0021367325,0.0009395347,0.0010269725,0.0014475418,0.0013743705,0.0051740254],"category_scores_gemma":[0.007943667,0.0008913756,0.0014068418,0.0024594914,0.0010010384,0.00210451,0.0029928375,0.0021403928,0.0034300685],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016554388,0.00005330859,0.0011400885,0.00019052347,0.00021203593,0.00014915047,0.00021163208,0.4652461,0.010850582,0.04404871,0.006827253,0.4709051],"study_design_scores_gemma":[0.000008656227,0.000051671566,0.00021890545,0.000015844413,0.00001613071,0.000072377195,0.000021438918,0.95869327,0.003054133,0.032943375,0.004882573,0.000021726788],"about_ca_topic_score_codex":0.0028283577,"about_ca_topic_score_gemma":0.0047659604,"teacher_disagreement_score":0.0051740254,"about_ca_system_score_codex":0.0008161083,"about_ca_system_score_gemma":0.0009599412,"threshold_uncertainty_score":0.017308772},"labels":[],"label_agreement":null},{"id":"W2159755860","doi":"","title":"Batch Tuning Strategies for Statistical Machine Translation","year":2012,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":338,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"Johns Hopkins University","keywords":"Computer science; Machine translation; Machine learning; Feature (linguistics); Artificial intelligence; Variety (cybernetics); Support vector machine; Translation (biology); Simple (philosophy); Sentence; Data mining","score_opus":0.028397383806373983,"score_gpt":0.3112397182538902,"score_spread":0.2828423344475162,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2159755860","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02981577,0.0019731366,0.9547424,0.00028488686,0.00015798939,0.00015705964,0.00024276346,0.008066224,0.004559814],"genre_scores_gemma":[0.42676044,0.0005049597,0.5646689,0.00040400517,0.00022602857,0.0005337071,0.00088823726,0.0015727897,0.004440979],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9961998,0.0020058698,0.00023331514,0.0007710664,0.0006238194,0.00016613472],"domain_scores_gemma":[0.9930962,0.004214146,0.00039538075,0.0015411803,0.00062227104,0.00013078398],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006563127,0.00089938217,0.0010835878,0.0012980606,0.00094217557,0.0014204044,0.0019278713,0.0013054215,0.0052479324],"category_scores_gemma":[0.015978806,0.00065289304,0.0006790788,0.001363982,0.0008336183,0.0027308161,0.001127122,0.0016762775,0.0033467077],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012239695,0.00053200923,0.0025989306,0.00033963815,0.00022970024,0.00013269407,0.00022948543,0.19061415,0.033683516,0.020307796,0.01542012,0.734688],"study_design_scores_gemma":[0.00012074965,0.00022746062,0.0010149417,0.000027485477,0.000039045146,0.00013668001,0.000050288294,0.9533692,0.012522249,0.027276294,0.0051703244,0.000045264016],"about_ca_topic_score_codex":0.0020152205,"about_ca_topic_score_gemma":0.0036050226,"teacher_disagreement_score":0.006563127,"about_ca_system_score_codex":0.0009782065,"about_ca_system_score_gemma":0.0011244618,"threshold_uncertainty_score":0.034709573},"labels":[],"label_agreement":null},{"id":"W2159939228","doi":"10.1007/s10590-011-9089-6","title":"TransSearch: from a bilingual concordancer to a translation finder","year":2010,"lang":"en","type":"article","venue":"Machine Translation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; Université du Québec à Montréal","funders":"","keywords":"Computer science; Machine translation; Natural language processing; Translation (biology); Artificial intelligence; Sentence; Word (group theory); Example-based machine translation; Computer-assisted translation; Bilingual dictionary; Feature (linguistics); Machine translation software usability; Identification (biology); Computational linguistics; Linguistics","score_opus":0.029544977834232312,"score_gpt":0.3163444565220911,"score_spread":0.28679947868785877,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2159939228","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016806966,0.00068799267,0.90005994,0.0007089375,0.0005561421,0.00033456893,0.004713112,0.065902896,0.0102294935],"genre_scores_gemma":[0.10851419,0.00050814805,0.85024697,0.00039785728,0.0002919882,0.00032255868,0.016714374,0.008270195,0.0147337],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99567586,0.0012873436,0.0004209819,0.0013776498,0.0010146647,0.00022346115],"domain_scores_gemma":[0.9932213,0.0024118477,0.0002683086,0.0019420702,0.001880341,0.00027611348],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003481609,0.0018446988,0.00221966,0.005649629,0.0020978574,0.0038494747,0.0031994649,0.00251119,0.030197147],"category_scores_gemma":[0.012482521,0.001479702,0.0017545826,0.005228509,0.0010559959,0.005715107,0.005807175,0.0023237949,0.022917923],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015493142,0.00037740127,0.0015411745,0.0010184915,0.00029357965,0.0016373403,0.0012236873,0.006150317,0.042981178,0.031768363,0.10804512,0.803414],"study_design_scores_gemma":[0.0006118783,0.0006732421,0.0021164753,0.00042555746,0.00060602394,0.0035027044,0.002408043,0.4500469,0.16245578,0.1671071,0.20962538,0.000420866],"about_ca_topic_score_codex":0.0032577221,"about_ca_topic_score_gemma":0.005131732,"teacher_disagreement_score":0.030197147,"about_ca_system_score_codex":0.00091751234,"about_ca_system_score_gemma":0.0029102282,"threshold_uncertainty_score":0.1010195},"labels":[],"label_agreement":null},{"id":"W2160547679","doi":"10.3115/1687878.1687898","title":"Reducing the annotation effort for letter-to-phoneme conversion","year":2009,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Annotation; Classifier (UML); Natural language processing; Artificial intelligence; Cluster analysis; Speech recognition; Training set; Machine learning","score_opus":0.011421269577353923,"score_gpt":0.2732580540577467,"score_spread":0.2618367844803928,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2160547679","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15379336,0.0011590575,0.8013087,0.0011371813,0.0003574265,0.0003501691,0.0014030093,0.0294905,0.011000544],"genre_scores_gemma":[0.31724125,0.00056670187,0.65727824,0.0005111253,0.00016643756,0.0004228178,0.0053468477,0.0025221661,0.015944438],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9971306,0.0010862236,0.00015881016,0.00062516914,0.00082643377,0.00017280834],"domain_scores_gemma":[0.9882604,0.005461008,0.00047972728,0.0030130004,0.00256461,0.00022126976],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017158256,0.0014800421,0.0013694435,0.001410327,0.0013171397,0.001860558,0.0024085958,0.0012639998,0.01004708],"category_scores_gemma":[0.012724529,0.0006203981,0.0007329097,0.0016906898,0.0006796838,0.003083465,0.0023452444,0.0018237403,0.0127257565],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00057401677,0.0003048406,0.0032220988,0.00029473467,0.00004176465,0.00020223465,0.0006950679,0.007120775,0.08115881,0.0018845641,0.011974815,0.8925263],"study_design_scores_gemma":[0.00020718003,0.0006620182,0.017393678,0.00017486318,0.00026960068,0.0017589624,0.002162522,0.41443133,0.42963663,0.018144134,0.114930674,0.00022836014],"about_ca_topic_score_codex":0.0060120756,"about_ca_topic_score_gemma":0.01205959,"teacher_disagreement_score":0.01004708,"about_ca_system_score_codex":0.0006299189,"about_ca_system_score_gemma":0.0018572437,"threshold_uncertainty_score":0.03361082},"labels":[],"label_agreement":null},{"id":"W2160603976","doi":"","title":"Presenting collocates in a dictionary of computing and the Internet according to user needs","year":2011,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"Russian Academy of Sciences; Social Sciences and Humanities Research Council of Canada; Ministerio de Ciencia e Innovación; European Commission","keywords":"The Internet; Computer science; World Wide Web","score_opus":0.0175504411022626,"score_gpt":0.2586285590062862,"score_spread":0.24107811790402356,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2160603976","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0855426,0.00071280106,0.85367244,0.0026612007,0.00048809993,0.0005266738,0.0020122298,0.0038724598,0.05051148],"genre_scores_gemma":[0.29952917,0.00067641994,0.6788671,0.0005326552,0.00010915261,0.00039120603,0.0020006963,0.0016313579,0.016262265],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9984193,0.0006609398,0.00030267244,0.00026740824,0.00027366588,0.00007608854],"domain_scores_gemma":[0.9946378,0.0024244285,0.00040466216,0.0013126393,0.0008361542,0.00038435863],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012440864,0.00052929885,0.0007201456,0.003910302,0.0021761307,0.0038422537,0.0007631537,0.0008844677,0.018472286],"category_scores_gemma":[0.007281074,0.0005035508,0.00040780785,0.005035428,0.00165622,0.009175893,0.0026734434,0.0016515754,0.0037308624],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000503663,0.00022337188,0.0073830183,0.0014823334,0.00004259128,0.00096594665,0.055815782,0.0018060067,0.06290803,0.29613736,0.040205196,0.53252673],"study_design_scores_gemma":[0.000071911956,0.00018776798,0.0059247985,0.0004403321,0.00006550376,0.0029795666,0.01974972,0.021243842,0.014578515,0.08425163,0.8503196,0.00018672584],"about_ca_topic_score_codex":0.0012829229,"about_ca_topic_score_gemma":0.0036268563,"teacher_disagreement_score":0.018472286,"about_ca_system_score_codex":0.0010174297,"about_ca_system_score_gemma":0.0011007792,"threshold_uncertainty_score":0.06179595},"labels":[],"label_agreement":null},{"id":"W2160765088","doi":"10.3115/1073427.1073438","title":"Automatically discovering word senses","year":2003,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Word (group theory); Cluster analysis; Computer science; Artificial intelligence; Natural language processing; Linguistics","score_opus":0.009390139955483549,"score_gpt":0.2565597385996673,"score_spread":0.24716959864418375,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2160765088","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17001855,0.0010193227,0.8058107,0.0009298106,0.0004639013,0.00032573607,0.0036875429,0.0060124123,0.011731968],"genre_scores_gemma":[0.36432898,0.0004548119,0.62134695,0.00023708449,0.0001313308,0.00028774698,0.008378881,0.00080629945,0.0040279194],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99739504,0.00069997396,0.00024464377,0.0008175537,0.0006596808,0.00018308789],"domain_scores_gemma":[0.9954397,0.0019525954,0.00030904595,0.0005393024,0.0015655214,0.00019386467],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001372125,0.0010208548,0.0009928893,0.006228708,0.0023435736,0.0017837121,0.0010880257,0.0011181053,0.0027136998],"category_scores_gemma":[0.009795718,0.00047199635,0.0012652194,0.003615474,0.0008452249,0.002823829,0.001828104,0.0013118802,0.0021238804],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005480081,0.00049086835,0.036995575,0.001219574,0.00038270743,0.0017084025,0.003161009,0.014596827,0.09929799,0.072807156,0.052254435,0.7165375],"study_design_scores_gemma":[0.00009755296,0.00027930914,0.020333074,0.00034653675,0.00038684026,0.004537567,0.004746879,0.532736,0.094072595,0.23241603,0.10974893,0.00029860385],"about_ca_topic_score_codex":0.0026851313,"about_ca_topic_score_gemma":0.007925064,"teacher_disagreement_score":0.006228708,"about_ca_system_score_codex":0.0005714306,"about_ca_system_score_gemma":0.0014919725,"threshold_uncertainty_score":0.009078264},"labels":[],"label_agreement":null},{"id":"W2160988762","doi":"","title":"Joint Processing and Discriminative Training for Letter-to-Phoneme Conversion","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":87,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Discriminative model; Computer science; Pipeline (software); Speech recognition; Task (project management); Joint (building); Artificial intelligence; Key (lock); Segmentation; Training set; Speech processing; Training (meteorology); Sequence labeling; Machine learning; Pattern recognition (psychology); Natural language processing; Engineering","score_opus":0.057056728881810063,"score_gpt":0.2883153528475371,"score_spread":0.23125862396572705,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2160988762","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.044553686,0.00033627838,0.9422024,0.00025057697,0.00009787109,0.00006246968,0.00045758506,0.009672052,0.002367061],"genre_scores_gemma":[0.59925455,0.00022823519,0.38812876,0.00036720582,0.00013378846,0.00023012511,0.0034237714,0.0005211967,0.0077123144],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992594,0.00016701089,0.000029969342,0.00032472383,0.00011875032,0.00010000619],"domain_scores_gemma":[0.9987482,0.000542618,0.000075684446,0.00040142055,0.00016487422,0.000067109955],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008843266,0.0011028266,0.0010327568,0.0005312106,0.0005391226,0.0008476073,0.0018293738,0.0011179509,0.0040759016],"category_scores_gemma":[0.0031683184,0.0004977806,0.0005381556,0.0010001799,0.0005890328,0.001888911,0.0012235583,0.0020240818,0.0030412495],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004748101,0.0005098618,0.002603855,0.00012070459,0.0000824622,0.00017630715,0.00010820303,0.14108953,0.043563783,0.0038589148,0.011015227,0.79639643],"study_design_scores_gemma":[0.000014958141,0.000057189525,0.00076260086,0.0000038203234,0.000016038028,0.00008355107,0.000013396198,0.98165995,0.013315605,0.0027705734,0.0012875622,0.000014787244],"about_ca_topic_score_codex":0.0059381863,"about_ca_topic_score_gemma":0.011361543,"teacher_disagreement_score":0.0059381863,"about_ca_system_score_codex":0.0005118801,"about_ca_system_score_gemma":0.0011628757,"threshold_uncertainty_score":0.013635278},"labels":[],"label_agreement":null},{"id":"W2161014506","doi":"","title":"A Corpus-based Method for Extracting Paraphrases of Emotion Terms","year":2010,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"WordNet; Bootstrapping (finance); Computer science; Natural language processing; Artificial intelligence","score_opus":0.01788797651845101,"score_gpt":0.33900779439204803,"score_spread":0.321119817873597,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2161014506","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.040766694,0.0014663897,0.93193287,0.0005448395,0.00025763235,0.0015985998,0.0074088946,0.005521974,0.010502125],"genre_scores_gemma":[0.08000858,0.0006557767,0.90096253,0.00011652327,0.00011158541,0.0014015862,0.012063173,0.00043401215,0.0042462633],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.998613,0.0003616355,0.00020387402,0.00033040368,0.00044895278,0.00004208254],"domain_scores_gemma":[0.99371654,0.0022561706,0.00048425017,0.0012861124,0.0021539745,0.000102977574],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012332357,0.0010098943,0.0006505595,0.0044919527,0.0011066815,0.0011882909,0.0010105396,0.00082589395,0.007017356],"category_scores_gemma":[0.009240662,0.00047225083,0.0006584213,0.0042320876,0.00054273225,0.0022513631,0.0011591348,0.0013039266,0.0036807097],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022419791,0.00030461935,0.0023292017,0.0018666055,0.00020138336,0.0008514687,0.0016657296,0.0020299584,0.16644377,0.006746701,0.023581287,0.79375505],"study_design_scores_gemma":[0.00037032997,0.0012556126,0.050903898,0.00091818,0.00094310613,0.014012755,0.00354201,0.23286663,0.32472035,0.029091028,0.34090918,0.0004668184],"about_ca_topic_score_codex":0.0013158226,"about_ca_topic_score_gemma":0.0030339858,"teacher_disagreement_score":0.007017356,"about_ca_system_score_codex":0.0004605291,"about_ca_system_score_gemma":0.0011210701,"threshold_uncertainty_score":0.023475349},"labels":[],"label_agreement":null},{"id":"W2162017663","doi":"10.3115/1620932.1620940","title":"Multiple word alignment with profile hidden Markov models","year":2009,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":26,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Hidden Markov model; Computer science; Word (group theory); Cognate; Task (project management); Set (abstract data type); Artificial intelligence; Maximum-entropy Markov model; Pattern recognition (psychology); Matching (statistics); Markov chain; Sequence labeling; Markov model; Speech recognition; Natural language processing; Variable-order Markov model; Machine learning; Mathematics; Statistics; Linguistics; Engineering","score_opus":0.011688440058398846,"score_gpt":0.23958822137634536,"score_spread":0.22789978131794653,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2162017663","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0059130117,0.00017772376,0.9889206,0.0001118292,0.0000382299,0.00005703849,0.00043906897,0.003580825,0.0007617072],"genre_scores_gemma":[0.19500653,0.0004758696,0.797799,0.00014663693,0.00006744357,0.00029960685,0.0032903475,0.0007352916,0.0021792839],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9973869,0.0013755531,0.00017698458,0.00057380536,0.00038578044,0.000100893],"domain_scores_gemma":[0.993753,0.004012743,0.00046733452,0.0011207797,0.0005200481,0.00012601641],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002795274,0.0009944765,0.0011217779,0.0017147257,0.00082589756,0.0016996824,0.001650982,0.001561822,0.004029671],"category_scores_gemma":[0.015788976,0.001066859,0.0017201321,0.0024127283,0.0005330289,0.005904312,0.0020331745,0.0022287702,0.0034145277],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011066265,0.00033840313,0.004250447,0.00071839243,0.0003994344,0.00063565245,0.0007299018,0.29455823,0.013644254,0.06633245,0.013528861,0.6037573],"study_design_scores_gemma":[0.000031027164,0.000059730613,0.00036442967,0.00003460588,0.000033155713,0.00013592273,0.000060234124,0.9188928,0.005015883,0.070890956,0.004448427,0.000032916396],"about_ca_topic_score_codex":0.0031875053,"about_ca_topic_score_gemma":0.004854502,"teacher_disagreement_score":0.004029671,"about_ca_system_score_codex":0.0007538606,"about_ca_system_score_gemma":0.0016896076,"threshold_uncertainty_score":0.014783025},"labels":[],"label_agreement":null},{"id":"W2162501006","doi":"10.1093/oxfordhb/9780199544004.013.0017","title":"Lexical-Functional Grammar","year":2012,"lang":"en","type":"book-chapter","venue":"Oxford University Press eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Generative grammar; Computer science; Section (typography); Linguistics; Grammar; Head-driven phrase structure grammar; Natural language processing; Lexical functional grammar; Phrase structure rules; Artificial intelligence; Constraint (computer-aided design); Mildly context-sensitive grammar formalism; Mathematics; Philosophy","score_opus":0.027639773355824637,"score_gpt":0.2093949663428124,"score_spread":0.18175519298698778,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2162501006","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0077206525,0.007932494,0.40396392,0.007196587,0.0005028459,0.00014575612,0.001104362,0.0012034518,0.57022995],"genre_scores_gemma":[0.52617365,0.007301289,0.2160057,0.0033734476,0.0008765972,0.00051026244,0.0032354116,0.001090193,0.24143349],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9992207,0.00029168255,0.00005401258,0.00018642955,0.000189938,0.00005718898],"domain_scores_gemma":[0.9993467,0.0003433868,0.00002406627,0.00011973907,0.00013531283,0.000030745934],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009922879,0.00055083894,0.00047824075,0.0013906097,0.00097217655,0.002973651,0.0010525354,0.0014123498,0.022820186],"category_scores_gemma":[0.0016858525,0.0003016082,0.00065154483,0.0014833948,0.0035202205,0.0040435353,0.0011836027,0.0013015597,0.0043502543],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000029831494,0.0000032740809,0.00005428752,0.000030980806,0.0000026708285,0.00003924735,0.000108726366,0.000533451,0.0001602993,0.9793241,0.0059022754,0.013837794],"study_design_scores_gemma":[0.0000047615968,0.000004288555,0.00009081239,0.000041585074,0.0000031157201,0.000093156756,0.000036994716,0.002399402,0.00017714228,0.8847041,0.112438835,0.0000057844913],"about_ca_topic_score_codex":0.0025308682,"about_ca_topic_score_gemma":0.0021000917,"teacher_disagreement_score":0.022820186,"about_ca_system_score_codex":0.002816719,"about_ca_system_score_gemma":0.0013131909,"threshold_uncertainty_score":0.07634109},"labels":[],"label_agreement":null},{"id":"W2163364265","doi":"10.3115/1614108.1614116","title":"A fast method for parallel document identification","year":2007,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; University of Alberta","keywords":"Computer science; Identification (biology); Homogeneous; Information retrieval; Domain (mathematical analysis); Parallel corpora; Frequency domain; Parallel algorithm; Natural language processing; Data mining; Artificial intelligence; Algorithm; Mathematics","score_opus":0.015423030749939518,"score_gpt":0.35153861150695154,"score_spread":0.33611558075701203,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2163364265","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0052837674,0.0006144142,0.9862091,0.000123841,0.00025509007,0.00021760014,0.00056548783,0.004851453,0.0018792308],"genre_scores_gemma":[0.028910434,0.00034627804,0.96152186,0.00008283825,0.00018754254,0.00030627873,0.0014556559,0.0005219921,0.006667098],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9964218,0.0004055357,0.00031390463,0.0010239951,0.001597754,0.00023695217],"domain_scores_gemma":[0.994158,0.00142291,0.00043960576,0.001539687,0.0022408674,0.00019899708],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016132947,0.0014610989,0.0013611193,0.006939046,0.00184158,0.0023096148,0.001852555,0.0016088807,0.009357414],"category_scores_gemma":[0.008341824,0.0007458765,0.0010871436,0.005688404,0.0009071027,0.0027753965,0.002315731,0.0021798892,0.013029624],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019424077,0.00008680417,0.001072613,0.00026954725,0.00007578896,0.0002499487,0.00019027393,0.0027297456,0.034637086,0.004479729,0.011396481,0.94461775],"study_design_scores_gemma":[0.00026637665,0.00044193916,0.007208276,0.00020315991,0.00029089994,0.007807797,0.0006504609,0.4673138,0.19663352,0.053501423,0.26516157,0.0005206872],"about_ca_topic_score_codex":0.0022745188,"about_ca_topic_score_gemma":0.002715134,"teacher_disagreement_score":0.009357414,"about_ca_system_score_codex":0.00049379416,"about_ca_system_score_gemma":0.0019830207,"threshold_uncertainty_score":0.031303704},"labels":[],"label_agreement":null},{"id":"W2163542005","doi":"10.2196/medinform.4211","title":"Context-Sensitive Spelling Correction of Consumer-Generated Content on Health Care","year":2015,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Spelling; Context (archaeology); Perspective (graphical); Computer science; Social media; Health care; Content (measure theory); User-generated content; Internet privacy; World Wide Web; Advertising; Artificial intelligence; Linguistics; Business","score_opus":0.04371315099819198,"score_gpt":0.3247550993567014,"score_spread":0.2810419483585094,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2163542005","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6579985,0.0014464554,0.29786703,0.0006161033,0.000503648,0.0021216013,0.0044167438,0.025008703,0.01002121],"genre_scores_gemma":[0.6906237,0.00047804601,0.29970282,0.00020564775,0.0001512491,0.00037929713,0.0032834618,0.00089670304,0.0042791367],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9957547,0.0014058235,0.00051406876,0.0006959059,0.0014796421,0.00014985615],"domain_scores_gemma":[0.9741717,0.011520581,0.003498199,0.0025242642,0.007930086,0.0003552093],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033278912,0.0010214336,0.0005713083,0.0036671702,0.00063901325,0.0014403022,0.0007173602,0.0006525848,0.0027904024],"category_scores_gemma":[0.02298904,0.00020310002,0.00047737264,0.0018586523,0.00055189314,0.0013233117,0.00090588914,0.00046961664,0.0014950915],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014817083,0.000283993,0.03244091,0.001666511,0.00017006275,0.0012331123,0.0032515475,0.0047928733,0.12381929,0.0016546235,0.010330563,0.8188748],"study_design_scores_gemma":[0.00017680784,0.0014173051,0.13469295,0.0005855736,0.0006392296,0.0028839896,0.0026631427,0.24135445,0.5523466,0.004544067,0.058239806,0.00045611826],"about_ca_topic_score_codex":0.0040505156,"about_ca_topic_score_gemma":0.004910361,"teacher_disagreement_score":0.0040505156,"about_ca_system_score_codex":0.000999816,"about_ca_system_score_gemma":0.0016052573,"threshold_uncertainty_score":0.017599761},"labels":[],"label_agreement":null},{"id":"W2164017066","doi":"10.1145/2509558.2509573","title":"Knowledge base population and visualization using an ontology based on semantic roles","year":2013,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Boeing","keywords":"Computer science; Ontology; Knowledge base; Visualization; Predicate (mathematical logic); Information retrieval; Upper ontology; Natural language processing; Semantic Web; Exploit; Natural language; Semantics (computer science); World Wide Web; Artificial intelligence; Programming language","score_opus":0.024266261766510897,"score_gpt":0.3222496998435002,"score_spread":0.2979834380769893,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2164017066","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01982306,0.0002036368,0.9598971,0.0009335209,0.000055428198,0.00021605221,0.0014834742,0.008510455,0.008877295],"genre_scores_gemma":[0.13513963,0.00035702935,0.85705453,0.00014418481,0.00002933676,0.00020233545,0.0026182854,0.00068793946,0.0037667998],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99931896,0.00014667212,0.000069352944,0.00017684104,0.00024670546,0.000041530708],"domain_scores_gemma":[0.9982278,0.00089431705,0.00015668018,0.0003009919,0.0003092443,0.0001109759],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013828935,0.00048227227,0.00044065266,0.0044883443,0.0010506585,0.004425316,0.00097177556,0.0006777201,0.0037104243],"category_scores_gemma":[0.0046571586,0.00040161234,0.0007745779,0.0028524352,0.00070023414,0.0050598904,0.002008166,0.0012873982,0.00097379467],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028573276,0.00035999162,0.006498638,0.00065944897,0.00011932841,0.0014918962,0.011090865,0.020062577,0.03901026,0.17429252,0.032759935,0.7133688],"study_design_scores_gemma":[0.00012786638,0.00011484956,0.0054888865,0.00038166557,0.00020627205,0.0012549504,0.00409936,0.39932582,0.0605749,0.21034041,0.31790102,0.00018405975],"about_ca_topic_score_codex":0.0060846633,"about_ca_topic_score_gemma":0.012036229,"teacher_disagreement_score":0.0060846633,"about_ca_system_score_codex":0.0007958903,"about_ca_system_score_gemma":0.001434719,"threshold_uncertainty_score":0.012412608},"labels":[],"label_agreement":null},{"id":"W2164441316","doi":"","title":"PORT: a Precision-Order-Recall MT Evaluation Metric for Tuning","year":2012,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"BLEU; Computer science; Metric (unit); Port (circuit theory); Machine translation; Artificial intelligence; Precision and recall; Translation (biology); Recall; Natural language processing; Machine learning; Engineering; Electronic engineering","score_opus":0.03752141005855442,"score_gpt":0.33421211678035184,"score_spread":0.2966907067217974,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2164441316","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13625686,0.0053895614,0.75013036,0.0005762035,0.0007267901,0.0017297906,0.013820157,0.054241333,0.037128925],"genre_scores_gemma":[0.46881056,0.0008422564,0.49382964,0.00037589876,0.00027255437,0.001975679,0.020844506,0.0050996523,0.007949305],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.988186,0.0048416886,0.0017675065,0.0014785825,0.0034754213,0.00025094586],"domain_scores_gemma":[0.973435,0.013644919,0.0021599883,0.0040233745,0.0063503897,0.00038629072],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009344005,0.002170245,0.0015462597,0.0071317195,0.0011635432,0.002539852,0.0016431881,0.0018303894,0.004670732],"category_scores_gemma":[0.044300083,0.00056534435,0.0009153574,0.0048596677,0.0007074289,0.0036509705,0.0016763789,0.0012267213,0.0032000106],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015185094,0.0005703535,0.023732094,0.0020320674,0.0011540493,0.00031374363,0.000530909,0.043130934,0.06470606,0.005760361,0.06690937,0.7896416],"study_design_scores_gemma":[0.0005229572,0.0034401112,0.076262176,0.0003684706,0.0009791697,0.0035282094,0.00043610987,0.5907631,0.21967041,0.016225182,0.08699013,0.0008139299],"about_ca_topic_score_codex":0.0027863784,"about_ca_topic_score_gemma":0.004085547,"teacher_disagreement_score":0.009344005,"about_ca_system_score_codex":0.0014485802,"about_ca_system_score_gemma":0.0012522707,"threshold_uncertainty_score":0.049416423},"labels":[],"label_agreement":null},{"id":"W2164788644","doi":"10.3115/1118693.1118713","title":"User-friendly text prediction for translators","year":2002,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":74,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Productivity; Text generation; Artificial intelligence; Machine translation; Production (economics); Translation (biology); Natural language processing; Machine learning","score_opus":0.012878219089769764,"score_gpt":0.24698374995249364,"score_spread":0.23410553086272387,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2164788644","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06403967,0.0003537438,0.8691216,0.0013617828,0.00011104407,0.00030432257,0.0008541937,0.056299526,0.007554108],"genre_scores_gemma":[0.5682597,0.00033560253,0.41634938,0.00038901088,0.00018865637,0.00055248453,0.00226041,0.0020880974,0.009576623],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99749446,0.0014398884,0.00008931807,0.0004810234,0.0004150538,0.00008039573],"domain_scores_gemma":[0.9806276,0.013089704,0.0007071636,0.0029046927,0.0022398552,0.00043115154],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003427965,0.0015813706,0.0011477568,0.0007171507,0.000764814,0.0016133323,0.0016189181,0.0016423459,0.011709879],"category_scores_gemma":[0.026134836,0.00062714296,0.00058025017,0.0010414534,0.00052605197,0.0032965215,0.001488456,0.0015887144,0.008041794],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0033165643,0.0005715312,0.012024275,0.00082883035,0.0001318078,0.0011065028,0.0030407144,0.073222436,0.04568955,0.006670165,0.06755529,0.7858424],"study_design_scores_gemma":[0.00018433058,0.00037736935,0.0023455443,0.00008505419,0.000076886376,0.00047162492,0.00052410114,0.9206556,0.029293515,0.014583384,0.031296127,0.00010649035],"about_ca_topic_score_codex":0.002415012,"about_ca_topic_score_gemma":0.0037171727,"teacher_disagreement_score":0.011709879,"about_ca_system_score_codex":0.0005120675,"about_ca_system_score_gemma":0.000982921,"threshold_uncertainty_score":0.039173424},"labels":[],"label_agreement":null},{"id":"W2164948578","doi":"10.1162/089120103321337458","title":"Word Reordering and a Dynamic Programming Beam Search Algorithm for Statistical Machine Translation","year":2003,"lang":"en","type":"article","venue":"Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":257,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Word (group theory); Task (project management); Machine translation; Vocabulary; Beam search; Word error rate; Sentence; Translation (biology); Artificial intelligence; Natural language processing; Dynamic programming; Search algorithm; Speech recognition; Set (abstract data type); Algorithm; Linguistics; Programming language","score_opus":0.01830414375166094,"score_gpt":0.31965641621504903,"score_spread":0.3013522724633881,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2164948578","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0014083534,0.00013782909,0.9973579,0.00006170898,0.000023110566,0.000034261197,0.000031381398,0.00037788934,0.0005675186],"genre_scores_gemma":[0.02167759,0.00017119285,0.97608644,0.00010596549,0.000032599582,0.00027717944,0.0002047074,0.00022751244,0.001216878],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99808973,0.00097165554,0.00011424692,0.00031017498,0.00042452133,0.000089611196],"domain_scores_gemma":[0.9977291,0.0015398007,0.00011212955,0.00022372176,0.00035806146,0.000037247435],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002154728,0.0011037582,0.0014034577,0.0015253222,0.0007651085,0.0010682808,0.0015756215,0.0013374164,0.00535028],"category_scores_gemma":[0.0062715933,0.00096797367,0.0010908123,0.0031136554,0.0011334885,0.002148295,0.0012496593,0.001946921,0.0023577474],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021296818,0.00012143035,0.00043493017,0.00028256673,0.00016347099,0.00015146837,0.00029468865,0.2715408,0.010630956,0.113588475,0.008608317,0.59396994],"study_design_scores_gemma":[0.000063685104,0.00006043009,0.00010538772,0.000018915709,0.00002505707,0.00007336041,0.000026868063,0.9234301,0.0028920993,0.066458926,0.006819246,0.000025901725],"about_ca_topic_score_codex":0.0031225486,"about_ca_topic_score_gemma":0.0043194434,"teacher_disagreement_score":0.00535028,"about_ca_system_score_codex":0.00074928574,"about_ca_system_score_gemma":0.0017172563,"threshold_uncertainty_score":0.0178985},"labels":[],"label_agreement":null},{"id":"W2165710864","doi":"10.1145/360487.360477","title":"A personal view of APL","year":2000,"lang":"en","type":"article","venue":"ACM SIGAPL APL Quote Quad","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Public Health","funders":"","keywords":"Spelling; Terminology; Grammar; Linguistics; Function (biology); Computer science; Philosophy","score_opus":0.015012439088719055,"score_gpt":0.2736670959864559,"score_spread":0.2586546568977368,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2165710864","genre_codex":"other","genre_gemma":"commentary","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"commentary","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.002628261,0.015897388,0.015046967,0.31503928,0.013618225,0.000011874561,0.00017197827,0.0009099109,0.63667613],"genre_scores_gemma":[0.08878994,0.025223386,0.0105248215,0.1350406,0.015308101,0.000047735342,0.00017215619,0.0008446498,0.7240486],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99816704,0.00057698874,0.00006843134,0.00030568003,0.00072015455,0.00016177175],"domain_scores_gemma":[0.99762976,0.000708028,0.00008907965,0.00026002753,0.00086958386,0.00044352646],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016274358,0.00035629934,0.00022090334,0.0007818308,0.0048045493,0.008048073,0.0006646031,0.0022918666,0.029189356],"category_scores_gemma":[0.004170092,0.00022731152,0.00020256912,0.0010605737,0.004293723,0.007683004,0.0025414883,0.00795152,0.013485818],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00001811141,0.000020775231,0.0002740991,0.000043942262,0.000004908913,0.0001484829,0.0042165266,0.00013571023,0.0005005175,0.3582327,0.5813937,0.055010468],"study_design_scores_gemma":[9.230705e-7,0.000004229566,0.000037061516,0.000030922118,0.000001063581,0.00012492396,0.0003693888,0.00005720521,0.00008170374,0.014193767,0.9850948,0.000003878852],"about_ca_topic_score_codex":0.00212803,"about_ca_topic_score_gemma":0.0021386172,"teacher_disagreement_score":0.029189356,"about_ca_system_score_codex":0.001766554,"about_ca_system_score_gemma":0.0023512007,"threshold_uncertainty_score":0.097648084},"labels":[],"label_agreement":null},{"id":"W2166001215","doi":"","title":"Going Beyond Word Cooccurrences in Global Lexical Selection for Statistical Machine Translation using a Multilayer Perceptron","year":2011,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Phrase; Artificial intelligence; Natural language processing; Machine translation; Word (group theory); Sentence; Perceptron; Selection (genetic algorithm); Speech recognition; Context (archaeology); Translation (biology); Example-based machine translation; Artificial neural network; Linguistics","score_opus":0.05879728227907942,"score_gpt":0.34460445237535325,"score_spread":0.2858071700962738,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2166001215","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.047074758,0.00040284975,0.9492538,0.00025279992,0.000050096336,0.00004057927,0.00009830656,0.0016083252,0.0012185025],"genre_scores_gemma":[0.73483366,0.00046842417,0.25840497,0.00023719606,0.00012424683,0.00013059807,0.0004342915,0.00034393542,0.0050227703],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99885476,0.0004571888,0.000108366745,0.00025690912,0.00021715234,0.000105630235],"domain_scores_gemma":[0.9979163,0.0010666336,0.00020002747,0.00038015243,0.00035824033,0.00007871283],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024147134,0.000886542,0.0011710525,0.0008141723,0.00054105156,0.0010979553,0.0009384566,0.0008300403,0.0028147697],"category_scores_gemma":[0.0046021743,0.00054875534,0.0007866679,0.001297607,0.0007461413,0.002951903,0.0012669107,0.0015477351,0.0016346283],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007229701,0.00025460793,0.0053928443,0.00037939384,0.0004033892,0.00053079275,0.00036931105,0.27092874,0.08163612,0.015545797,0.003962205,0.6198739],"study_design_scores_gemma":[0.000030149396,0.00015646948,0.0012180993,0.000023632912,0.0000925325,0.00012372645,0.000034879915,0.96058154,0.016674606,0.01950551,0.0015218793,0.000037054127],"about_ca_topic_score_codex":0.0031160614,"about_ca_topic_score_gemma":0.0071311216,"teacher_disagreement_score":0.0031160614,"about_ca_system_score_codex":0.00050419226,"about_ca_system_score_gemma":0.00096453296,"threshold_uncertainty_score":0.012770414},"labels":[],"label_agreement":null},{"id":"W2166202273","doi":"","title":"Bootstrapping a Stochastic Transducer for Arabic-English Transliteration Extraction","year":2007,"lang":"en","type":"article","venue":"Meeting of the Association for Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Bootstrapping (finance); Computer science; Transducer; Scripting language; Artificial intelligence; Natural language processing; Metric (unit); Transliteration; Task (project management); Speech recognition; Similarity (geometry); Arabic; Pattern recognition (psychology); Programming language; Linguistics; Acoustics; Engineering; Mathematics","score_opus":0.014098183086053236,"score_gpt":0.2948093699806726,"score_spread":0.2807111868946193,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2166202273","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.047072686,0.000107674845,0.94682884,0.00030352827,0.00006859448,0.0000640031,0.00021193104,0.0043966672,0.0009460948],"genre_scores_gemma":[0.63805217,0.00011862488,0.3563839,0.00031698906,0.000093244984,0.000271262,0.0015461756,0.0003394151,0.002878192],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989686,0.00037671614,0.00006543864,0.00028745402,0.00021425955,0.00008745825],"domain_scores_gemma":[0.996691,0.0022547306,0.00013747632,0.0002654483,0.00055128516,0.00009999697],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001306289,0.00069780764,0.00083025,0.00066302216,0.0004962951,0.000598632,0.0011350202,0.0010299319,0.0026476888],"category_scores_gemma":[0.006989904,0.00042091834,0.00068795244,0.0005384317,0.0005704576,0.0012944487,0.0012572567,0.001568389,0.002431586],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00058879476,0.00028494856,0.0045080297,0.00020808469,0.00011220447,0.00049985806,0.00051595946,0.12273517,0.101019904,0.008974833,0.0064389626,0.75411326],"study_design_scores_gemma":[0.000011386517,0.00009159778,0.0005117472,0.00000664707,0.0000145657295,0.00010961638,0.00003794138,0.9742848,0.019387374,0.0043167663,0.0012104389,0.000016993632],"about_ca_topic_score_codex":0.0022320377,"about_ca_topic_score_gemma":0.0037447782,"teacher_disagreement_score":0.0026476888,"about_ca_system_score_codex":0.00039810245,"about_ca_system_score_gemma":0.0010772066,"threshold_uncertainty_score":0.008857429},"labels":[],"label_agreement":null},{"id":"W2166281503","doi":"","title":"DalTREC 2004: Question Answering using Regular Expression Rewriting","year":2004,"lang":"en","type":"article","venue":"Text REtrieval Conference","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Dalhousie University","funders":"","keywords":"Rewriting; Regular expression; Computer science; Expression (computer science); Question answering; Search engine; Information retrieval; Track (disk drive); Programming language; Natural language processing; World Wide Web; Artificial intelligence; Operating system","score_opus":0.028250012493525753,"score_gpt":0.2986974757111274,"score_spread":0.27044746321760166,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2166281503","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05558458,0.003463844,0.57046056,0.0030892123,0.0009544568,0.0023454223,0.01587123,0.30800667,0.040223982],"genre_scores_gemma":[0.27398852,0.0012416862,0.57071567,0.0022419137,0.00040752813,0.0010471623,0.08058278,0.012791768,0.05698301],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9944887,0.0021359182,0.00033993943,0.0013528571,0.001348785,0.00033379774],"domain_scores_gemma":[0.9940136,0.0021863894,0.00020036373,0.0017836149,0.0015725913,0.00024349625],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0050573153,0.0014108045,0.0016734877,0.001758084,0.0009932478,0.003024858,0.0035656083,0.0022874332,0.018586976],"category_scores_gemma":[0.010510463,0.000895262,0.0012459375,0.001162159,0.0010838478,0.004356581,0.002328982,0.003061239,0.011126745],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014841567,0.0009335515,0.0015840753,0.0019325733,0.00038149062,0.0007738547,0.0012717266,0.017940503,0.08060946,0.021937454,0.3731783,0.49797282],"study_design_scores_gemma":[0.0010704933,0.00079660176,0.003957591,0.0001442709,0.00025004422,0.0013886854,0.0005377437,0.2815157,0.1685955,0.026552418,0.5148561,0.0003349071],"about_ca_topic_score_codex":0.012521504,"about_ca_topic_score_gemma":0.011731834,"teacher_disagreement_score":0.018586976,"about_ca_system_score_codex":0.0020801795,"about_ca_system_score_gemma":0.0019075205,"threshold_uncertainty_score":0.062179565},"labels":[],"label_agreement":null},{"id":"W2166323096","doi":"10.3115/1117586.1117592","title":"Pre-processing closed captions for machine translation","year":2000,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Pipeline (software); Machine translation; Natural language processing; Artificial intelligence; Translation (biology); Speech recognition; Speech translation; Translation system; Machine translation system; Programming language","score_opus":0.018399525557642404,"score_gpt":0.29350888927938096,"score_spread":0.27510936372173855,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2166323096","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0051858765,0.00056899467,0.972808,0.00034960167,0.00077297457,0.0003319151,0.0009975602,0.009288757,0.009696347],"genre_scores_gemma":[0.06591547,0.000699162,0.9145966,0.00031985465,0.0006918866,0.0007055948,0.0054196753,0.0025138443,0.0091378745],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9972556,0.0011299107,0.00024239814,0.0005536127,0.00067588815,0.00014265081],"domain_scores_gemma":[0.9923028,0.0026941947,0.0004172492,0.0014171746,0.0029945245,0.00017410419],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018198757,0.0020086295,0.0011350537,0.0016735673,0.0018098708,0.004034198,0.001259837,0.0017240067,0.027753327],"category_scores_gemma":[0.010979117,0.00078523107,0.0010509889,0.0018959144,0.0011951016,0.002837423,0.002013378,0.0025744115,0.021969596],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00064406835,0.0002560593,0.0008133398,0.0017958645,0.00012718442,0.0015889014,0.002125358,0.014884398,0.19615293,0.047375455,0.076935716,0.6573008],"study_design_scores_gemma":[0.0000963514,0.00068038807,0.0025915683,0.0004263558,0.00016941498,0.002810462,0.0011601807,0.18902344,0.2608942,0.09187619,0.4499599,0.00031153974],"about_ca_topic_score_codex":0.00079411146,"about_ca_topic_score_gemma":0.0014792003,"teacher_disagreement_score":0.027753327,"about_ca_system_score_codex":0.00063652475,"about_ca_system_score_gemma":0.0011583494,"threshold_uncertainty_score":0.09284413},"labels":[],"label_agreement":null},{"id":"W2166391512","doi":"10.1162/089120101317066122","title":"Automatic Verb Classification Based on Statistical Distributions of Argument Structure","year":2001,"lang":"en","type":"article","venue":"Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":199,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada; University of Toronto; National Science Foundation","keywords":"Computer science; Natural language processing; Artificial intelligence; Verb; Sentence; Argument (complex analysis); Classifier (UML); Predicate (mathematical logic)","score_opus":0.01714931419756963,"score_gpt":0.3015708389109142,"score_spread":0.28442152471334453,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2166391512","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.25223574,0.00018228244,0.7391645,0.0002967717,0.000038145998,0.00015791577,0.0006159189,0.0039415783,0.003367217],"genre_scores_gemma":[0.7937898,0.000090302434,0.20212162,0.000069380316,0.000040239887,0.00019695482,0.002203321,0.00027135477,0.001217125],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99777883,0.0008937017,0.00016368803,0.00056866114,0.00044758295,0.00014751487],"domain_scores_gemma":[0.98463583,0.011046579,0.0013224498,0.001235742,0.0015916907,0.00016772025],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034816982,0.00053998147,0.0007507592,0.0029598977,0.0005772026,0.0017786938,0.0011197393,0.00092800235,0.0015468303],"category_scores_gemma":[0.0184218,0.0003331444,0.00054880057,0.001384856,0.00081181293,0.0029430967,0.00091376435,0.0011377904,0.0013255482],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005950482,0.00043215856,0.06581441,0.00025726852,0.00010328807,0.00020263555,0.00083717133,0.040349167,0.04970795,0.016767776,0.0067420118,0.8181911],"study_design_scores_gemma":[0.000031945594,0.000054987926,0.010830311,0.00002383741,0.000014685691,0.00014692786,0.0001444319,0.95185494,0.014440876,0.020807667,0.0016264728,0.00002299279],"about_ca_topic_score_codex":0.0012300536,"about_ca_topic_score_gemma":0.0019477764,"teacher_disagreement_score":0.0034816982,"about_ca_system_score_codex":0.0007458882,"about_ca_system_score_gemma":0.0008791592,"threshold_uncertainty_score":0.018413186},"labels":[],"label_agreement":null},{"id":"W2166456646","doi":"10.3115/1613704.1613706","title":"Distinguishing subtypes of multiword expressions using linguistically-motivated statistical measures","year":2007,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":67,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Lexicon; Idiosyncrasy; Natural language processing; Artificial intelligence; Property (philosophy); Encoding (memory); Semantics (computer science)","score_opus":0.031241313900526126,"score_gpt":0.3303614893980023,"score_spread":0.29912017549747616,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2166456646","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.60504884,0.00048258962,0.38076088,0.0008996966,0.000113673304,0.00037323602,0.0008788094,0.00097330177,0.010469016],"genre_scores_gemma":[0.8585174,0.00013987329,0.13835505,0.00018214718,0.00011852318,0.0003098796,0.0010997952,0.00029260054,0.0009847727],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9954314,0.0010611879,0.0008612928,0.0011763173,0.0012493026,0.00022056907],"domain_scores_gemma":[0.9667213,0.016522316,0.0065232525,0.004256006,0.0047858264,0.0011912932],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004271748,0.0005707252,0.00092191447,0.004641969,0.001540457,0.0043972437,0.0013859953,0.0011734539,0.0021128985],"category_scores_gemma":[0.028366065,0.00041189505,0.00076794927,0.0030305292,0.002097036,0.006026143,0.0019409475,0.001491594,0.0007508171],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012281031,0.0004727601,0.1862635,0.00084496534,0.00024869895,0.0012129146,0.00628765,0.011161462,0.12482761,0.25775424,0.004155061,0.405543],"study_design_scores_gemma":[0.00010471412,0.0004795476,0.11453605,0.00022919835,0.00025531277,0.003036696,0.005597788,0.3152317,0.055644423,0.4888057,0.015714522,0.00036424334],"about_ca_topic_score_codex":0.0010864297,"about_ca_topic_score_gemma":0.0015607636,"teacher_disagreement_score":0.004641969,"about_ca_system_score_codex":0.0011023507,"about_ca_system_score_gemma":0.0010962778,"threshold_uncertainty_score":0.022591412},"labels":[],"label_agreement":null},{"id":"W2166557989","doi":"10.1007/978-3-642-21538-4_21","title":"Question Type Classification Using a Part-of-Speech Hierarchy","year":2011,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Lakehead University","funders":"","keywords":"Computer science; Hierarchy; Type (biology); Artificial intelligence; Natural language processing; Task (project management)","score_opus":0.048633084525883524,"score_gpt":0.30054311330116584,"score_spread":0.25191002877528235,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2166557989","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06308928,0.0025484576,0.8868565,0.0010221823,0.0004947181,0.0016391587,0.011179078,0.017489934,0.015680697],"genre_scores_gemma":[0.22855233,0.0009294403,0.7362159,0.00034595543,0.00029847305,0.0006387597,0.020662205,0.00071268645,0.0116441995],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9972784,0.00039666335,0.0004017667,0.0008311742,0.00077182427,0.00032025494],"domain_scores_gemma":[0.99438584,0.0025698585,0.00034700485,0.0005122338,0.0018071862,0.00037797066],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025941138,0.0013863596,0.0019174223,0.009384253,0.0017405562,0.0037122208,0.0026160458,0.0020421096,0.012897888],"category_scores_gemma":[0.0065202164,0.00058538426,0.0025402422,0.0049924045,0.0007923627,0.0051387236,0.0019021509,0.0020378463,0.008213534],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010077752,0.0005503328,0.01406734,0.0010101132,0.00019042594,0.0005244551,0.0009586667,0.00398656,0.06730761,0.011541429,0.02485213,0.8740032],"study_design_scores_gemma":[0.00021539924,0.000676852,0.029779345,0.0005616156,0.0011152957,0.002128159,0.0023058585,0.69250786,0.10484946,0.09087343,0.07465676,0.00032998022],"about_ca_topic_score_codex":0.0075831455,"about_ca_topic_score_gemma":0.010245076,"teacher_disagreement_score":0.012897888,"about_ca_system_score_codex":0.0014321669,"about_ca_system_score_gemma":0.0025360002,"threshold_uncertainty_score":0.043147683},"labels":[],"label_agreement":null},{"id":"W2166669788","doi":"10.1007/11841920_8","title":"Efficient Incremental Validation of XML Documents After Composite Updates","year":2006,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Computer science; XML validation; RELAX NG; XML; XML Schema (W3C); Document Structure Description; Scalability; Streaming XML; Schema (genetic algorithms); Efficient XML Interchange; Programming language; XML Schema Editor; Database; Information retrieval; Document type definition; World Wide Web","score_opus":0.006390539843762291,"score_gpt":0.24618337710327273,"score_spread":0.23979283725951045,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2166669788","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11157799,0.001225126,0.82751954,0.00053210347,0.000528161,0.00066052686,0.0016112971,0.050711315,0.005633876],"genre_scores_gemma":[0.52451104,0.0004091646,0.44969416,0.0003333047,0.00018319553,0.00032302257,0.0050322977,0.007524458,0.011989367],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98913366,0.0026629656,0.0010034551,0.0014214868,0.0048716697,0.00090680475],"domain_scores_gemma":[0.9347841,0.03286778,0.0023016525,0.022293933,0.007204689,0.00054775947],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00786305,0.0014739829,0.0020711687,0.0026070534,0.0015089083,0.004954949,0.0049332944,0.0021948162,0.0057866536],"category_scores_gemma":[0.033859678,0.0018277002,0.0020455536,0.002608193,0.0018959588,0.007573678,0.004853144,0.002144955,0.0025941394],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0033433167,0.0005276998,0.008475868,0.00116425,0.00029782104,0.0016221472,0.0018430105,0.06406957,0.09468793,0.031528387,0.026168132,0.7662719],"study_design_scores_gemma":[0.0003142681,0.00041597465,0.0035787255,0.00024014869,0.0004454117,0.0009851684,0.0005135249,0.59384775,0.31065828,0.05899124,0.029763516,0.0002460193],"about_ca_topic_score_codex":0.0042034034,"about_ca_topic_score_gemma":0.0056271534,"teacher_disagreement_score":0.00786305,"about_ca_system_score_codex":0.0012264409,"about_ca_system_score_gemma":0.0027236973,"threshold_uncertainty_score":0.041584253},"labels":[],"label_agreement":null},{"id":"W2166769824","doi":"10.1111/0824-7935.00190","title":"Generate and Repair Machine Translation","year":2002,"lang":"en","type":"article","venue":"Computational Intelligence","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Synchronous context-free grammar; Rule-based machine translation; Computer science; Machine translation; Example-based machine translation; Transfer-based machine translation; Machine translation software usability; Natural language processing; Phrase; Artificial intelligence; Translation (biology); Evaluation of machine translation; Dynamic and formal equivalence; Computer-assisted translation; Programming language; Syntax","score_opus":0.044620027198693665,"score_gpt":0.28815948407822406,"score_spread":0.24353945687953038,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2166769824","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006417581,0.00020480876,0.98012084,0.00039634533,0.00011366541,0.00017673713,0.00034529442,0.0058493232,0.0063754837],"genre_scores_gemma":[0.075994104,0.0003151576,0.91534793,0.00031013813,0.00007157742,0.00016646646,0.001468144,0.0013056555,0.005020736],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9965322,0.0012646933,0.00028907196,0.00070016464,0.0010254631,0.00018843506],"domain_scores_gemma":[0.99555343,0.0015761844,0.00036052405,0.0015407884,0.0008743603,0.00009467023],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030106269,0.0010966955,0.0010427769,0.0012928753,0.0008663404,0.0023951444,0.0017531468,0.001425165,0.008431555],"category_scores_gemma":[0.008250397,0.00054695667,0.0013589872,0.001269145,0.0010626282,0.002370494,0.0018780741,0.0012913124,0.0052444804],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003564552,0.00023308628,0.0021633746,0.0010654618,0.00023752838,0.0013563957,0.0012616048,0.067704536,0.03715029,0.15800619,0.03626815,0.6941969],"study_design_scores_gemma":[0.00015518138,0.0003082313,0.0010961572,0.00016972459,0.00018526643,0.0017366377,0.0004682163,0.5679156,0.07330708,0.14453806,0.20993444,0.00018527597],"about_ca_topic_score_codex":0.0012409148,"about_ca_topic_score_gemma":0.001545306,"teacher_disagreement_score":0.008431555,"about_ca_system_score_codex":0.0006292252,"about_ca_system_score_gemma":0.0016663625,"threshold_uncertainty_score":0.028206348},"labels":[],"label_agreement":null},{"id":"W2167302977","doi":"10.3115/1075096.1075115","title":"A comparative study on reordering constraints in statistical machine translation","year":2003,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":114,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Task (project management); Word (group theory); IBM; Translation (biology); Machine translation; Viterbi algorithm; Constraint (computer-aided design); Algorithm; Artificial intelligence; Theoretical computer science; Arithmetic; Decoding methods; Mathematics","score_opus":0.04258568960460794,"score_gpt":0.3498494213666288,"score_spread":0.3072637317620208,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2167302977","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.660036,0.033801228,0.25454086,0.0030901,0.00026959606,0.00036641795,0.0021537438,0.0028269975,0.042915076],"genre_scores_gemma":[0.8543249,0.005488832,0.13286921,0.00039353955,0.00025225713,0.00018742189,0.0040454706,0.00081671076,0.0016216462],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9880471,0.0067419508,0.0011420775,0.0009125021,0.0026451787,0.00051118556],"domain_scores_gemma":[0.9023264,0.087227404,0.0022872568,0.0042473148,0.0033571958,0.000554376],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009889782,0.0008578505,0.0017074528,0.002914223,0.0011574974,0.0023420625,0.0011287363,0.0013423957,0.0043039713],"category_scores_gemma":[0.041641004,0.0004657331,0.0007292458,0.0075262045,0.0012872466,0.005977451,0.001263374,0.0014049261,0.0007909608],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004028309,0.0005814332,0.013413855,0.0025296176,0.00036915764,0.00056954805,0.0010101794,0.25612354,0.015719015,0.055280887,0.010381421,0.639993],"study_design_scores_gemma":[0.00051435054,0.0018608115,0.014499037,0.00039028,0.00025327483,0.001081414,0.0012172889,0.85252774,0.022191245,0.07649457,0.028776452,0.00019355874],"about_ca_topic_score_codex":0.0070476034,"about_ca_topic_score_gemma":0.014958078,"teacher_disagreement_score":0.009889782,"about_ca_system_score_codex":0.001737657,"about_ca_system_score_gemma":0.0022708424,"threshold_uncertainty_score":0.052302778},"labels":[],"label_agreement":null},{"id":"W2167528246","doi":"10.1145/1099554.1099665","title":"Document clustering using character N-grams","year":2005,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":38,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Cluster analysis; Computer science; Character (mathematics); Word (group theory); Preprocessor; Artificial intelligence; Robustness (evolution); Curse of dimensionality; Document clustering; Pattern recognition (psychology); Dimension (graph theory); n-gram; Natural language processing; Language model; Mathematics; Combinatorics","score_opus":0.017233856515494885,"score_gpt":0.28882159379222083,"score_spread":0.27158773727672597,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2167528246","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011367679,0.000739899,0.9841412,0.00010751021,0.0001258007,0.00012822209,0.00033012682,0.0018803485,0.0011792006],"genre_scores_gemma":[0.0667794,0.0004953223,0.9280775,0.000082103914,0.00014312868,0.00017978848,0.0012084139,0.000202628,0.0028316642],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99860054,0.00023357803,0.00012888547,0.0003866234,0.00057675433,0.00007361974],"domain_scores_gemma":[0.9983565,0.00034335902,0.0002568074,0.00036647354,0.00061111496,0.000065625456],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000566967,0.0009284326,0.0011415939,0.004432191,0.0009170509,0.001288704,0.0011084862,0.00080706435,0.0012949865],"category_scores_gemma":[0.0030209564,0.0003271508,0.00092800497,0.0061078095,0.0005816671,0.0016730353,0.0009218433,0.0007812987,0.0023196086],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001850746,0.00016960951,0.0026900053,0.00037182073,0.00017856153,0.00012983804,0.00029173496,0.018895175,0.038587295,0.011859814,0.007051804,0.91958934],"study_design_scores_gemma":[0.00008063859,0.00037306183,0.0067454083,0.00012690587,0.00017956637,0.0014686947,0.00029580246,0.7858903,0.081662506,0.056783292,0.06610778,0.0002860618],"about_ca_topic_score_codex":0.0024479006,"about_ca_topic_score_gemma":0.00396837,"teacher_disagreement_score":0.004432191,"about_ca_system_score_codex":0.0006314492,"about_ca_system_score_gemma":0.0009741121,"threshold_uncertainty_score":0.0048673153},"labels":[],"label_agreement":null},{"id":"W2167742478","doi":"10.1017/s1351324911000210","title":"Exploring patterns in dictionary definitions for synonym extraction","year":2011,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":28,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Synonym (taxonomy); Natural language processing; Artificial intelligence; Lexicon; Quality (philosophy)","score_opus":0.07898671688241453,"score_gpt":0.262412322812195,"score_spread":0.18342560592978047,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2167742478","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17149584,0.0022649944,0.80573595,0.000957816,0.0002046421,0.000964707,0.00617021,0.0047111576,0.0074947183],"genre_scores_gemma":[0.33989075,0.0007448734,0.651971,0.00011062637,0.00005296311,0.0002647203,0.005856051,0.0002179523,0.00089106837],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99657464,0.0012657958,0.00065408606,0.00076917006,0.0006452357,0.00009114416],"domain_scores_gemma":[0.991148,0.005068125,0.0011694964,0.0010528294,0.0013550505,0.00020648392],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014325291,0.0007350294,0.0008512812,0.006610184,0.00065042306,0.0019610436,0.0012294223,0.00075102755,0.0037047141],"category_scores_gemma":[0.012591949,0.00036080924,0.0006039606,0.0060331165,0.00062089495,0.0037805948,0.0015385125,0.0008603398,0.0020292404],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035522858,0.00038500156,0.020545712,0.002177106,0.00023360724,0.00096183456,0.0013020609,0.0036922437,0.05378551,0.011541228,0.010354377,0.89466614],"study_design_scores_gemma":[0.0004227407,0.0011157221,0.054256026,0.001331186,0.0006685416,0.010483638,0.007544276,0.49331415,0.18670344,0.08120048,0.16261695,0.00034281076],"about_ca_topic_score_codex":0.00065900356,"about_ca_topic_score_gemma":0.0015901482,"teacher_disagreement_score":0.006610184,"about_ca_system_score_codex":0.00037710817,"about_ca_system_score_gemma":0.0010452305,"threshold_uncertainty_score":0.012393534},"labels":[],"label_agreement":null},{"id":"W2167907745","doi":"10.1109/rcis.2011.6006865","title":"Extracting the multilingual terminology from a web-based encyclopedia","year":2011,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Computer science; Terminology; Encyclopedia; Rank (graph theory); Natural language processing; Information retrieval; Exploit; Artificial intelligence; Term (time); Filter (signal processing); World Wide Web; Linguistics; Library science","score_opus":0.030854006936886218,"score_gpt":0.2778228173745137,"score_spread":0.24696881043762747,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2167907745","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.31007713,0.011575597,0.5241128,0.0020476084,0.0009311048,0.002254553,0.073675625,0.009564707,0.06576091],"genre_scores_gemma":[0.2513137,0.004984304,0.6400761,0.00026919757,0.00026857638,0.0010205434,0.09269117,0.0013290241,0.008047447],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99916553,0.00012021161,0.00022973494,0.00023249153,0.00018620459,0.00006590288],"domain_scores_gemma":[0.99809176,0.00065989775,0.00021081774,0.00019384948,0.0007474676,0.00009619272],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007024066,0.0013181431,0.0010541365,0.016825711,0.0010858048,0.0025424408,0.0008071228,0.00053983094,0.0059229275],"category_scores_gemma":[0.00477401,0.00044613905,0.0010447049,0.012777664,0.0005060401,0.0031636427,0.0018071919,0.00077895145,0.0051104454],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000358234,0.0002325216,0.011621885,0.006815078,0.00032999704,0.0045078243,0.003472178,0.0028130799,0.1568483,0.021356907,0.028987218,0.76265675],"study_design_scores_gemma":[0.00016215336,0.0008540282,0.0668779,0.0024134906,0.0016668871,0.011009872,0.0147530725,0.039239973,0.13902967,0.02293279,0.70058477,0.00047547068],"about_ca_topic_score_codex":0.0031401883,"about_ca_topic_score_gemma":0.005761028,"teacher_disagreement_score":0.016825711,"about_ca_system_score_codex":0.00073186704,"about_ca_system_score_gemma":0.0024344688,"threshold_uncertainty_score":0.019814134},"labels":[],"label_agreement":null},{"id":"W2167925143","doi":"10.3115/1118905.1118925","title":"Aligning and using an English-Inuktitut parallel corpus","year":2003,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":32,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Agglutinative language; Computer science; Sentence; Substring; Natural language processing; Artificial intelligence; Word (group theory); Speech recognition; Morpheme; Linguistics; Programming language","score_opus":0.023714388834000063,"score_gpt":0.27661286063885243,"score_spread":0.25289847180485236,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2167925143","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.84202605,0.0015915646,0.10895088,0.00071652513,0.00033145153,0.0018459938,0.018391976,0.0021351236,0.024010386],"genre_scores_gemma":[0.68331605,0.0006705402,0.2551648,0.00020657567,0.0000988712,0.0017135161,0.04884778,0.000661085,0.009320806],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9985862,0.00042263398,0.00019849371,0.0004792031,0.00023090189,0.00008262368],"domain_scores_gemma":[0.9958769,0.0018328903,0.00026731455,0.00040660898,0.0015128449,0.00010344917],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010138443,0.0006323915,0.0005216303,0.0029368151,0.0025814371,0.001501504,0.0008169833,0.00054358516,0.004332794],"category_scores_gemma":[0.005801509,0.00043126298,0.00035442555,0.0036369609,0.0007501385,0.0012045514,0.0013592708,0.0006985199,0.0015650346],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015573517,0.0011072119,0.030671926,0.0049717883,0.00027518792,0.008647174,0.029487072,0.013922899,0.27333832,0.014827162,0.03265629,0.58853775],"study_design_scores_gemma":[0.000364618,0.00060720806,0.1402751,0.0005662328,0.0007854402,0.006397836,0.030291962,0.08451835,0.2442608,0.006625708,0.48493797,0.00036878392],"about_ca_topic_score_codex":0.028572135,"about_ca_topic_score_gemma":0.058432195,"teacher_disagreement_score":0.028572135,"about_ca_system_score_codex":0.0014282253,"about_ca_system_score_gemma":0.0019863937,"threshold_uncertainty_score":0.05681169},"labels":[],"label_agreement":null},{"id":"W2168129207","doi":"10.1145/1390749.1390754","title":"Successfully detecting and correcting false friends using channel profiles","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Trigram; Security token; Computer science; Word (group theory); Set (abstract data type); Channel (broadcasting); Error detection and correction; Artificial intelligence; Point (geometry); Natural language processing; Speech recognition; Process (computing); Pattern recognition (psychology); Algorithm; Mathematics","score_opus":0.027108173006178155,"score_gpt":0.2719335257534153,"score_spread":0.24482535274723713,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2168129207","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3987558,0.00052128464,0.571126,0.00041421648,0.00020693432,0.00013661559,0.00081891223,0.025388833,0.0026314494],"genre_scores_gemma":[0.76908433,0.00020251366,0.22524887,0.00007654708,0.00008612102,0.00005600342,0.0010586629,0.0010597396,0.003127398],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99574775,0.000988672,0.00035249413,0.0008710667,0.0016693128,0.00037071278],"domain_scores_gemma":[0.9689085,0.014177511,0.0044108685,0.007258297,0.0045515792,0.00069326756],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021920353,0.0010574633,0.0012475977,0.0034605595,0.0010673983,0.0016940138,0.00141305,0.0014061407,0.0013723504],"category_scores_gemma":[0.027336467,0.0005180652,0.00041209257,0.0018939504,0.000659203,0.0036568951,0.001619999,0.001064311,0.0023985044],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012833974,0.0002107628,0.046433363,0.00037442148,0.00012274612,0.0014189315,0.0019766532,0.018427385,0.080542974,0.004355911,0.009577281,0.8352761],"study_design_scores_gemma":[0.000053735526,0.00025783756,0.018361289,0.00009017186,0.0001419271,0.0035629547,0.0013315707,0.5957497,0.35360372,0.0133911865,0.013163207,0.00029261888],"about_ca_topic_score_codex":0.0022435063,"about_ca_topic_score_gemma":0.0033607685,"teacher_disagreement_score":0.0034605595,"about_ca_system_score_codex":0.00041710222,"about_ca_system_score_gemma":0.0011093393,"threshold_uncertainty_score":0.011592746},"labels":[],"label_agreement":null},{"id":"W2168138178","doi":"10.5296/ijl.v3i1.648","title":"Machine Learning for Automatic Labeling of Frames and Frame Elements in Text","year":2011,"lang":"en","type":"article","venue":"International Journal of Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"FrameNet; Computer science; Frame (networking); Task (project management); Natural language processing; Artificial intelligence; Representation (politics); Semantics (computer science); Programming language; Engineering; Parsing","score_opus":0.023054528345659166,"score_gpt":0.3157570860943017,"score_spread":0.2927025577486425,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2168138178","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0108187115,0.0005802014,0.9709608,0.00045375442,0.00012865088,0.00030826504,0.0018673975,0.012345086,0.0025372682],"genre_scores_gemma":[0.05566559,0.00032852832,0.9352149,0.00012109631,0.00007264909,0.00031372273,0.0057647577,0.00039623282,0.002122647],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9974788,0.0009863048,0.0002302485,0.00076008495,0.0004021355,0.00014251111],"domain_scores_gemma":[0.9952081,0.0023351125,0.00054746616,0.00086658786,0.00093395956,0.00010872852],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030808612,0.0012341617,0.00077732455,0.0044966293,0.0018497866,0.0015326594,0.0017811388,0.0016987507,0.006085362],"category_scores_gemma":[0.0101798,0.0007314982,0.0011760136,0.003074462,0.00094289175,0.004296607,0.0013869619,0.0019632184,0.006048809],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036557458,0.00023804007,0.002145092,0.00068801874,0.00007806399,0.00016147277,0.0007419852,0.0077853813,0.04147715,0.021525279,0.042337824,0.88245606],"study_design_scores_gemma":[0.00014851503,0.00024107749,0.004284251,0.00038712303,0.00013279667,0.00043445706,0.0007331218,0.7085612,0.11012695,0.08681014,0.08800414,0.00013627374],"about_ca_topic_score_codex":0.007940668,"about_ca_topic_score_gemma":0.014283591,"teacher_disagreement_score":0.007940668,"about_ca_system_score_codex":0.0021041057,"about_ca_system_score_gemma":0.00226739,"threshold_uncertainty_score":0.02035755},"labels":[],"label_agreement":null},{"id":"W2168148476","doi":"10.1017/s0022226713000108","title":"Massimo Piattelli-Palmarini, Juan Uriagereka &amp; Pello Salaburu (eds.), Of minds and language: A dialogue with Noam Chomsky in the Basque Country. Oxford: Oxford University Press, 2009. Pp. x+458.","year":2013,"lang":"en","type":"article","venue":"Journal of Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Humanities; Philosophy; Classics; Theology; Art","score_opus":0.008310177040640979,"score_gpt":0.22908536133434063,"score_spread":0.22077518429369963,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2168148476","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0006095152,0.94965184,0.0020923843,0.023977349,0.0048690652,0.000012599901,0.00011848376,0.0000968888,0.018571949],"genre_scores_gemma":[0.010266674,0.91274434,0.005339445,0.004055345,0.009588644,0.000053915886,0.00023259998,0.00017189853,0.05754701],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9994978,0.00015585725,0.000021108159,0.0001294643,0.00014500061,0.000050774866],"domain_scores_gemma":[0.99873275,0.0007175964,0.000107654545,0.000046806526,0.00019595564,0.00019928179],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013695152,0.0021602407,0.0018698386,0.0025615008,0.001917669,0.004183712,0.0013160914,0.0026196053,0.035036802],"category_scores_gemma":[0.0016207573,0.0010999715,0.0006133748,0.002686816,0.0017692063,0.009459191,0.0015619098,0.0040594824,0.015533246],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012788075,0.0000409768,0.00045899386,0.0013944653,0.000024295066,0.00023818787,0.0021909962,0.00025252104,0.00048604686,0.013073725,0.7480662,0.23364568],"study_design_scores_gemma":[0.000009722218,0.000015421789,0.001829215,0.00068045297,0.00002290327,0.000746062,0.0012304395,0.00016609547,0.00021756621,0.007586565,0.9874697,0.000025969244],"about_ca_topic_score_codex":0.010224577,"about_ca_topic_score_gemma":0.013499992,"teacher_disagreement_score":0.035036802,"about_ca_system_score_codex":0.0021647112,"about_ca_system_score_gemma":0.0018729111,"threshold_uncertainty_score":0.11720979},"labels":[],"label_agreement":null},{"id":"W2168341766","doi":"","title":"SemEval-2 Task 9: The Interpretation of Noun Compounds Using Paraphrasing Verbs and Prepositions","year":2010,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":39,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"SemEval; Computer science; Task (project management); Noun; Natural language processing; Verb; Artificial intelligence; Interpretation (philosophy); Meaning (existential); Linguistics; Theme (computing); Noun phrase; Quality (philosophy); Philosophy; World Wide Web; Programming language","score_opus":0.008367889668326712,"score_gpt":0.2782778270113288,"score_spread":0.2699099373430021,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2168341766","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.45206696,0.004773842,0.2723465,0.005060937,0.002081322,0.0031592222,0.11959015,0.09470966,0.04621137],"genre_scores_gemma":[0.44599137,0.00059734745,0.35629025,0.001392288,0.00025865025,0.0011013285,0.17764801,0.0042057256,0.01251502],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99023914,0.0043405616,0.00096443587,0.0024580287,0.0014800765,0.0005177947],"domain_scores_gemma":[0.9741132,0.015191761,0.0010709012,0.005946587,0.003053096,0.00062445726],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0072765425,0.0031038225,0.0025084657,0.0024830042,0.0017844118,0.00405213,0.0049753767,0.0047063865,0.01584226],"category_scores_gemma":[0.033471864,0.0010007985,0.0022246004,0.0022457412,0.0014217584,0.009597035,0.0054394505,0.003593128,0.009468431],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.003657325,0.002237415,0.011799336,0.006344656,0.0005098539,0.0022914296,0.004827046,0.015747776,0.03661461,0.012858403,0.34965846,0.5534536],"study_design_scores_gemma":[0.0020105995,0.0018826474,0.027033128,0.0010398602,0.00047457407,0.008763686,0.012503618,0.39357635,0.13872485,0.05346532,0.35998356,0.000541793],"about_ca_topic_score_codex":0.0061196918,"about_ca_topic_score_gemma":0.008768447,"teacher_disagreement_score":0.01584226,"about_ca_system_score_codex":0.0016517132,"about_ca_system_score_gemma":0.0022938573,"threshold_uncertainty_score":0.05299765},"labels":[],"label_agreement":null},{"id":"W2168727904","doi":"10.1111/0824-7935.00107","title":"Integrating Web‐Based Documents, Shared Knowledge Bases, and Information Retrieval for User Help","year":2000,"lang":"en","type":"article","venue":"Computational Intelligence","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Information retrieval; Paragraph; Terminology; World Wide Web; Sentence; Knowledge base; Search engine; Frame (networking); Artificial intelligence","score_opus":0.018403862805794646,"score_gpt":0.30848898935031954,"score_spread":0.2900851265445249,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2168727904","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06303732,0.0011967436,0.89194745,0.0010910222,0.00011666316,0.0007820877,0.00024717877,0.021095676,0.02048577],"genre_scores_gemma":[0.18866391,0.00044583372,0.80088645,0.00027440416,0.000067800844,0.00043149848,0.0006677169,0.00068105885,0.007881261],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9974445,0.0010024203,0.0002122494,0.00029776,0.00082104316,0.00022207364],"domain_scores_gemma":[0.99444026,0.0027586045,0.00029672758,0.0015212074,0.00067781785,0.00030538125],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005266942,0.0005598694,0.0010856723,0.0032931312,0.0012841846,0.0041573374,0.0018564155,0.0012073456,0.00555083],"category_scores_gemma":[0.009406989,0.00074663677,0.0006799997,0.0035018742,0.0014696607,0.007569733,0.0028503933,0.0013895048,0.0028632283],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00081152946,0.0012249949,0.0030083202,0.0009473618,0.00017320686,0.0006284134,0.0047904537,0.0070408867,0.023794077,0.06803857,0.015337846,0.87420434],"study_design_scores_gemma":[0.0015443112,0.0032171106,0.0089046275,0.0008418772,0.0013012895,0.0028581803,0.004290162,0.18399721,0.13419738,0.2642085,0.39380652,0.0008328431],"about_ca_topic_score_codex":0.004185399,"about_ca_topic_score_gemma":0.005470481,"teacher_disagreement_score":0.00555083,"about_ca_system_score_codex":0.001111611,"about_ca_system_score_gemma":0.002316186,"threshold_uncertainty_score":0.027854621},"labels":[],"label_agreement":null},{"id":"W2168801328","doi":"10.1162/089120103322711587","title":"Embedding Web-Based Statistical Translation Models in Cross-Language Information Retrieval","year":2003,"lang":"en","type":"article","venue":"Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":106,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; Université du Québec à Montréal","funders":"","keywords":"Computer science; Cross-language information retrieval; Machine translation; Natural language processing; Information retrieval; Artificial intelligence; Process (computing); Translation (biology); Machine translation software usability; Language model; Embedding; Example-based machine translation; Programming language","score_opus":0.019591668201522305,"score_gpt":0.3257819259314265,"score_spread":0.3061902577299042,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2168801328","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029683745,0.0006860976,0.9654546,0.00050115946,0.00005434126,0.00007006529,0.00015102152,0.001362248,0.0020367706],"genre_scores_gemma":[0.59846324,0.0013637204,0.39190063,0.0003444765,0.00017761273,0.00036100103,0.0011345301,0.0005012973,0.005753377],"study_design_codex":"simulation_or_modeling","study_design_gemma":"not_applicable","domain_scores_codex":[0.9977093,0.0015457375,0.00012446423,0.0002656126,0.00027061172,0.00008442893],"domain_scores_gemma":[0.9948868,0.0035503593,0.0003219531,0.00062135246,0.00055888935,0.0000606088],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002948029,0.0006858851,0.0009962313,0.0020624246,0.00061531935,0.002010587,0.0008486329,0.0015922906,0.0028926025],"category_scores_gemma":[0.012559234,0.00082487665,0.00107615,0.002539771,0.0010933888,0.004334592,0.0014236307,0.0012577723,0.0024793325],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00027245085,0.00024929666,0.0014390044,0.00018700362,0.00015257842,0.00023264941,0.00033341144,0.67946255,0.003470942,0.06309647,0.0029440103,0.24815963],"study_design_scores_gemma":[0.000013765319,0.000026642732,0.000121970064,0.000008505562,0.0000121794965,0.000029343671,0.000024737035,0.9709681,0.0005532609,0.027362023,0.00086719304,0.00001224516],"about_ca_topic_score_codex":0.007951599,"about_ca_topic_score_gemma":0.007478223,"teacher_disagreement_score":0.007951599,"about_ca_system_score_codex":0.0010884337,"about_ca_system_score_gemma":0.0012224395,"threshold_uncertainty_score":0.015810609},"labels":[],"label_agreement":null},{"id":"W2169218969","doi":"10.18653/v1/d13-1110","title":"Efficient Left-to-Right Hierarchical Phrase-Based Translation with Improved Reordering","year":2013,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Decoding methods; Phrase; Computer science; Machine translation; Translation (biology); Algorithm; Sentence; Speech recognition; Artificial intelligence; Natural language processing","score_opus":0.006980057795418121,"score_gpt":0.23002735962996146,"score_spread":0.22304730183454333,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2169218969","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019296274,0.00021637346,0.97025573,0.00016034994,0.00006568785,0.000109702414,0.0004891442,0.006333245,0.0030735454],"genre_scores_gemma":[0.16257274,0.00017462023,0.82865125,0.00018487421,0.00006898573,0.00013119339,0.0018792844,0.0015830535,0.0047540204],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99895275,0.0003633996,0.00009614549,0.00023860845,0.00027045462,0.000078633784],"domain_scores_gemma":[0.9977552,0.0008432395,0.00014236705,0.000666582,0.00053916266,0.00005347896],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00085932505,0.0010765006,0.0010363172,0.0007169607,0.0004930238,0.00096623955,0.0009691609,0.0008125578,0.0052217157],"category_scores_gemma":[0.0035568886,0.00045886612,0.0006560154,0.0010459457,0.0005093344,0.0015862545,0.0012375024,0.0013027523,0.005181802],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005108801,0.00025189278,0.0011978311,0.0004636151,0.00007608778,0.00042074698,0.0005327157,0.05883928,0.1300255,0.016168034,0.01857667,0.7729367],"study_design_scores_gemma":[0.00020788806,0.00029830635,0.0010439391,0.00003123719,0.000082218154,0.00059091195,0.00015897781,0.8490834,0.11411209,0.018740121,0.015561937,0.000089052264],"about_ca_topic_score_codex":0.0027881863,"about_ca_topic_score_gemma":0.00612929,"teacher_disagreement_score":0.0052217157,"about_ca_system_score_codex":0.0004261486,"about_ca_system_score_gemma":0.0013946082,"threshold_uncertainty_score":0.017468393},"labels":[],"label_agreement":null},{"id":"W2169256506","doi":"","title":"코퍼스기반 번역학 연구에서 정량적 인자가 정성적 분석 결과에 미치는 영향","year":2013,"lang":"ko","type":"article","venue":"번역학연구","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Translation (biology); Linguistics; Noun phrase; Corpus linguistics; Quarter (Canadian coin); Natural language processing; Noun; Zero (linguistics); Artificial intelligence; Computer science; History; Philosophy; Biology","score_opus":0.010726173817322712,"score_gpt":0.2582661328891293,"score_spread":0.2475399590718066,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2169256506","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8061328,0.0017027615,0.14190947,0.0026825743,0.0003172706,0.0023294066,0.003279559,0.00021698722,0.04142923],"genre_scores_gemma":[0.86957145,0.0006789446,0.11784748,0.00038420557,0.00011879805,0.005909575,0.001980316,0.00012558803,0.0033835738],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9933035,0.0042688493,0.0005143647,0.000757502,0.0010015465,0.00015421849],"domain_scores_gemma":[0.962005,0.029131584,0.0016301738,0.0028825984,0.0040999786,0.0002506601],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009473288,0.00025402338,0.0005146882,0.0021676095,0.0016225638,0.002051538,0.00085108454,0.00041571326,0.0056214915],"category_scores_gemma":[0.03620376,0.0003705803,0.00033900174,0.002308322,0.0019889162,0.0028308197,0.0013037956,0.00042936398,0.00087173603],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013300154,0.0009821927,0.18676296,0.0056735845,0.00030755345,0.0017774606,0.112787426,0.003920555,0.09723588,0.08137249,0.012728076,0.49512187],"study_design_scores_gemma":[0.00028619295,0.0026934166,0.44191116,0.0022074126,0.0007405196,0.0025628076,0.18433407,0.021846356,0.083500184,0.050274603,0.20930752,0.00033571772],"about_ca_topic_score_codex":0.0018923872,"about_ca_topic_score_gemma":0.0040028514,"teacher_disagreement_score":0.009473288,"about_ca_system_score_codex":0.0009995817,"about_ca_system_score_gemma":0.0014991389,"threshold_uncertainty_score":0.050100148},"labels":[],"label_agreement":null},{"id":"W2169503695","doi":"10.1111/0824-7935.00118","title":"Choosing Rhetorical Structures To Plan Instructional Texts","year":2000,"lang":"en","type":"article","venue":"Computational Intelligence","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":43,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Rhetorical question; Computer science; Heuristics; Natural language processing; Natural language generation; Artificial intelligence; Natural language; Set (abstract data type); Natural (archaeology); Linguistics; Natural language understanding; Programming language","score_opus":0.029826901023934045,"score_gpt":0.31508870331775013,"score_spread":0.2852618022938161,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2169503695","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12374523,0.00059129944,0.8451614,0.0018928802,0.00010410274,0.0012652507,0.00042560176,0.0030209895,0.023793358],"genre_scores_gemma":[0.33860746,0.00031816523,0.6570053,0.00022938185,0.000042068867,0.0005110834,0.00061646244,0.00044880903,0.0022212192],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9965025,0.0022360317,0.0001917086,0.00043080442,0.00048062453,0.00015832539],"domain_scores_gemma":[0.9886645,0.008260388,0.00092369964,0.0007843256,0.0010233623,0.0003437906],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004221052,0.00087820645,0.0005151829,0.0024435427,0.0012748435,0.0030553567,0.0012823269,0.0010986793,0.0047401534],"category_scores_gemma":[0.025596224,0.00068923447,0.0005808915,0.0010324371,0.0022714452,0.004698642,0.0015405297,0.0014880884,0.0013855841],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00065624295,0.0005009006,0.0071305935,0.0015644293,0.00009874668,0.0009405205,0.018119356,0.04453333,0.035642955,0.41013968,0.012095708,0.46857753],"study_design_scores_gemma":[0.0005932039,0.00066258485,0.004318464,0.00090676703,0.00023571154,0.0005824211,0.012250324,0.3057097,0.057646662,0.51746684,0.09937781,0.00024953412],"about_ca_topic_score_codex":0.0011360315,"about_ca_topic_score_gemma":0.0030013565,"teacher_disagreement_score":0.0047401534,"about_ca_system_score_codex":0.0014927875,"about_ca_system_score_gemma":0.002316942,"threshold_uncertainty_score":0.02232331},"labels":[],"label_agreement":null},{"id":"W2169526216","doi":"10.1109/slt.2010.5700850","title":"An efficient approach for two-stage open vocabulary spoken term detection","year":2010,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Vysoké Učení Technické v Brně; McGill University","keywords":"Search engine indexing; Computer science; Vocabulary; Term (time); Word (group theory); Speech recognition; Domain (mathematical analysis); Natural language processing; Artificial intelligence; Task (project management); Index (typography); Database index; Information retrieval; World Wide Web","score_opus":0.015540315263242297,"score_gpt":0.3104132214472516,"score_spread":0.2948729061840093,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2169526216","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01842389,0.00032898903,0.9722812,0.000076599725,0.00008506614,0.00028621496,0.0005590211,0.006291657,0.001667271],"genre_scores_gemma":[0.07952086,0.00014496667,0.9138716,0.000054878816,0.000076519515,0.00032242597,0.0020690628,0.0003050564,0.0036347671],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9982186,0.00019099076,0.00016627464,0.0003343911,0.00094798824,0.00014186218],"domain_scores_gemma":[0.99698263,0.001019643,0.00016640296,0.0006352991,0.0010113184,0.00018467252],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00096591975,0.0009971794,0.001476776,0.0031775604,0.000812619,0.0020894702,0.0020457513,0.0008586105,0.007231719],"category_scores_gemma":[0.0050168894,0.00046624415,0.00077118445,0.0024908087,0.00045277315,0.0028688277,0.0020689494,0.0010801717,0.0072945766],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00047012337,0.00024583796,0.0011425737,0.00021730906,0.000052123494,0.00011127507,0.00025369736,0.0021108387,0.12105358,0.004504795,0.003817368,0.8660206],"study_design_scores_gemma":[0.00033563224,0.0012108502,0.006004685,0.00006045775,0.00019828646,0.002239727,0.00076639943,0.6286523,0.29208547,0.024955386,0.043186806,0.00030408375],"about_ca_topic_score_codex":0.002292712,"about_ca_topic_score_gemma":0.004324598,"teacher_disagreement_score":0.007231719,"about_ca_system_score_codex":0.000507966,"about_ca_system_score_gemma":0.0018546468,"threshold_uncertainty_score":0.024192512},"labels":[],"label_agreement":null},{"id":"W216953234","doi":"10.1353/lan.2000.0128","title":"Towards a 'natural' narratology ByMonika Fludernik (review)","year":2000,"lang":"en","type":"article","venue":"Language","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Parsing; Computer science; Artificial intelligence; Bottom-up parsing; Natural language processing; Top-down parsing; Top-down parsing language; S-attributed grammar; Parser combinator; Programming language; Dependency grammar; Memoization; Sentence; Grammar; Natural language; Linguistics","score_opus":0.0061293855597686344,"score_gpt":0.27202795288163695,"score_spread":0.26589856732186834,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W216953234","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.000026661353,0.99462986,0.000057673875,0.0026586426,0.0014498106,0.0000036152414,0.000020267265,0.000006765855,0.0011467157],"genre_scores_gemma":[0.00044187045,0.9905282,0.000119838354,0.0043146005,0.0020763506,0.000017897348,0.00008356656,0.000006713535,0.002410914],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9996749,0.000071649316,0.000049614697,0.000065941895,0.00010356707,0.000034353794],"domain_scores_gemma":[0.99933773,0.00029468205,0.00009054491,0.000018380077,0.00017155243,0.00008715304],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00085275393,0.00081007346,0.0013129658,0.0025523494,0.00050726626,0.0023627686,0.0011207691,0.0017777687,0.015541735],"category_scores_gemma":[0.0017710836,0.00037960563,0.0004556057,0.002444219,0.0010723982,0.004336681,0.0016230951,0.0039129793,0.009475177],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000043158492,0.000027214293,0.00010147224,0.01107203,0.000037841975,0.0001339053,0.0001867642,0.0000647178,0.0003459886,0.005145768,0.5264643,0.4563768],"study_design_scores_gemma":[0.000006429854,0.00001097279,0.00024665,0.0028421422,0.000014377,0.00039327107,0.000052380463,0.000006488181,0.00003235825,0.0007325879,0.99565566,0.0000066645875],"about_ca_topic_score_codex":0.0019742937,"about_ca_topic_score_gemma":0.0035442796,"teacher_disagreement_score":0.015541735,"about_ca_system_score_codex":0.0013943837,"about_ca_system_score_gemma":0.0017575141,"threshold_uncertainty_score":0.051992297},"labels":[],"label_agreement":null},{"id":"W2169852957","doi":"","title":"Summarization Techniques at DUC 2004","year":2004,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Lethbridge","funders":"","keywords":"Automatic summarization; Byte; Computer science; Context (archaeology); Artificial intelligence; Programming language; History","score_opus":0.007476894249547869,"score_gpt":0.24941034655666142,"score_spread":0.24193345230711355,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2169852957","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05175497,0.011669648,0.7217893,0.003310866,0.0022952429,0.0017812048,0.033417895,0.13525105,0.03872977],"genre_scores_gemma":[0.096926644,0.0022140369,0.769298,0.0005773097,0.0005345621,0.0010978099,0.07934891,0.0045732176,0.045429535],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9968215,0.0009572437,0.00030272172,0.00051797647,0.0011894814,0.00021098743],"domain_scores_gemma":[0.99531037,0.0007367235,0.00016632628,0.00067804905,0.0029276428,0.0001808593],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030442856,0.0014452589,0.0012808553,0.0037383887,0.0015797685,0.0023132688,0.0015448917,0.0012003062,0.009148056],"category_scores_gemma":[0.0067592817,0.0006144058,0.00060669816,0.0024625794,0.0003560506,0.0018469134,0.00076955033,0.00196348,0.005122779],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00054099446,0.000267696,0.0005897369,0.0006825447,0.00015105322,0.00035276075,0.00051023194,0.008584501,0.03816271,0.005304553,0.30166295,0.64319026],"study_design_scores_gemma":[0.00048092954,0.001124168,0.0042989873,0.00015589742,0.00026894486,0.0008939775,0.00049977604,0.13279177,0.15115601,0.009749283,0.69832253,0.00025768304],"about_ca_topic_score_codex":0.015228171,"about_ca_topic_score_gemma":0.022547973,"teacher_disagreement_score":0.015228171,"about_ca_system_score_codex":0.0017074993,"about_ca_system_score_gemma":0.0013427805,"threshold_uncertainty_score":0.03060329},"labels":[],"label_agreement":null},{"id":"W2169943035","doi":"10.48550/arxiv.1302.4813","title":"Probabilistic Frame Induction","year":2013,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":87,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Merge (version control); Parsing; Probabilistic logic; Artificial intelligence; Frame (networking); Natural language processing; Natural language; Event (particle physics); Set (abstract data type); Information retrieval; Programming language","score_opus":0.038696906805177664,"score_gpt":0.17766298116822438,"score_spread":0.1389660743630467,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2169943035","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005506608,0.00024994052,0.9879475,0.00027451146,0.00007965316,0.00021470472,0.0010476168,0.0022388517,0.0024406523],"genre_scores_gemma":[0.17308225,0.0004420022,0.8085427,0.00033251976,0.0003018764,0.0009727206,0.009280557,0.00066136767,0.0063839406],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99738115,0.00080511475,0.00014171086,0.00097416516,0.00048607658,0.00021180906],"domain_scores_gemma":[0.99405164,0.0042108316,0.0003049067,0.00053698156,0.000769763,0.00012587194],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023062837,0.0015840431,0.0010674893,0.0029666515,0.0015557797,0.0013443866,0.0032561556,0.0015938933,0.009481852],"category_scores_gemma":[0.008812217,0.0009070634,0.002728102,0.0024337464,0.0012249699,0.0033246235,0.0027419275,0.002670155,0.0037358317],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005834987,0.00025562622,0.003639033,0.00079715194,0.00019592048,0.00074914744,0.0012490425,0.063302785,0.010705876,0.12933317,0.03831448,0.7508743],"study_design_scores_gemma":[0.00010148767,0.00007979454,0.00124553,0.000120887715,0.00013605675,0.00037077043,0.00021860626,0.77333117,0.012487615,0.18953155,0.02231619,0.000060245922],"about_ca_topic_score_codex":0.0035502682,"about_ca_topic_score_gemma":0.0061976463,"teacher_disagreement_score":0.009481852,"about_ca_system_score_codex":0.0016048729,"about_ca_system_score_gemma":0.0019853387,"threshold_uncertainty_score":0.031719923},"labels":[],"label_agreement":null},{"id":"W2170270347","doi":"","title":"Lexically-Triggered Hidden Markov Models for Clinical Document Coding","year":2011,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Hidden Markov model; Discriminative model; Computer science; Coding (social sciences); Artificial intelligence; Natural language processing; Phrase; Set (abstract data type); Speech recognition; Mathematics","score_opus":0.08503920427112521,"score_gpt":0.34829158072323213,"score_spread":0.2632523764521069,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2170270347","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.034705456,0.0007301902,0.95999783,0.00062893046,0.00012285444,0.000108858265,0.0009166977,0.0016323844,0.001156919],"genre_scores_gemma":[0.75369656,0.0006625129,0.23788792,0.00032292455,0.00018722193,0.00048067098,0.0025447602,0.00018623145,0.004031214],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99869317,0.0006735732,0.000107635395,0.00023542714,0.00019647257,0.000093808136],"domain_scores_gemma":[0.99047893,0.007994813,0.0004686025,0.00049029564,0.00046261825,0.000104663304],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021714536,0.0006156464,0.000697662,0.0010033757,0.00041333676,0.0008833745,0.0016338619,0.0012557165,0.0023625595],"category_scores_gemma":[0.011450706,0.00045201715,0.00083890854,0.0011224663,0.0005766923,0.0015149378,0.00088405824,0.0017952141,0.0010804438],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00067048933,0.00020979425,0.0045024464,0.00023421433,0.00011990926,0.00045776126,0.0003782036,0.7363187,0.0056331432,0.038871523,0.0051409644,0.20746288],"study_design_scores_gemma":[0.000012226017,0.000014958587,0.000190433,0.0000065552836,0.000006649046,0.000023271192,0.00000655998,0.9868622,0.0004543853,0.012145454,0.00026795876,0.000009362103],"about_ca_topic_score_codex":0.0070142923,"about_ca_topic_score_gemma":0.0091524655,"teacher_disagreement_score":0.0070142923,"about_ca_system_score_codex":0.0011262585,"about_ca_system_score_gemma":0.0010835237,"threshold_uncertainty_score":0.01394695},"labels":[],"label_agreement":null},{"id":"W2170441435","doi":"10.22329/il.v23i3.2172","title":"Against the \"Ordinary Summing\" Test for Convergence","year":2004,"lang":"en","type":"article","venue":"Informal Logic","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Argument (complex analysis); Generalization; Convergence (economics); Test (biology); Mathematics; Epistemology; Applied mathematics; Philosophy","score_opus":0.018095908422299578,"score_gpt":0.2676428179057791,"score_spread":0.24954690948347952,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2170441435","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.111934535,0.0016014833,0.48691225,0.047626246,0.0015160884,0.0005112199,0.0008782617,0.0013721198,0.34764782],"genre_scores_gemma":[0.86984307,0.0005512208,0.1033956,0.011948473,0.0013637472,0.000946839,0.0006514114,0.00057713035,0.010722453],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.954043,0.024059812,0.0029361628,0.005511354,0.011570757,0.0018789228],"domain_scores_gemma":[0.8259529,0.12949717,0.005428159,0.014738671,0.02107997,0.0033031413],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03956764,0.0013189153,0.0029115153,0.0051740645,0.0050862804,0.0054938365,0.004016486,0.0053419815,0.017307408],"category_scores_gemma":[0.19025211,0.0005932067,0.0031567449,0.0032373478,0.0273167,0.022574889,0.011684378,0.0096817445,0.0035789316],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021962302,0.000040891424,0.0025937844,0.0001434928,0.000067264176,0.00013426336,0.00072687067,0.0006400663,0.0002924867,0.9688676,0.004805339,0.021468354],"study_design_scores_gemma":[0.00009718419,0.00006592412,0.000903659,0.000103912804,0.000030608862,0.00019780624,0.0005688972,0.0037606074,0.0007245715,0.9867877,0.006727106,0.000032026248],"about_ca_topic_score_codex":0.0013200645,"about_ca_topic_score_gemma":0.00085896126,"teacher_disagreement_score":0.03956764,"about_ca_system_score_codex":0.00230975,"about_ca_system_score_gemma":0.0037226123,"threshold_uncertainty_score":0.20925617},"labels":[],"label_agreement":null},{"id":"W2170471032","doi":"10.1162/089120103322753365","title":"Disambiguating Nouns, Verbs, and Adjectives Using Automatically Acquired Selectional Preferences","year":2003,"lang":"en","type":"article","venue":"Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":141,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Engineering and Physical Sciences Research Council; Atomic Energy of Canada Limited","keywords":"Computer science; Adjective; Noun; Natural language processing; Verb; Artificial intelligence; Word (group theory); Heuristic; Argument (complex analysis); Part of speech; Linguistics","score_opus":0.024330433789640626,"score_gpt":0.30231777277498023,"score_spread":0.2779873389853396,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2170471032","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8695164,0.0006923937,0.11766265,0.00026161212,0.00005557213,0.00027069298,0.0027101473,0.003195768,0.005634833],"genre_scores_gemma":[0.8423027,0.0002715034,0.14929533,0.00009192689,0.000027528875,0.00015083898,0.0065135225,0.0002892957,0.0010573],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99693227,0.00126989,0.00037254707,0.0006800415,0.0006304519,0.00011477616],"domain_scores_gemma":[0.98269373,0.011718534,0.0011201567,0.0017830541,0.0024325412,0.00025194045],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0038786654,0.000853889,0.0009596024,0.004175608,0.00097741,0.0019458754,0.000637814,0.0007250617,0.0009121499],"category_scores_gemma":[0.017518032,0.00041141175,0.0006133033,0.0032795942,0.000696438,0.0030135524,0.0016253724,0.00070784404,0.0008941206],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002014697,0.00063072995,0.093862854,0.0013649003,0.0005414154,0.0006306854,0.0047616283,0.043993376,0.116376735,0.0054104766,0.007701236,0.7227112],"study_design_scores_gemma":[0.0003190016,0.00081118895,0.09614705,0.00017753337,0.0005131208,0.0018705868,0.005853472,0.58939415,0.2513878,0.023155821,0.03000402,0.0003662433],"about_ca_topic_score_codex":0.0036587738,"about_ca_topic_score_gemma":0.007457117,"teacher_disagreement_score":0.004175608,"about_ca_system_score_codex":0.0005978033,"about_ca_system_score_gemma":0.0010463274,"threshold_uncertainty_score":0.02051258},"labels":[],"label_agreement":null},{"id":"W2170694014","doi":"10.1145/1316874.1316896","title":"Combining resources with confidence measures for cross language information retrieval","year":2007,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Cross-language information retrieval; Natural language processing; Artificial intelligence; Translation (biology); Arabic; Measure (data warehouse); Machine translation; Information retrieval; Confidence interval; Data mining; Statistics; Linguistics; Mathematics","score_opus":0.01099490196256783,"score_gpt":0.29088982205894826,"score_spread":0.27989492009638045,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2170694014","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029628076,0.0017635612,0.9643804,0.00021431386,0.00005888659,0.00015993911,0.00017382317,0.0019868303,0.0016341164],"genre_scores_gemma":[0.46797606,0.0006845973,0.5280448,0.00022324293,0.00027969712,0.000451526,0.0010892455,0.00045056394,0.00080025475],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.97207,0.012148927,0.0027571423,0.0025416159,0.009695072,0.00078727107],"domain_scores_gemma":[0.90538555,0.06614802,0.0057342383,0.011849221,0.010027769,0.00085512246],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.021869186,0.0017893839,0.0024476945,0.013518012,0.0011820102,0.0036896002,0.0031107247,0.0026164998,0.0019312294],"category_scores_gemma":[0.11413986,0.0013018945,0.0017054577,0.010565477,0.0020912287,0.012696884,0.0054323096,0.0032110722,0.0014397475],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00076919625,0.0006534512,0.012454856,0.0005341483,0.0005784814,0.00027673933,0.000418988,0.08879885,0.012592359,0.024463367,0.0037902696,0.8546692],"study_design_scores_gemma":[0.00012089572,0.00043140855,0.0052776234,0.000121371326,0.00029418888,0.000641238,0.00013584663,0.9111498,0.024839086,0.052814227,0.0038738651,0.00030048317],"about_ca_topic_score_codex":0.001706287,"about_ca_topic_score_gemma":0.001523761,"teacher_disagreement_score":0.021869186,"about_ca_system_score_codex":0.0013117153,"about_ca_system_score_gemma":0.0010672656,"threshold_uncertainty_score":0.115656674},"labels":[],"label_agreement":null},{"id":"W2170774542","doi":"10.3115/1614049.1614065","title":"Investigating cross-language speech retrieval for a spontaneous conversational speech collection","year":2006,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Clef; Natural language processing; Machine translation; Artificial intelligence; Speech recognition; Transcription (linguistics); Speech translation; Language model; Language translation; Speech corpus; Metadata; Task (project management); Information retrieval; Speech synthesis; World Wide Web; Linguistics","score_opus":0.011763319945289346,"score_gpt":0.2831661493475073,"score_spread":0.27140282940221794,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2170774542","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.89133483,0.0009006162,0.10079284,0.00040930894,0.000099406556,0.0008087297,0.0014054158,0.0019148226,0.0023340445],"genre_scores_gemma":[0.87733346,0.0004934927,0.10810781,0.0002528741,0.0002184031,0.0010617672,0.009101784,0.00041136082,0.0030190593],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9873169,0.008037499,0.00095533486,0.0015113712,0.0017279622,0.00045093056],"domain_scores_gemma":[0.9548992,0.032996554,0.0015405705,0.00411001,0.005787697,0.00066589593],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012416982,0.0011609874,0.0012387572,0.0023832722,0.001747132,0.0023929693,0.0012778705,0.0016130282,0.0017538004],"category_scores_gemma":[0.035265725,0.0004887161,0.0012412451,0.0023051505,0.0012282722,0.0039132414,0.0023270294,0.0011326693,0.0014248453],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0038524,0.0063155345,0.04644087,0.0063463324,0.0014028291,0.00444355,0.020600667,0.052146394,0.3559641,0.006900083,0.013710144,0.48187715],"study_design_scores_gemma":[0.0014070204,0.0075459615,0.092745386,0.00019763388,0.00092611427,0.0072107594,0.02119461,0.5397876,0.29409105,0.008848859,0.025501892,0.0005431877],"about_ca_topic_score_codex":0.005161944,"about_ca_topic_score_gemma":0.006046679,"teacher_disagreement_score":0.012416982,"about_ca_system_score_codex":0.0010532456,"about_ca_system_score_gemma":0.0011267462,"threshold_uncertainty_score":0.065668106},"labels":[],"label_agreement":null},{"id":"W2170921593","doi":"10.1109/fuzz.2002.1006661","title":"Natural language understanding through fuzzy logic inference and its application to speech recognition","year":2003,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Fuzzy logic; Artificial intelligence; Inference; Adaptive neuro fuzzy inference system; Natural language; Natural language processing; Rule of inference; Speech recognition; Neuro-fuzzy; Fuzzy control system","score_opus":0.049678600399321045,"score_gpt":0.3188503847888839,"score_spread":0.26917178438956285,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2170921593","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014053751,0.00049585896,0.9800828,0.00029619987,0.000029742,0.00007672022,0.000039697872,0.0006702895,0.00425487],"genre_scores_gemma":[0.25470743,0.0010452378,0.7411441,0.00017852621,0.00006576869,0.00014439881,0.0001522139,0.00007024067,0.0024920192],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9994912,0.00014350511,0.000039569575,0.00011317425,0.00017724979,0.000035259724],"domain_scores_gemma":[0.99904555,0.00061869365,0.00006441605,0.00007916753,0.0001729897,0.00001907528],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010930148,0.00040454915,0.0003448394,0.00083185313,0.00061258406,0.0013168168,0.0007764713,0.0008291802,0.0026351307],"category_scores_gemma":[0.0040392936,0.00024103966,0.00059882546,0.000679172,0.0010349725,0.0016449405,0.00050088536,0.0009307518,0.00058644393],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019862609,0.00022339843,0.0017475773,0.00036373225,0.00014411683,0.0006433145,0.0016880933,0.08148695,0.051996637,0.07199915,0.0027012518,0.7868071],"study_design_scores_gemma":[0.000038784998,0.00013490705,0.0016413827,0.00009121132,0.00010119549,0.00041772414,0.0002481117,0.83130646,0.040970556,0.11258507,0.012383806,0.00008082],"about_ca_topic_score_codex":0.0049657095,"about_ca_topic_score_gemma":0.0040895767,"teacher_disagreement_score":0.0049657095,"about_ca_system_score_codex":0.00049949053,"about_ca_system_score_gemma":0.00064389175,"threshold_uncertainty_score":0.009873629},"labels":[],"label_agreement":null},{"id":"W2171095728","doi":"10.1109/asru.1997.659106","title":"Techniques to achieve fast lexical access","year":2002,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Institut National de la Recherche Scientifique","funders":"","keywords":"Computer science; Lexicon; Language model; Pruning; Heuristic; Vocabulary; Artificial intelligence; Natural language processing; Task (project management); Overhead (engineering); Workstation; Word (group theory); Lexical database; Stack (abstract data type); Speech recognition; Programming language; Linguistics; WordNet","score_opus":0.02907606266598196,"score_gpt":0.3124485945903496,"score_spread":0.2833725319243677,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2171095728","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006078422,0.0008138153,0.9812992,0.00030556254,0.0001557776,0.00037687307,0.00012520114,0.00550692,0.005338335],"genre_scores_gemma":[0.06233121,0.0006011112,0.92837334,0.00032622516,0.00021415525,0.00078176556,0.0005476088,0.0006772951,0.006147295],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9978192,0.00055006903,0.00025557104,0.00032944125,0.0007350488,0.0003105727],"domain_scores_gemma":[0.9944812,0.00265997,0.00028033444,0.0013619032,0.0010680712,0.00014852641],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021466673,0.0017413326,0.0014799034,0.0036822958,0.0019090872,0.003193509,0.0028461164,0.0015770663,0.022549886],"category_scores_gemma":[0.0099707255,0.0012916396,0.0011665563,0.003427372,0.0012978412,0.007177548,0.005073031,0.0022793782,0.012010853],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00046814574,0.00026177132,0.0005240117,0.0006543057,0.000091060836,0.00040749385,0.0009323233,0.0035279193,0.0500359,0.07277572,0.018646237,0.8516752],"study_design_scores_gemma":[0.0011594743,0.0007312604,0.0012824789,0.0003416421,0.00034760014,0.0038204018,0.0011308709,0.22334094,0.10847373,0.43185785,0.22718461,0.00032924346],"about_ca_topic_score_codex":0.0011562543,"about_ca_topic_score_gemma":0.0023745776,"teacher_disagreement_score":0.022549886,"about_ca_system_score_codex":0.0005522973,"about_ca_system_score_gemma":0.001394421,"threshold_uncertainty_score":0.07543695},"labels":[],"label_agreement":null},{"id":"W2171160649","doi":"10.7202/009357ar","title":"A Methodological Proposal for the Study of Semantic Functions across Languages","year":2004,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Natural language processing; Linguistics; Artificial intelligence; Process (computing); Function (biology); Contrastive analysis; Programming language; Philosophy","score_opus":0.12105628847692165,"score_gpt":0.3931444725885981,"score_spread":0.27208818411167646,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2171160649","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.002995265,0.000643441,0.9728879,0.0055675167,0.00040831746,0.00066310004,0.00037134075,0.00019430042,0.016268862],"genre_scores_gemma":[0.068120785,0.00075886946,0.9192833,0.0014960807,0.00033754559,0.0052969265,0.00038779152,0.00017204435,0.0041465973],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9848317,0.009744312,0.0012222099,0.0026146031,0.0012120238,0.0003751464],"domain_scores_gemma":[0.98533326,0.008217784,0.0010490551,0.0029856428,0.0017944634,0.0006198233],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.028176408,0.0014322305,0.0017587577,0.010130435,0.0042401436,0.01083344,0.0050575864,0.0036109283,0.009497144],"category_scores_gemma":[0.026470285,0.0013184337,0.0026197894,0.008278105,0.022612924,0.018161379,0.006945374,0.0062208204,0.0020747574],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000010570667,0.000023373923,0.00029676224,0.0001024404,0.00001745125,0.000033421467,0.0024683583,0.00011347113,0.00034215781,0.9882379,0.0005740944,0.007780052],"study_design_scores_gemma":[0.000028893373,0.00005193645,0.0003865946,0.00013731304,0.00002443812,0.0001288041,0.0023848927,0.0012499717,0.00034731103,0.9650554,0.030184485,0.00001992792],"about_ca_topic_score_codex":0.0018813118,"about_ca_topic_score_gemma":0.0017006722,"teacher_disagreement_score":0.028176408,"about_ca_system_score_codex":0.002985699,"about_ca_system_score_gemma":0.009882809,"threshold_uncertainty_score":0.14901286},"labels":[],"label_agreement":null},{"id":"W2171199696","doi":"10.1109/fuzzy.2006.1681769","title":"A Methodology for Extracting and Representing Actions in Texts","year":2006,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Sentence; Computer science; Natural language processing; Artificial intelligence; Focus (optics); Classifier (UML); Domain (mathematical analysis); Information retrieval; Mathematics","score_opus":0.09055476831964546,"score_gpt":0.390880960278032,"score_spread":0.3003261919583865,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2171199696","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0008145169,0.000112423644,0.9966846,0.00020705954,0.000029885336,0.00021175887,0.0003745054,0.0009850976,0.0005801557],"genre_scores_gemma":[0.007948572,0.00015148874,0.9900921,0.00005796818,0.000032800566,0.00028445126,0.0006753792,0.00006614772,0.00069105317],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9960567,0.0010612481,0.00077606586,0.0011663275,0.00083698536,0.00010263759],"domain_scores_gemma":[0.99433786,0.002595294,0.0008958381,0.0010515061,0.0009561581,0.00016327295],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0043736114,0.0019869239,0.0009635579,0.007427926,0.0016858163,0.0046359366,0.0025960885,0.0019704753,0.0038820675],"category_scores_gemma":[0.009878204,0.00095495774,0.0025692522,0.0043973858,0.002930377,0.0075420956,0.0020370055,0.0027944352,0.004126133],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014625586,0.00030979616,0.0023093463,0.002013803,0.00019590935,0.00075214443,0.0033775982,0.012102724,0.03431822,0.26759502,0.012041377,0.66483784],"study_design_scores_gemma":[0.000110209,0.00045049548,0.0032907415,0.0007683365,0.00038327498,0.0027715052,0.0018743904,0.25902998,0.067435615,0.41507,0.24842095,0.00039445198],"about_ca_topic_score_codex":0.0026140457,"about_ca_topic_score_gemma":0.0030269383,"teacher_disagreement_score":0.007427926,"about_ca_system_score_codex":0.0012770474,"about_ca_system_score_gemma":0.0029059916,"threshold_uncertainty_score":0.023130119},"labels":[],"label_agreement":null},{"id":"W2171444999","doi":"","title":"EXPRESSING PROBABILISTIC CONTEXT-FREE GRAMMARS IN THE RELAXED UNIFICATION FORMALISM","year":2003,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Unification; Formalism (music); Probabilistic logic; Computer science; Inference; Rule-based machine translation; Theoretical computer science; Rule of inference; Context-sensitive grammar; Context-free grammar; Unified field theory; Algorithm; Mathematics; Artificial intelligence; Programming language; Theoretical physics; Physics","score_opus":0.019131700380524638,"score_gpt":0.2566125783062081,"score_spread":0.23748087792568345,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2171444999","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.003517274,0.00018279963,0.992126,0.00023811302,0.00004543383,0.000045992285,0.00011605653,0.00069280935,0.003035446],"genre_scores_gemma":[0.121830866,0.0007555946,0.8718554,0.00032935117,0.00015625637,0.00024127487,0.00064619444,0.00038803756,0.0037970631],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9963904,0.0014690486,0.00032270423,0.00049988175,0.0010287394,0.00028921725],"domain_scores_gemma":[0.99725384,0.0014141386,0.00026694813,0.00066277676,0.0003255751,0.00007679576],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0046695042,0.00075092947,0.0010431618,0.001473762,0.0011059095,0.0033462846,0.00296216,0.0016112545,0.0029816849],"category_scores_gemma":[0.0066247624,0.0013751554,0.0028569836,0.001812036,0.0036454862,0.0059496984,0.0033199168,0.003861564,0.0013675028],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000031274576,0.000019246072,0.0001285729,0.00009511751,0.000045942645,0.00029638666,0.00057117385,0.040516492,0.0032527503,0.9314649,0.0012988722,0.022279384],"study_design_scores_gemma":[0.000034094795,0.000024960895,0.00007703528,0.00006147678,0.00007116369,0.00023313945,0.000065362845,0.17723204,0.004550848,0.8000662,0.017527035,0.00005671538],"about_ca_topic_score_codex":0.002741089,"about_ca_topic_score_gemma":0.0028419124,"teacher_disagreement_score":0.0046695042,"about_ca_system_score_codex":0.0009865175,"about_ca_system_score_gemma":0.0019591965,"threshold_uncertainty_score":0.02469498},"labels":[],"label_agreement":null},{"id":"W2171519600","doi":"10.1007/978-3-642-30353-1_14","title":"Domain Adaptation Techniques for Machine Translation and Their Evaluation in a Real-World Setting","year":2012,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Machine translation; Adaptation (eye); Artificial intelligence; Domain (mathematical analysis); Machine learning; Context (archaeology); Word (group theory); Domain adaptation; BLEU; Quality (philosophy); Evaluation of machine translation; Baseline (sea); Natural language processing; Machine translation software usability; Example-based machine translation","score_opus":0.03277191709911327,"score_gpt":0.3040883991362745,"score_spread":0.27131648203716124,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2171519600","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.49419284,0.023090318,0.42795303,0.0010738118,0.0014924255,0.0014319689,0.0040581445,0.025509609,0.021197826],"genre_scores_gemma":[0.5948863,0.005817453,0.37914324,0.00044603704,0.00027470058,0.000842436,0.010886409,0.0015373601,0.0061659715],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9937484,0.0038436342,0.0004658837,0.0008346596,0.0008582399,0.0002490594],"domain_scores_gemma":[0.98768485,0.008708501,0.00040196662,0.001614616,0.0013548292,0.00023517842],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0057250033,0.0020902317,0.0013668186,0.0023788796,0.0010510004,0.0016867853,0.0017110939,0.002410927,0.0041581686],"category_scores_gemma":[0.017070998,0.00050735235,0.0010129551,0.0044261124,0.0008709515,0.0023102136,0.0021510443,0.0020086963,0.0026342948],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0031785402,0.0014877038,0.003263173,0.0019980494,0.0008660193,0.0007205858,0.00066007953,0.078068785,0.021298736,0.0018559108,0.017234512,0.8693679],"study_design_scores_gemma":[0.0016514795,0.0036307645,0.015944678,0.00039605165,0.0014458597,0.0030583928,0.0016088242,0.8649913,0.06816617,0.010859814,0.027911674,0.00033502566],"about_ca_topic_score_codex":0.0041090744,"about_ca_topic_score_gemma":0.0039813016,"teacher_disagreement_score":0.0057250033,"about_ca_system_score_codex":0.0006523927,"about_ca_system_score_gemma":0.0009462576,"threshold_uncertainty_score":0.030277014},"labels":[],"label_agreement":null},{"id":"W2171694517","doi":"10.3115/1219044.1219071","title":"Automatic clustering of collocation for detecting practical sense boundary","year":2004,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Collocation (remote sensing); Computer science; Boundary (topology); Bottleneck; Cluster analysis; Natural language processing; Sense (electronics); Confusion; Resource (disambiguation); Word-sense disambiguation; Artificial intelligence; Information retrieval; Machine learning; Mathematics; WordNet; Engineering","score_opus":0.020979735011627643,"score_gpt":0.3230780631393331,"score_spread":0.3020983281277055,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2171694517","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.067405,0.0010813884,0.92096674,0.000294666,0.00026710777,0.00034304973,0.00091718405,0.003960114,0.004764732],"genre_scores_gemma":[0.26175755,0.0003058595,0.7308832,0.00010326643,0.00006118823,0.00026218747,0.0035828114,0.000550587,0.0024934367],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99552035,0.001315587,0.00037966308,0.0015840989,0.0009565177,0.00024383722],"domain_scores_gemma":[0.9936813,0.002138239,0.0005306528,0.0012387342,0.0021673348,0.0002436922],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018122558,0.0012710455,0.0010146893,0.007941107,0.0025882637,0.0017144115,0.0014157386,0.0015324316,0.0040542837],"category_scores_gemma":[0.01100694,0.00067460025,0.00091889064,0.0050556934,0.0011902273,0.003239264,0.0023143103,0.0012380289,0.003444033],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006807579,0.00022323459,0.012137911,0.001356068,0.0002556584,0.0009170616,0.0038982942,0.012560395,0.15592727,0.022535512,0.028569251,0.7609386],"study_design_scores_gemma":[0.00019253817,0.00038990934,0.033746667,0.00040526874,0.00033237328,0.0034820645,0.0059546023,0.649538,0.14075936,0.06324829,0.101478204,0.00047272578],"about_ca_topic_score_codex":0.0044006514,"about_ca_topic_score_gemma":0.0057664462,"teacher_disagreement_score":0.007941107,"about_ca_system_score_codex":0.00082439533,"about_ca_system_score_gemma":0.0016898562,"threshold_uncertainty_score":0.013562918},"labels":[],"label_agreement":null},{"id":"W2172057406","doi":"10.1162/002438900554497","title":"The Mental Representation of Semitic Words","year":2000,"lang":"en","type":"article","venue":"Linguistic Inquiry","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":157,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Institut Universitaire de Gériatrie de Montréal; Université du Québec à Montréal","funders":"","keywords":"Morpheme; Linguistics; Semitic languages; Representation (politics); Computer science; Psychology; Natural language processing; Philosophy; Arabic","score_opus":0.020909619772314807,"score_gpt":0.32211379689900027,"score_spread":0.30120417712668546,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2172057406","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6194126,0.0016357551,0.24938653,0.0015383601,0.000108528366,0.00010556194,0.0005335584,0.0005921282,0.12668695],"genre_scores_gemma":[0.96694493,0.00055277924,0.02904962,0.000070679795,0.000025971856,0.000033277232,0.00039259478,0.00006262104,0.0028675213],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9993826,0.00020148055,0.00005686446,0.00013479195,0.00017921093,0.000045102737],"domain_scores_gemma":[0.99840194,0.0006794459,0.00028958012,0.00039271012,0.00018341208,0.000052974774],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007451099,0.00024734676,0.0001836439,0.0014186071,0.00036742858,0.0046839938,0.0007007035,0.00047490498,0.002952846],"category_scores_gemma":[0.0060154335,0.0002147155,0.00043667894,0.00069445866,0.0027961961,0.0051192828,0.001395,0.00060322566,0.0005170413],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015697576,0.000042346666,0.008844712,0.00056453556,0.00007822428,0.0006544893,0.030314995,0.0029438597,0.029065384,0.7129464,0.002160568,0.21222754],"study_design_scores_gemma":[0.000029145627,0.0001633166,0.014860161,0.00021516383,0.00008980875,0.0024406812,0.011970198,0.0199598,0.014653478,0.88659936,0.048901692,0.00011718588],"about_ca_topic_score_codex":0.0006129979,"about_ca_topic_score_gemma":0.0005867374,"teacher_disagreement_score":0.0046839938,"about_ca_system_score_codex":0.00055087695,"about_ca_system_score_gemma":0.00037068652,"threshold_uncertainty_score":0.009878218},"labels":[],"label_agreement":null},{"id":"W2178237250","doi":"","title":"Automatic identification of words with novel but infrequent senses","year":2011,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Semantic change; Focus (optics); Identification (biology); Task (project management); Word (group theory); Contrast (vision); Baseline (sea); Linguistics","score_opus":0.022973635592338358,"score_gpt":0.2528105134744438,"score_spread":0.2298368778821054,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2178237250","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7370705,0.0012184839,0.2495122,0.00044117708,0.0004955455,0.00042320718,0.0033107742,0.003075388,0.0044527827],"genre_scores_gemma":[0.8672548,0.00017748403,0.12714599,0.00010440626,0.000074883326,0.0002035536,0.0038185888,0.00032440363,0.00089595444],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9968059,0.000804003,0.00035877936,0.0010083928,0.0009173731,0.000105592764],"domain_scores_gemma":[0.98375875,0.008776067,0.0021494364,0.0020506105,0.0029056063,0.00035968167],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020264727,0.0005668906,0.0007575925,0.0032588742,0.00091197086,0.0017302567,0.00092164613,0.0009689175,0.0016054392],"category_scores_gemma":[0.015818192,0.00039880438,0.0006664587,0.0021327862,0.00097099313,0.0030735019,0.0016475753,0.000916023,0.0009115257],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002237497,0.0006772538,0.09818155,0.0021982035,0.0005679781,0.0017812235,0.002054789,0.019093225,0.2280126,0.014453981,0.014848697,0.615893],"study_design_scores_gemma":[0.00038124924,0.00094563235,0.079858914,0.00037980493,0.00041545433,0.007335814,0.0026799266,0.61489105,0.207185,0.04736951,0.038197603,0.00036008624],"about_ca_topic_score_codex":0.0007550163,"about_ca_topic_score_gemma":0.0016619896,"teacher_disagreement_score":0.0032588742,"about_ca_system_score_codex":0.00037454345,"about_ca_system_score_gemma":0.0007347826,"threshold_uncertainty_score":0.010717094},"labels":[],"label_agreement":null},{"id":"W2179150490","doi":"10.18653/v1/s15-1016","title":"Mapping Different Rhetorical Relation Annotations: A Proposal","year":2015,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":37,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Relation (database); Rhetorical question; Computer science; Natural language processing; Linguistics; Data mining; Philosophy","score_opus":0.039607641836695476,"score_gpt":0.2967326707298348,"score_spread":0.2571250288931393,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2179150490","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016981049,0.0014110116,0.93274814,0.013196192,0.0009611683,0.0005203083,0.0010860689,0.0042383526,0.02885774],"genre_scores_gemma":[0.11008455,0.0011099493,0.8736034,0.0019759773,0.0004882238,0.00074464804,0.0022243361,0.0012982218,0.008470746],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.98307437,0.005668707,0.0013375247,0.0052578608,0.003474076,0.0011874689],"domain_scores_gemma":[0.9588554,0.011159795,0.002029668,0.0142139625,0.011492637,0.0022484395],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.023882864,0.0014674859,0.0016878912,0.015123966,0.0071761785,0.01165935,0.0052785194,0.005799768,0.008019949],"category_scores_gemma":[0.042855956,0.0019541031,0.002618591,0.014629515,0.008184797,0.042523026,0.009668258,0.0062348004,0.005188647],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001847507,0.00033300178,0.008246928,0.0005334511,0.00008052517,0.0002242989,0.00855241,0.002335731,0.0065149516,0.5663121,0.01428559,0.39239624],"study_design_scores_gemma":[0.000079499514,0.00018298296,0.003646671,0.0006070324,0.00019635772,0.0006677586,0.0068723885,0.032537308,0.006058595,0.7216417,0.2272939,0.00021586913],"about_ca_topic_score_codex":0.01362436,"about_ca_topic_score_gemma":0.008208044,"teacher_disagreement_score":0.023882864,"about_ca_system_score_codex":0.003512101,"about_ca_system_score_gemma":0.010779152,"threshold_uncertainty_score":0.12630618},"labels":[],"label_agreement":null},{"id":"W2179916386","doi":"10.1007/978-3-540-89778-1_1","title":"Ambiguity in Natural Language Requirements Documents","year":2008,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":48,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Ambiguity; Natural (archaeology); Computer science; Linguistics; Natural language; Natural language processing; Programming language; History; Philosophy; Archaeology","score_opus":0.015582989552096003,"score_gpt":0.2882899828273541,"score_spread":0.27270699327525805,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2179916386","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025666729,0.0028887063,0.9048654,0.0023141021,0.0003104845,0.00028523713,0.00055318995,0.0016269069,0.061489213],"genre_scores_gemma":[0.49815404,0.002894182,0.46201405,0.00080812536,0.0004424506,0.00038011698,0.002597705,0.0017021418,0.031007165],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9882882,0.0041828873,0.001309008,0.00095784914,0.0047318386,0.0005301675],"domain_scores_gemma":[0.9767973,0.01712197,0.0011836499,0.0023055898,0.002338831,0.00025263225],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0056367693,0.000821854,0.0011195969,0.0033101737,0.0019102368,0.005762046,0.0014939778,0.0018178066,0.006054046],"category_scores_gemma":[0.0314485,0.0018194839,0.0011828888,0.003396703,0.0027772763,0.012941245,0.0030857096,0.0037514856,0.0024062707],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001378815,0.00006777481,0.00076556765,0.0005371257,0.000021189266,0.00093952485,0.003981256,0.008285034,0.004411249,0.7896033,0.010399202,0.1808509],"study_design_scores_gemma":[0.000028300268,0.000041866067,0.00038056867,0.0003396309,0.000028679773,0.001279213,0.001222512,0.026425848,0.009165588,0.87645787,0.08455548,0.000074434094],"about_ca_topic_score_codex":0.0016280458,"about_ca_topic_score_gemma":0.0012229041,"teacher_disagreement_score":0.006054046,"about_ca_system_score_codex":0.0019278665,"about_ca_system_score_gemma":0.0021004756,"threshold_uncertainty_score":0.029810429},"labels":[],"label_agreement":null},{"id":"W2180877780","doi":"","title":"Reduplication and initial change in Sheshatshiu Innu-aimun","year":2009,"lang":"en","type":"dissertation","venue":"Memorial University Research Repository (Memorial University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Reduplication; Vowel; Linguistics; Sound change; Verb; History; Geography; Philosophy","score_opus":0.03458871754571986,"score_gpt":0.31166738627046753,"score_spread":0.2770786687247477,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2180877780","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9946911,0.00019595584,0.000121699944,0.000035104444,0.0000058630844,0.000009585497,0.000030419678,0.000003425077,0.004906901],"genre_scores_gemma":[0.99691796,0.00014474252,0.00040366084,0.000023812527,0.0000024994267,0.000009662509,0.000064333806,0.000007574576,0.0024257444],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.999757,0.000061515035,0.000031891308,0.00005518728,0.000042477954,0.00005199606],"domain_scores_gemma":[0.999491,0.00023662877,0.00008059477,0.000038439768,0.00011210088,0.000041233754],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00039159055,0.00018874202,0.00024952783,0.00078095216,0.0013815338,0.0007393221,0.00020318563,0.00022335311,0.0020469588],"category_scores_gemma":[0.0009647884,0.00015680045,0.00013433993,0.000883032,0.0009906499,0.00039902248,0.0005039657,0.00037384697,0.00026108898],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013823876,0.0004287385,0.2757553,0.00058367883,0.00003838237,0.005662823,0.37750575,0.00037797325,0.13370998,0.018730003,0.0010104508,0.18481451],"study_design_scores_gemma":[0.00002822472,0.00036347992,0.8915313,0.00009153544,0.00003683287,0.0021427744,0.07407866,0.0005255373,0.012072987,0.0020850871,0.016986158,0.00005751592],"about_ca_topic_score_codex":0.014314763,"about_ca_topic_score_gemma":0.04569831,"teacher_disagreement_score":0.014314763,"about_ca_system_score_codex":0.0008505823,"about_ca_system_score_gemma":0.00075807126,"threshold_uncertainty_score":0.028462887},"labels":[],"label_agreement":null},{"id":"W2181194706","doi":"","title":"A Continuum-Based Approach for Tightness Analysis of Chinese Semantic Units","year":2009,"lang":"en","type":"article","venue":"Institutional Repositories DataBase (IRDB)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Ranking (information retrieval); Natural language processing; Artificial intelligence; Information retrieval; Segmentation; Search engine; Connection (principal bundle); Variety (cybernetics); Mathematics","score_opus":0.016237351576211964,"score_gpt":0.2835296992195975,"score_spread":0.2672923476433855,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2181194706","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.28900084,0.0012363716,0.6859715,0.00016079478,0.000050364233,0.00042682124,0.0024000397,0.002730977,0.018022243],"genre_scores_gemma":[0.74331975,0.00036682255,0.24915613,0.000067844616,0.000055931712,0.000564079,0.003331742,0.0002483952,0.0028893433],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9982476,0.0002414696,0.00022187931,0.0005112694,0.0006156247,0.00016217006],"domain_scores_gemma":[0.9977543,0.0007345816,0.00033158457,0.00039547987,0.0006294263,0.00015459738],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014791455,0.0006854495,0.000727306,0.018776342,0.0018818221,0.0022451237,0.0011074081,0.00061194104,0.0053222678],"category_scores_gemma":[0.0050894283,0.00046834504,0.0009579485,0.01076953,0.0017931149,0.0040642736,0.0025687108,0.001204275,0.0011028012],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009173218,0.00037161296,0.06461016,0.0009533226,0.00035835686,0.0014136032,0.006804856,0.020724095,0.13116854,0.17317478,0.0064007235,0.5931027],"study_design_scores_gemma":[0.00010263048,0.0006823484,0.15514383,0.00023849402,0.00034804744,0.0023259216,0.005396855,0.49301204,0.05365149,0.24697922,0.0417176,0.00040146871],"about_ca_topic_score_codex":0.0057681226,"about_ca_topic_score_gemma":0.005329654,"teacher_disagreement_score":0.018776342,"about_ca_system_score_codex":0.0013304562,"about_ca_system_score_gemma":0.0011967443,"threshold_uncertainty_score":0.017804801},"labels":[],"label_agreement":null},{"id":"W2181540934","doi":"","title":"The LIA Update Summarization Systems at TAC-2008 (Draft)","year":2008,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Overfitting; Computer science; Automatic summarization; Sentence; Natural language processing; Artificial intelligence; Artificial neural network","score_opus":0.0064350971868102685,"score_gpt":0.2320949621177877,"score_spread":0.22565986493097745,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2181540934","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.102395706,0.010753801,0.47073674,0.008317499,0.006951353,0.0063159624,0.09324653,0.20537466,0.09590776],"genre_scores_gemma":[0.1786518,0.0012338018,0.5111871,0.0012452813,0.0014514591,0.0026916321,0.16047536,0.005114754,0.13794883],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99713874,0.0011031156,0.0002103575,0.00040700845,0.00090402085,0.00023675275],"domain_scores_gemma":[0.9909574,0.0007512372,0.0003453813,0.0009327679,0.0065903133,0.00042287348],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004989781,0.0013694135,0.0010626793,0.002891299,0.0011062457,0.0028165034,0.0015834589,0.0011990177,0.012363622],"category_scores_gemma":[0.008412043,0.00047299036,0.0005864677,0.0011991714,0.00029019904,0.0015509807,0.0010730231,0.0016152266,0.0109378435],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005642133,0.0002590505,0.00142741,0.00052147475,0.00009561957,0.00019785004,0.00050525164,0.0033283273,0.015612467,0.0020370528,0.51983804,0.45561326],"study_design_scores_gemma":[0.00046884074,0.0017203464,0.011021558,0.00021497,0.0002890016,0.00039833185,0.00078411103,0.06755888,0.08006096,0.002790287,0.834442,0.00025069033],"about_ca_topic_score_codex":0.018008366,"about_ca_topic_score_gemma":0.032192566,"teacher_disagreement_score":0.018008366,"about_ca_system_score_codex":0.0021380568,"about_ca_system_score_gemma":0.0018377021,"threshold_uncertainty_score":0.04136038},"labels":[],"label_agreement":null},{"id":"W2182244728","doi":"","title":"The CIST Summarization System at TAC 2010","year":2010,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Automatic summarization; Computer science; Multi-document summarization; Information retrieval","score_opus":0.004082541518875471,"score_gpt":0.23154461198033158,"score_spread":0.2274620704614561,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2182244728","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08041747,0.0044609993,0.43521604,0.0024579691,0.0028610618,0.003535869,0.21143669,0.212089,0.047524974],"genre_scores_gemma":[0.085663125,0.0008654327,0.38122228,0.00026805673,0.0005477679,0.0014160317,0.49404293,0.0077818637,0.028192487],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9969855,0.0008592637,0.00035940157,0.0005355175,0.0010710784,0.0001892341],"domain_scores_gemma":[0.98842037,0.0016864452,0.0006250878,0.0018885679,0.006813322,0.00056624843],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0043231444,0.0015506357,0.0010395368,0.0055951783,0.001353314,0.0025345772,0.0016375108,0.00089012587,0.011167617],"category_scores_gemma":[0.01177144,0.00036982982,0.00074880576,0.0043915254,0.00026450932,0.0023519616,0.0011061787,0.0016363147,0.008548023],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00061735377,0.00032154148,0.0026096713,0.0012206964,0.00028036523,0.00033519435,0.0012564271,0.0050941235,0.029858017,0.002684888,0.51295215,0.44276953],"study_design_scores_gemma":[0.0005406461,0.0013400582,0.012303446,0.00021491678,0.0006099149,0.00050674164,0.0009260119,0.075352386,0.071769275,0.005933225,0.8302678,0.00023557307],"about_ca_topic_score_codex":0.009449273,"about_ca_topic_score_gemma":0.01359955,"teacher_disagreement_score":0.011167617,"about_ca_system_score_codex":0.0010874103,"about_ca_system_score_gemma":0.0022189938,"threshold_uncertainty_score":0.037359416},"labels":[],"label_agreement":null},{"id":"W2182343512","doi":"","title":"Sidestepping the combinatorial explosion: Towards a processing model based on discriminative learning","year":2010,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Discriminative model; Computer science; Artificial intelligence; Natural language processing; Word (group theory); Word lists by frequency; Decomposition; Speech recognition; Linguistics","score_opus":0.0193519442914011,"score_gpt":0.2846260567293512,"score_spread":0.2652741124379501,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2182343512","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019664712,0.00012500478,0.9772071,0.0005642874,0.000022376798,0.000028822973,0.00007759659,0.00029246105,0.002017577],"genre_scores_gemma":[0.5622754,0.0005248287,0.42699203,0.00073110545,0.00015164097,0.0003145012,0.0004832773,0.0002877075,0.008239571],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99942905,0.000209956,0.000027817447,0.00014501261,0.00013489298,0.00005337062],"domain_scores_gemma":[0.9968827,0.0021401462,0.00020869156,0.00048919406,0.00019940473,0.00007987032],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001319727,0.0006056273,0.0011196913,0.00082123245,0.0005381056,0.0014039242,0.0024100235,0.0009883167,0.0032683562],"category_scores_gemma":[0.00602622,0.00073495624,0.0012446009,0.0010330736,0.002348672,0.004336831,0.0016663524,0.0028828185,0.0007697751],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015648424,0.00011841814,0.0015436074,0.0001838185,0.00008618941,0.00013183641,0.00025350746,0.5222107,0.0038264305,0.37804082,0.0024791404,0.09096903],"study_design_scores_gemma":[0.000007590405,0.000012635187,0.00006196523,0.0000039383076,0.0000061279334,0.000017302746,0.000004668323,0.8710552,0.00029955755,0.1281421,0.00038339745,0.0000054667357],"about_ca_topic_score_codex":0.0024450668,"about_ca_topic_score_gemma":0.003391464,"teacher_disagreement_score":0.0032683562,"about_ca_system_score_codex":0.001085903,"about_ca_system_score_gemma":0.0009080898,"threshold_uncertainty_score":0.010933757},"labels":[],"label_agreement":null},{"id":"W2182572069","doi":"","title":"University of Lethbridge's Participation in DUC-2007 Main Task","year":2007,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Lethbridge","funders":"","keywords":"Automatic summarization; Computer science; Task (project management); Feature (linguistics); Natural language processing; Multi-document summarization; Artificial intelligence; Information retrieval; Linguistics; Engineering","score_opus":0.010213491033252926,"score_gpt":0.2706711809058771,"score_spread":0.2604576898726242,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2182572069","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.31939042,0.019704117,0.19953689,0.023435261,0.009475673,0.0052676653,0.12844801,0.10943488,0.18530707],"genre_scores_gemma":[0.24585862,0.0023379866,0.2594614,0.0020491558,0.00094683445,0.0015746133,0.29376876,0.0043748417,0.1896277],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99396825,0.0028466939,0.0002607692,0.0010356674,0.0014669785,0.0004215679],"domain_scores_gemma":[0.99147147,0.0019146053,0.00016098049,0.0013157872,0.003759922,0.001377184],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008171044,0.0018932411,0.001655237,0.0022139626,0.0025363415,0.002435933,0.0016672178,0.0018327946,0.016055144],"category_scores_gemma":[0.009086772,0.00038529185,0.0003551585,0.0017991062,0.0005500383,0.001544227,0.001693388,0.0014722849,0.010720072],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009858195,0.00071746914,0.0015060959,0.0006012531,0.000089364796,0.00033910177,0.0008866471,0.0027954613,0.021204025,0.0016908697,0.64998543,0.31919846],"study_design_scores_gemma":[0.0007074667,0.0015981385,0.01123912,0.00015662797,0.0001229938,0.0006482969,0.001228393,0.04013419,0.06352055,0.0028918907,0.8775416,0.00021079711],"about_ca_topic_score_codex":0.04217441,"about_ca_topic_score_gemma":0.06453404,"teacher_disagreement_score":0.04217441,"about_ca_system_score_codex":0.0020015019,"about_ca_system_score_gemma":0.0033431628,"threshold_uncertainty_score":0.083857834},"labels":[],"label_agreement":null},{"id":"W2182677133","doi":"","title":"A Textual Entailment System using Anaphora Resolution.","year":2011,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Novelty; Task (project management); Textual entailment; Computer science; Natural language processing; Set (abstract data type); Test set; Artificial intelligence; Logical consequence; Test (biology); Novelty detection; Resolution (logic); Programming language","score_opus":0.0174668792673123,"score_gpt":0.2588937167682667,"score_spread":0.24142683750095442,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2182677133","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.057252448,0.0016938429,0.74037635,0.0012738799,0.00039685515,0.0013404264,0.011505764,0.17066576,0.015494598],"genre_scores_gemma":[0.15394774,0.00052472,0.7949235,0.0005220231,0.00018485694,0.000536877,0.030600678,0.0023775,0.016382143],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9977641,0.00050088856,0.00036612322,0.000683121,0.00061603705,0.000069758986],"domain_scores_gemma":[0.9965873,0.0016317052,0.00032567943,0.0005626163,0.00078956276,0.00010319767],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025918803,0.001105384,0.00087620126,0.0020479164,0.00085333636,0.0018664419,0.001665525,0.0014782,0.014412053],"category_scores_gemma":[0.008770407,0.0006261644,0.0013394004,0.0010225923,0.0002969619,0.003969748,0.0020712751,0.0010931961,0.010210977],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010192784,0.00059932336,0.0026327074,0.0017308698,0.00027133242,0.0011341609,0.00077235664,0.0045450036,0.09396044,0.008525405,0.07920325,0.8056059],"study_design_scores_gemma":[0.00086070626,0.001015036,0.011396227,0.00030856236,0.0006859895,0.0063532996,0.0011909001,0.4153585,0.27854007,0.027096232,0.25691432,0.00028012774],"about_ca_topic_score_codex":0.0019248191,"about_ca_topic_score_gemma":0.0016912066,"teacher_disagreement_score":0.014412053,"about_ca_system_score_codex":0.00062360876,"about_ca_system_score_gemma":0.0011785649,"threshold_uncertainty_score":0.048213124},"labels":[],"label_agreement":null},{"id":"W2182907248","doi":"","title":"Knowledge Base Augmentation using Tabular Data","year":2014,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":43,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Knowledge base; Information retrieval; Semantics (computer science); Natural language; Base (topology); Entity linking; Linked data; Probabilistic logic; Natural language processing; World Wide Web; Semantic Web; Artificial intelligence; Programming language","score_opus":0.06400391790445405,"score_gpt":0.3500177096186218,"score_spread":0.28601379171416774,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2182907248","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.046943836,0.0013017622,0.90740144,0.001529376,0.00054700027,0.0008406199,0.012702671,0.01737688,0.011356414],"genre_scores_gemma":[0.15759327,0.0009346083,0.81564075,0.00046976793,0.00016145234,0.0006634685,0.019613523,0.00041957578,0.004503665],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9978003,0.0006552888,0.00022189306,0.00042087972,0.0007984323,0.000103204715],"domain_scores_gemma":[0.9824225,0.008349932,0.0007000039,0.0038341524,0.0044668675,0.00022660915],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032135348,0.0007538599,0.0008977494,0.0057788873,0.0009557169,0.0027229127,0.0024597172,0.0009775818,0.008490821],"category_scores_gemma":[0.025966179,0.00067855965,0.0012085249,0.0059687844,0.00063873717,0.0057138447,0.0030685107,0.0017862185,0.0035067028],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004558917,0.00060854433,0.006680395,0.000990637,0.00026627732,0.00082755514,0.00073003245,0.063777186,0.012842227,0.011854836,0.027922777,0.8730437],"study_design_scores_gemma":[0.00011725516,0.0002615337,0.0032358193,0.00055062043,0.000407417,0.00076470466,0.0006875732,0.8027703,0.0381616,0.047859978,0.10506298,0.000120198216],"about_ca_topic_score_codex":0.0052134027,"about_ca_topic_score_gemma":0.009254581,"teacher_disagreement_score":0.008490821,"about_ca_system_score_codex":0.00083869195,"about_ca_system_score_gemma":0.0023744805,"threshold_uncertainty_score":0.028404653},"labels":[],"label_agreement":null},{"id":"W2183009455","doi":"","title":"University of Ottawa's participation in the CL-SR task at CLEF 2006","year":2006,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Ottawa","funders":"","keywords":"Clef; Computer science; Natural language processing; Information retrieval; Task (project management); Scripting language; Query expansion; Document retrieval; Search engine indexing; Artificial intelligence; Word (group theory); Automatic indexing; Weighting; Linguistics","score_opus":0.00795903993459232,"score_gpt":0.24182006334073794,"score_spread":0.23386102340614562,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2183009455","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.76882464,0.0067743715,0.0122833215,0.01325654,0.0031481201,0.0022875927,0.09937949,0.019930279,0.07411572],"genre_scores_gemma":[0.6429794,0.0010022288,0.038589437,0.002002586,0.00049298204,0.00075466855,0.19684629,0.0031258587,0.11420657],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9908235,0.003295497,0.0004880339,0.0018366034,0.002068535,0.00148779],"domain_scores_gemma":[0.9805572,0.0046454486,0.00037118426,0.0037036603,0.007718293,0.003004248],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010454864,0.0021567435,0.0031311589,0.0016847615,0.0053439736,0.0029791498,0.00208295,0.0027765066,0.023492355],"category_scores_gemma":[0.015315959,0.0008439308,0.0010041393,0.0019871586,0.001250456,0.001635877,0.0025392745,0.001755714,0.013112272],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004725108,0.0018345352,0.00893732,0.0011066928,0.00033001657,0.0012642657,0.0029252924,0.0072350493,0.030108443,0.0009822946,0.76729155,0.17325948],"study_design_scores_gemma":[0.0043815514,0.0032932747,0.11175876,0.0002521012,0.00041255282,0.0014931837,0.011084084,0.053632542,0.10206503,0.0013312912,0.7091981,0.0010975472],"about_ca_topic_score_codex":0.45016265,"about_ca_topic_score_gemma":0.5894521,"teacher_disagreement_score":0.45016265,"about_ca_system_score_codex":0.008410718,"about_ca_system_score_gemma":0.011015327,"threshold_uncertainty_score":0.8950848},"labels":[],"label_agreement":null},{"id":"W2183255880","doi":"","title":"Illinois Cognitive Computation Group UI-CCG TAC 2013 Entity Linking and Slot Filler Validation Systems","year":2013,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Filler (materials); Computer science; Computation; Knowledge base; Logical consequence; Entity linking; Natural language processing; Group (periodic table); Base (topology); Information retrieval; Artificial intelligence; Programming language; Mathematics; Engineering","score_opus":0.008501356371943021,"score_gpt":0.25016348355651735,"score_spread":0.24166212718457433,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2183255880","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04518654,0.0010967863,0.5771549,0.010283222,0.0019406189,0.002085076,0.011277259,0.09189014,0.2590855],"genre_scores_gemma":[0.25641462,0.00043115357,0.596993,0.0015145593,0.0004290678,0.0023590787,0.03482488,0.006485255,0.10054835],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9913662,0.0032045878,0.0005307202,0.0012627735,0.0030310452,0.0006045227],"domain_scores_gemma":[0.97892886,0.005850336,0.00045105565,0.0030696057,0.010386498,0.0013135485],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008343401,0.0011474867,0.001000057,0.0034008666,0.0036277163,0.006552155,0.0026140197,0.0017014358,0.038346075],"category_scores_gemma":[0.027795032,0.00083926634,0.0009669892,0.0028752089,0.0014284273,0.0046040136,0.0046478994,0.0024585987,0.02041766],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005873418,0.00030098626,0.0021722768,0.00032067593,0.000065806234,0.00041721915,0.0018776342,0.0071456423,0.007353471,0.04264723,0.5560796,0.38103208],"study_design_scores_gemma":[0.00027512817,0.00020123865,0.0050820727,0.0002532575,0.00009754057,0.0006418463,0.0017554164,0.18877819,0.027941942,0.06369611,0.710977,0.00030021384],"about_ca_topic_score_codex":0.041337673,"about_ca_topic_score_gemma":0.061782293,"teacher_disagreement_score":0.041337673,"about_ca_system_score_codex":0.004688596,"about_ca_system_score_gemma":0.012244879,"threshold_uncertainty_score":0.1282804},"labels":[],"label_agreement":null},{"id":"W2183577233","doi":"","title":"TAC 2009 Update Summarization of ICL","year":2009,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Automatic summarization; Computer science; Graph; Cluster analysis; Task (project management); Ranking (information retrieval); Cluster (spacecraft); Information retrieval; Natural language processing; Node (physics); Artificial intelligence; Theoretical computer science","score_opus":0.004624052201121328,"score_gpt":0.24870546263396479,"score_spread":0.24408141043284345,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2183577233","genre_codex":"empirical","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.30921438,0.004915189,0.2554633,0.0044023977,0.0054113977,0.0053276713,0.078321315,0.284722,0.052222338],"genre_scores_gemma":[0.37804213,0.0004705485,0.3713711,0.0010450417,0.0012960895,0.0018807881,0.20382628,0.007819685,0.03424845],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9963198,0.001320626,0.00031597933,0.00082823983,0.00085252605,0.000362917],"domain_scores_gemma":[0.9907176,0.0028450927,0.00039498296,0.0019507007,0.0035699992,0.0005216844],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0041751126,0.0025763088,0.001733793,0.0031238,0.0015675379,0.001829736,0.0019765077,0.0020746992,0.013108509],"category_scores_gemma":[0.015116963,0.0005155656,0.0010181988,0.0022540134,0.0004804195,0.0022506649,0.0012924112,0.002142662,0.0058479393],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018919779,0.00096452364,0.0017537328,0.0013741282,0.00037916336,0.0009083628,0.0007241085,0.028554538,0.032593485,0.0018453561,0.46777588,0.46123472],"study_design_scores_gemma":[0.0016486681,0.0029332219,0.0105412565,0.00013108706,0.0005160634,0.0009350951,0.0012731161,0.59865415,0.1208328,0.0056878803,0.2565463,0.00030031978],"about_ca_topic_score_codex":0.012336692,"about_ca_topic_score_gemma":0.022148654,"teacher_disagreement_score":0.013108509,"about_ca_system_score_codex":0.0015577384,"about_ca_system_score_gemma":0.0015137922,"threshold_uncertainty_score":0.04385239},"labels":[],"label_agreement":null},{"id":"W2184088174","doi":"","title":"BEwT-E for TAC 2009's AESOP Task","year":2009,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Automatic summarization; Task (project management); Computer science; Natural language processing; Metric (unit); Artificial intelligence; Information retrieval; Engineering","score_opus":0.0063801030929812146,"score_gpt":0.2674144851177682,"score_spread":0.261034382024787,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2184088174","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.34852296,0.0050752843,0.22019087,0.004796113,0.005928003,0.009842093,0.14920016,0.19581275,0.06063188],"genre_scores_gemma":[0.26488546,0.0005607918,0.29041287,0.00093846605,0.00053947186,0.0031889451,0.4069172,0.0044814576,0.02807538],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99287814,0.0027759739,0.00073842885,0.0011338919,0.0020450756,0.00042851706],"domain_scores_gemma":[0.9885943,0.0032522075,0.0004768733,0.0025189286,0.00437892,0.00077884074],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006396064,0.0020263253,0.0015926893,0.003054642,0.0016961771,0.0023075775,0.0019151479,0.0020994942,0.0092492355],"category_scores_gemma":[0.02069672,0.000440672,0.0010555539,0.0019393242,0.00052518747,0.0043012314,0.002451964,0.0023337856,0.008836035],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002055051,0.001803937,0.0038751806,0.0013185708,0.0003632734,0.0004635625,0.000736186,0.009630024,0.02956395,0.0026075228,0.4886925,0.4588903],"study_design_scores_gemma":[0.0022318454,0.00501859,0.03638262,0.00029992877,0.00044474142,0.0014066832,0.0025474369,0.27537638,0.11082516,0.011346156,0.5536315,0.00048893754],"about_ca_topic_score_codex":0.014551967,"about_ca_topic_score_gemma":0.019197686,"teacher_disagreement_score":0.014551967,"about_ca_system_score_codex":0.0013698759,"about_ca_system_score_gemma":0.002002263,"threshold_uncertainty_score":0.033826053},"labels":[],"label_agreement":null},{"id":"W2184107510","doi":"","title":"IKOMA at TAC2011: A Method for Recognizing Textual Entailment using Lexical-level and Sentence Structure-level features","year":2011,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Textual entailment; Logical consequence; Natural language processing; Computer science; Sentence; Predicate (mathematical logic); Artificial intelligence; Argument (complex analysis); Matching (statistics); Word (group theory); Linguistics; Mathematics; Programming language","score_opus":0.051837636617153786,"score_gpt":0.3178900933876772,"score_spread":0.2660524567705234,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2184107510","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03662354,0.0007655247,0.9165519,0.00027624748,0.00020546089,0.000626195,0.005385582,0.030907542,0.008658031],"genre_scores_gemma":[0.14631028,0.00024418187,0.8286776,0.00016057392,0.00010387016,0.00057110295,0.016351143,0.0012142385,0.0063669593],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99863654,0.00021901373,0.00018589244,0.00035849173,0.000491226,0.000108750726],"domain_scores_gemma":[0.99769175,0.0005732972,0.000301951,0.0005358131,0.0007714855,0.00012563313],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015224367,0.0011610138,0.0009906405,0.0053807613,0.0013601416,0.001944745,0.0016993209,0.0014203828,0.0075361305],"category_scores_gemma":[0.0067371577,0.00064487866,0.0012830506,0.0021306705,0.00055086386,0.0029061395,0.0016085727,0.0013621295,0.0046034697],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008649235,0.0005174382,0.010180231,0.0011614403,0.0004231617,0.00096844113,0.0012059908,0.0034187075,0.07608419,0.013831255,0.05041022,0.840934],"study_design_scores_gemma":[0.0004721133,0.00084581785,0.0336644,0.00031413313,0.0011783898,0.004382055,0.0015123467,0.47496545,0.2412529,0.053581554,0.18736866,0.00046211807],"about_ca_topic_score_codex":0.0041980855,"about_ca_topic_score_gemma":0.010320172,"teacher_disagreement_score":0.0075361305,"about_ca_system_score_codex":0.0008267632,"about_ca_system_score_gemma":0.00166556,"threshold_uncertainty_score":0.025210917},"labels":[],"label_agreement":null},{"id":"W2184852068","doi":"","title":"The CASIA Entity linking System at TAC 2013","year":2013,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Entity linking; Cluster analysis; String (physics); Task (project management); Matching (statistics); Knowledge base; Information retrieval; Hierarchical clustering; Variety (cybernetics); Rank (graph theory); Data mining; Base (topology); Artificial intelligence; Mathematics","score_opus":0.005015511076475368,"score_gpt":0.23223203286063226,"score_spread":0.2272165217841569,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2184852068","genre_codex":"software","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.038204756,0.0028874031,0.28792918,0.0036939925,0.0016040517,0.001960731,0.07964496,0.4499404,0.13413452],"genre_scores_gemma":[0.14103776,0.0012931341,0.49095464,0.0020103457,0.00055602816,0.0018060047,0.3073205,0.012731311,0.04229029],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9968465,0.00073482894,0.00029849698,0.00076517975,0.001137239,0.00021776925],"domain_scores_gemma":[0.9944225,0.0009850716,0.00031279412,0.0015752484,0.0021956007,0.0005088056],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0048039653,0.0014701225,0.0013102064,0.0065522175,0.0027125787,0.004262847,0.0034744842,0.0016170702,0.021475682],"category_scores_gemma":[0.008643091,0.00072136306,0.0009947148,0.004402888,0.00060619647,0.008462079,0.0031477455,0.0026823555,0.018045167],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00093435636,0.00040377284,0.0026123205,0.0008823689,0.00023528963,0.0012302293,0.0006080895,0.0037543818,0.012507042,0.019917702,0.7226089,0.23430549],"study_design_scores_gemma":[0.00032192632,0.00016371602,0.0042440635,0.00015249813,0.0002854588,0.001263665,0.0005729494,0.07353622,0.030441593,0.014994742,0.8737428,0.00028044605],"about_ca_topic_score_codex":0.027986167,"about_ca_topic_score_gemma":0.02718052,"teacher_disagreement_score":0.027986167,"about_ca_system_score_codex":0.0033873583,"about_ca_system_score_gemma":0.0045919996,"threshold_uncertainty_score":0.07184327},"labels":[],"label_agreement":null},{"id":"W2184983269","doi":"","title":"FBK Participation in the RTE-7 Main Task","year":2011,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Paraphrase; Security token; Task (project management); Computer science; Context (archaeology); Set (abstract data type); Contrast (vision); Measure (data warehouse); Natural language processing; Artificial intelligence; Training set; Data mining; Geography; Engineering","score_opus":0.014639711783241766,"score_gpt":0.2783076216650061,"score_spread":0.26366790988176436,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2184983269","genre_codex":"empirical","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.32900384,0.0062358854,0.16282527,0.020264909,0.012817513,0.008001936,0.12263667,0.17986268,0.1583514],"genre_scores_gemma":[0.32354972,0.0008283353,0.18928765,0.0025536176,0.0018144421,0.0044968575,0.3213115,0.030710796,0.12544712],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.98117137,0.005076859,0.00088457507,0.0034472747,0.006477784,0.0029421581],"domain_scores_gemma":[0.9583351,0.008553639,0.00048827284,0.010615562,0.015971327,0.0060360553],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.027763968,0.003570655,0.0043565542,0.003027409,0.0027677643,0.005308553,0.0061305244,0.004391534,0.037763942],"category_scores_gemma":[0.0368795,0.0011845964,0.00204213,0.0019196294,0.0014468686,0.0061535924,0.006375648,0.0054504666,0.06581567],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0034245956,0.002429267,0.0019156013,0.0006745184,0.00026687962,0.0005941621,0.0013966552,0.004468257,0.02028441,0.0018670269,0.7133109,0.24936768],"study_design_scores_gemma":[0.0024557777,0.0033628293,0.018041074,0.0002907089,0.00021923883,0.0019655132,0.0029793244,0.095086046,0.08421293,0.0074750637,0.78324485,0.0006666649],"about_ca_topic_score_codex":0.012135525,"about_ca_topic_score_gemma":0.009945936,"teacher_disagreement_score":0.037763942,"about_ca_system_score_codex":0.0035210035,"about_ca_system_score_gemma":0.0043785144,"threshold_uncertainty_score":0.14683163},"labels":[],"label_agreement":null},{"id":"W2185701500","doi":"10.1162/tacl_a_00239","title":"Measuring Machine Translation Errors in New Domains","year":2013,"lang":"en","type":"article","venue":"Transactions of the Association for Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":77,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"Defense Advanced Research Projects Agency; National Science Foundation","keywords":"Computer science; Machine translation; Natural language processing; Domain (mathematical analysis); Phrase; Machine translation software usability; Artificial intelligence; Evaluation of machine translation; Porting; Example-based machine translation; Rule-based machine translation; Translation (biology); Programming language","score_opus":0.02489167998879905,"score_gpt":0.26812889941354556,"score_spread":0.24323721942474652,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2185701500","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.92027324,0.0005929952,0.07176959,0.0003876248,0.0000814173,0.00023462955,0.0005695323,0.0007559647,0.005335064],"genre_scores_gemma":[0.936502,0.00024878437,0.059591707,0.00017852691,0.00006242957,0.00027578973,0.0012982327,0.0003394027,0.00150309],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9780115,0.009939123,0.0024730016,0.0024273954,0.006657641,0.00049140834],"domain_scores_gemma":[0.880864,0.079896905,0.011043936,0.008937624,0.01810442,0.0011531292],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009788002,0.0008871593,0.00096669904,0.0029992072,0.00086016924,0.0020391566,0.0009784834,0.0013674254,0.0011994895],"category_scores_gemma":[0.0958329,0.00053220266,0.0005606752,0.0038442363,0.0014049634,0.0041180835,0.0025354526,0.0021827498,0.0008006678],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0026223315,0.0017838477,0.22179864,0.0017206036,0.0011087285,0.0009711884,0.009714782,0.1574432,0.14900646,0.0053068697,0.003571769,0.44495156],"study_design_scores_gemma":[0.00018303681,0.0031140747,0.33062586,0.00018262898,0.0004841894,0.0014092657,0.004379227,0.35233277,0.28383154,0.012633031,0.010490172,0.0003341504],"about_ca_topic_score_codex":0.0021494543,"about_ca_topic_score_gemma":0.0021593773,"teacher_disagreement_score":0.009788002,"about_ca_system_score_codex":0.0009975034,"about_ca_system_score_gemma":0.00071213336,"threshold_uncertainty_score":0.051764548},"labels":[],"label_agreement":null},{"id":"W2185791191","doi":"","title":"TransSearch: What are translators looking for?","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Productivity; Machine translation; Quality (philosophy); Production (economics); Artificial intelligence; Natural language processing","score_opus":0.04046044932977954,"score_gpt":0.29009445069077333,"score_spread":0.24963400136099378,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2185791191","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.46041316,0.03997894,0.21229391,0.10042732,0.0019298564,0.0009573442,0.029171564,0.02454318,0.13028473],"genre_scores_gemma":[0.77734363,0.008730461,0.12833863,0.006740288,0.0012074158,0.0004912524,0.017343398,0.0039759013,0.055829022],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9940655,0.0027925272,0.00045670042,0.0013322802,0.0009449237,0.0004079638],"domain_scores_gemma":[0.9854953,0.00652004,0.0017175649,0.0015426936,0.0035373026,0.0011870824],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005039395,0.0008912176,0.0012886046,0.004035256,0.0025022882,0.0056209476,0.0012804839,0.0027514386,0.023025323],"category_scores_gemma":[0.02406542,0.00059415394,0.00059428817,0.006176393,0.0012291711,0.01045193,0.0014374966,0.0017234662,0.020304922],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015352188,0.00030615326,0.0801435,0.0028307203,0.00014925294,0.0016696998,0.017570162,0.0012904435,0.03049101,0.012135053,0.15154517,0.7003336],"study_design_scores_gemma":[0.00021510852,0.000619722,0.07993648,0.001451329,0.000320464,0.009323553,0.061014235,0.033777446,0.035644144,0.044092737,0.7330589,0.00054577447],"about_ca_topic_score_codex":0.004206027,"about_ca_topic_score_gemma":0.0071376744,"teacher_disagreement_score":0.023025323,"about_ca_system_score_codex":0.001408435,"about_ca_system_score_gemma":0.0028941198,"threshold_uncertainty_score":0.07702744},"labels":[],"label_agreement":null},{"id":"W2185810805","doi":"","title":"University of Ottawa's Contribution to CLEF 2005, the CL-SR Track","year":2005,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Clef; Computer science; Information retrieval; Search engine indexing; Natural language processing; Weighting; Document retrieval; Task (project management); Artificial intelligence; Scheme (mathematics)","score_opus":0.006362797367864719,"score_gpt":0.2330211480157243,"score_spread":0.22665835064785958,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2185810805","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.067232944,0.046906225,0.1113237,0.10457776,0.061146673,0.005242306,0.2761033,0.0981472,0.22931986],"genre_scores_gemma":[0.07136466,0.005606765,0.1205197,0.008399275,0.0044151247,0.0014056074,0.39223352,0.014049267,0.38200614],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9899884,0.0028262516,0.000462923,0.002033356,0.0036477072,0.0010413532],"domain_scores_gemma":[0.97927815,0.002924288,0.00046106486,0.0038620192,0.009563113,0.0039113206],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012426898,0.0032888222,0.003809936,0.004891894,0.005664891,0.0065639666,0.003937301,0.0036754154,0.08988871],"category_scores_gemma":[0.017172894,0.001215462,0.001322659,0.004314221,0.0015646904,0.0044192784,0.003551623,0.0031223658,0.05008792],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021611022,0.000108835266,0.00038641482,0.00016232282,0.000051035557,0.0000837096,0.00008605211,0.0008075534,0.0017237172,0.000601158,0.9548531,0.040920027],"study_design_scores_gemma":[0.0004639721,0.00017835696,0.0030357528,0.00008234202,0.00006728095,0.00026525604,0.00025865465,0.008215448,0.004732342,0.0012645787,0.9812971,0.00013893885],"about_ca_topic_score_codex":0.36081645,"about_ca_topic_score_gemma":0.463512,"teacher_disagreement_score":0.36081645,"about_ca_system_score_codex":0.013396787,"about_ca_system_score_gemma":0.015400752,"threshold_uncertainty_score":0.7174325},"labels":[],"label_agreement":null},{"id":"W2186534410","doi":"","title":"Saarland University Spoken Language Systems Group at TAC KBP 2011","year":2011,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Cluster analysis; Entity linking; Task (project management); Focus (optics); Natural language processing; Cluster (spacecraft); Similarity (geometry); Population; Artificial intelligence; Knowledge base; Information retrieval; Medicine; Programming language; Image (mathematics); Engineering","score_opus":0.007755455015914662,"score_gpt":0.21281596388376603,"score_spread":0.20506050886785138,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2186534410","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1190377,0.008843304,0.4289665,0.022470666,0.0071498808,0.0019371262,0.04603747,0.067765296,0.29779205],"genre_scores_gemma":[0.17693146,0.0027834051,0.2711284,0.001445031,0.0009203941,0.00143013,0.086230166,0.0062519624,0.45287904],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99716836,0.00084437785,0.00014946843,0.0006995093,0.0008931549,0.00024515894],"domain_scores_gemma":[0.99446845,0.0012312521,0.00013506606,0.00086441287,0.0025883045,0.00071256957],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005995767,0.0009772866,0.0013033777,0.0015306125,0.0019469573,0.0041646045,0.0012109845,0.0011832059,0.06611086],"category_scores_gemma":[0.006311984,0.00054957793,0.00043644005,0.001813274,0.0005160118,0.005036515,0.0021394885,0.0020115871,0.0391437],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005956056,0.0006883896,0.0016153608,0.00031123107,0.00006002778,0.00041875447,0.0020166512,0.0020771772,0.022101955,0.013901409,0.43164605,0.5245673],"study_design_scores_gemma":[0.0003273661,0.00051862857,0.004947991,0.00015024879,0.0000917423,0.00045370555,0.0015714983,0.045464315,0.032937847,0.010995929,0.90239054,0.00015022386],"about_ca_topic_score_codex":0.011788219,"about_ca_topic_score_gemma":0.010876032,"teacher_disagreement_score":0.06611086,"about_ca_system_score_codex":0.0015326571,"about_ca_system_score_gemma":0.0036463316,"threshold_uncertainty_score":0.2211628},"labels":[],"label_agreement":null},{"id":"W2186676076","doi":"","title":"UBC Entity Linking at TAC-KBP 2013: random forests for high accuracy","year":2013,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Cluster analysis; Random forest; Task (project management); Artificial intelligence; Natural language processing; Generative grammar; F1 score; Reuse; Machine learning; Data mining; Biology","score_opus":0.007377326336720849,"score_gpt":0.26194726154569653,"score_spread":0.2545699352089757,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2186676076","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19290468,0.0041198293,0.4354928,0.0026493035,0.0030572906,0.0014067474,0.03325948,0.3066565,0.020453379],"genre_scores_gemma":[0.38806352,0.0004313256,0.50395423,0.0008088416,0.00051649753,0.00087015575,0.08214323,0.014066785,0.009145452],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9885854,0.004481259,0.00065463025,0.0024602541,0.0026583595,0.0011601226],"domain_scores_gemma":[0.9805053,0.008574196,0.00040456426,0.0050320444,0.004702898,0.00078106474],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.014193547,0.0033264838,0.0020512422,0.0034553523,0.0031659456,0.0029879967,0.0034106825,0.003829863,0.0089853015],"category_scores_gemma":[0.03239938,0.0012935769,0.0020227982,0.0041677817,0.00095967145,0.0048157386,0.002867749,0.004281954,0.013407487],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0025576232,0.0012342312,0.008428264,0.00078355684,0.00083034387,0.0008237434,0.0006318429,0.12177959,0.015482502,0.0036504497,0.35885158,0.48494637],"study_design_scores_gemma":[0.00052795355,0.0003001944,0.0060819094,0.00010197075,0.00019303462,0.0004982976,0.00026664857,0.90999126,0.03248469,0.012458672,0.0369045,0.00019080863],"about_ca_topic_score_codex":0.01886444,"about_ca_topic_score_gemma":0.024164818,"teacher_disagreement_score":0.01886444,"about_ca_system_score_codex":0.0016059384,"about_ca_system_score_gemma":0.0023128793,"threshold_uncertainty_score":0.075063586},"labels":[],"label_agreement":null},{"id":"W2186705392","doi":"","title":"Na¨ ive but effective NIL clustering baselines - CMCRC at TAC 2011","year":2011,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Cluster analysis; Computer science; Artificial intelligence","score_opus":0.009456469064637365,"score_gpt":0.24566998416756497,"score_spread":0.2362135151029276,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2186705392","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.40454906,0.015292733,0.27336895,0.005494719,0.0048191757,0.002041715,0.035610177,0.1908994,0.06792405],"genre_scores_gemma":[0.5532016,0.0008911623,0.30701298,0.0021900954,0.0007723386,0.0005491863,0.10057986,0.004769241,0.030033508],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9869941,0.0037831934,0.00082377414,0.0035596378,0.0035277365,0.0013116247],"domain_scores_gemma":[0.9882693,0.0022542442,0.00034754476,0.0038222107,0.0047156434,0.0005910732],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00799585,0.0031759765,0.002801124,0.0057306597,0.0043713986,0.0042684693,0.0054714927,0.00449413,0.0078648105],"category_scores_gemma":[0.01589905,0.00078493817,0.0014338619,0.004253959,0.001384759,0.0063925474,0.0048280903,0.00297664,0.011867913],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0035266103,0.0016080582,0.008076557,0.0013326752,0.0007636983,0.0006539357,0.000704605,0.044354778,0.030945774,0.005894432,0.30965623,0.5924826],"study_design_scores_gemma":[0.0010326632,0.0018734243,0.010430111,0.00019569714,0.00053932436,0.0015665605,0.0023710346,0.74246466,0.0941103,0.017638078,0.12737812,0.00039998506],"about_ca_topic_score_codex":0.029281965,"about_ca_topic_score_gemma":0.04484574,"teacher_disagreement_score":0.029281965,"about_ca_system_score_codex":0.002941228,"about_ca_system_score_gemma":0.0025379092,"threshold_uncertainty_score":0.05822307},"labels":[],"label_agreement":null},{"id":"W2186889148","doi":"","title":"Generated Abstracts for TAC 2011","year":2011,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Automatic summarization; Computer science; Natural language processing; Parsing; Focus (optics); Artificial intelligence; Metric (unit); Pyramid (geometry); Quality (philosophy); Linguistics; Information retrieval; Mathematics","score_opus":0.017009620848164486,"score_gpt":0.26190103849086627,"score_spread":0.24489141764270178,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2186889148","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023338182,0.002195071,0.13140516,0.0038806007,0.0043595578,0.0017369087,0.57458323,0.14656572,0.11193562],"genre_scores_gemma":[0.033590548,0.0003709152,0.077966355,0.00030973103,0.0003491904,0.0006654054,0.8447098,0.0065502464,0.035487775],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99721503,0.00075262424,0.0003257528,0.00040997044,0.0010698017,0.00022688443],"domain_scores_gemma":[0.9964412,0.00070018007,0.00022139796,0.00081593374,0.0015361324,0.00028513643],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023834307,0.0025580064,0.0011607015,0.0037930554,0.0015806125,0.0039000544,0.0019205046,0.0017273552,0.09683775],"category_scores_gemma":[0.007207375,0.0007481342,0.0016419394,0.003210027,0.00032253092,0.0019783615,0.001369253,0.0021885086,0.058074784],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007717418,0.00016808271,0.0004554807,0.0010432816,0.00010306664,0.00061643205,0.00023960261,0.0035112759,0.007040329,0.005638596,0.9009529,0.07945932],"study_design_scores_gemma":[0.00040887852,0.0002949716,0.00217034,0.00010034592,0.00014750501,0.000661708,0.00022312155,0.021490635,0.017679723,0.008729978,0.94796747,0.00012544951],"about_ca_topic_score_codex":0.005727829,"about_ca_topic_score_gemma":0.010211498,"teacher_disagreement_score":0.09683775,"about_ca_system_score_codex":0.0018077209,"about_ca_system_score_gemma":0.002181372,"threshold_uncertainty_score":0.32395458},"labels":[],"label_agreement":null},{"id":"W2187127363","doi":"","title":"Linguistic Resources for 2012 Knowledge Base Population Evaluations","year":2012,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"NIST; Computer science; Annotation; Knowledge base; Selection (genetic algorithm); Population; Information extraction; Resource (disambiguation); Entity linking; Track (disk drive); Information retrieval; Base (topology); Natural language processing; Data science; Artificial intelligence","score_opus":0.016945637261835286,"score_gpt":0.3279130043055564,"score_spread":0.31096736704372113,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2187127363","genre_codex":"methods","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14507788,0.0029208201,0.4023303,0.0071621756,0.0020694975,0.015641214,0.13779296,0.030858984,0.25614616],"genre_scores_gemma":[0.27107796,0.00088076823,0.4938522,0.0015767296,0.00030638324,0.01886481,0.1782364,0.0052484097,0.029956382],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9642731,0.01790312,0.0032390126,0.0019172278,0.011542845,0.0011246158],"domain_scores_gemma":[0.9331685,0.02547824,0.0015004326,0.0065643983,0.031901225,0.0013872016],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02699335,0.0012990816,0.0011116528,0.009594987,0.0037236398,0.005528688,0.0031463231,0.0021183062,0.030054238],"category_scores_gemma":[0.10840353,0.0008812732,0.001078824,0.0071170405,0.0009689669,0.0050659985,0.0046112114,0.00251852,0.01306123],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019493074,0.0016103749,0.005956251,0.0020263258,0.00016530183,0.0007599157,0.0036857736,0.017980777,0.008958279,0.03669365,0.28728682,0.63292724],"study_design_scores_gemma":[0.0011364948,0.0010415033,0.014574252,0.0013567451,0.0004493171,0.00072251714,0.005620081,0.09013504,0.045596708,0.036768276,0.802114,0.0004849444],"about_ca_topic_score_codex":0.015547267,"about_ca_topic_score_gemma":0.01603324,"teacher_disagreement_score":0.030054238,"about_ca_system_score_codex":0.0052464264,"about_ca_system_score_gemma":0.008656884,"threshold_uncertainty_score":0.14275616},"labels":[],"label_agreement":null},{"id":"W2187464037","doi":"","title":"Towards Event-Based Discourse Analysis of Biomedical Text","year":2013,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Open Text (Canada)","funders":"","keywords":"Rhetorical question; Computer science; Event (particle physics); Relation (database); Natural language processing; Interpretation (philosophy); Linguistics; Information retrieval; Artificial intelligence; Data mining","score_opus":0.009015518458927056,"score_gpt":0.30157972225326113,"score_spread":0.29256420379433407,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2187464037","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.050838605,0.0022535834,0.9307899,0.001480032,0.00015475371,0.0006736583,0.0062075555,0.0026525594,0.0049493],"genre_scores_gemma":[0.15384665,0.0010314077,0.8349931,0.00014647024,0.0001798149,0.00071889546,0.007045193,0.0002158491,0.0018226639],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99182135,0.0040582786,0.0009985982,0.0014938046,0.0013767651,0.00025119423],"domain_scores_gemma":[0.97182965,0.01869901,0.0038462102,0.0017011213,0.003408197,0.00051580434],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008299071,0.0011568966,0.0008607092,0.018161893,0.0014179791,0.006898116,0.0018872648,0.0015234555,0.0029470718],"category_scores_gemma":[0.026374143,0.00060367346,0.0015037698,0.010964277,0.0014547256,0.006062927,0.003198452,0.0021911738,0.001515647],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010043625,0.0006069863,0.018216945,0.0037820074,0.00041135956,0.0022047742,0.034519184,0.01913946,0.07688761,0.16646722,0.012581686,0.6641785],"study_design_scores_gemma":[0.00012360324,0.00024299388,0.021003846,0.00087089156,0.00032470422,0.0012589232,0.019179588,0.45268607,0.047385603,0.3389119,0.11772689,0.00028494664],"about_ca_topic_score_codex":0.003317234,"about_ca_topic_score_gemma":0.0027499213,"teacher_disagreement_score":0.018161893,"about_ca_system_score_codex":0.0018090103,"about_ca_system_score_gemma":0.0021164233,"threshold_uncertainty_score":0.04389018},"labels":[],"label_agreement":null},{"id":"W2187479749","doi":"10.19173/irrodl.v16i6.2145","title":"A MOOC on Approaches to Machine Translation","year":2015,"lang":"en","type":"article","venue":"The International Review of Research in Open and Distributed Learning","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"European Regional Development Fund; Agència de Gestió d'Ajuts Universitaris i de Recerca; Ministerio de Economía y Competitividad; Generalitat de Catalunya; European Commission","keywords":"Computer science; Machine translation; Translation (biology); Software engineering; Multimedia; Artificial intelligence","score_opus":0.3687970938527921,"score_gpt":0.4602998079630409,"score_spread":0.0915027141102488,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2187479749","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.030855322,0.0034660962,0.901056,0.0033815706,0.0013046955,0.0012243675,0.00026322127,0.002859021,0.055589642],"genre_scores_gemma":[0.25639796,0.0033407663,0.66127,0.0015651406,0.00059582497,0.0012817383,0.0009936293,0.0011690226,0.07338581],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99808633,0.00080994423,0.0000922739,0.0003196222,0.0005459105,0.00014587861],"domain_scores_gemma":[0.9975896,0.0010644327,0.0001420497,0.00038591205,0.0005411717,0.0002768934],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019035527,0.00090887025,0.0003987367,0.0010737798,0.0014481043,0.002717981,0.0014099476,0.0013080505,0.008214284],"category_scores_gemma":[0.007219775,0.0003493933,0.0006192063,0.0007798442,0.001774214,0.0017917815,0.0029317958,0.0012971287,0.002578471],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023002944,0.0005328952,0.0018551773,0.0011554232,0.00011993019,0.0002482022,0.0032535414,0.02285553,0.018078703,0.16784686,0.020918336,0.76290536],"study_design_scores_gemma":[0.00011089488,0.00091976696,0.0045591732,0.0007522435,0.00009518489,0.00063833914,0.0013945017,0.081619255,0.018895522,0.2190746,0.67180896,0.00013147345],"about_ca_topic_score_codex":0.0045537483,"about_ca_topic_score_gemma":0.0036005876,"teacher_disagreement_score":0.008214284,"about_ca_system_score_codex":0.0013929384,"about_ca_system_score_gemma":0.0020315063,"threshold_uncertainty_score":0.02747953},"labels":[],"label_agreement":null},{"id":"W2187967184","doi":"10.5539/ijel.v5n6p84","title":"Types and Features of Noun Phrase in Chinese Scholars’ Abstracts","year":2015,"lang":"en","type":"article","venue":"International Journal of English Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Adjective; Noun phrase; Linguistics; Noun; Determiner phrase; Phrase; Psychology; Computer science; Philosophy","score_opus":0.012094702986068689,"score_gpt":0.30695848987741753,"score_spread":0.29486378689134884,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2187967184","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9672538,0.0061482615,0.005019862,0.0007252721,0.0003360317,0.0002540175,0.003237982,0.00015684601,0.016867971],"genre_scores_gemma":[0.98388493,0.0032364493,0.006063426,0.00017409488,0.0001985323,0.00028531073,0.002642831,0.000106190106,0.0034081931],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99580437,0.0009517466,0.0011907757,0.00042334816,0.0014941986,0.00013564508],"domain_scores_gemma":[0.97755367,0.011452855,0.0050595053,0.00065991096,0.004591315,0.0006827366],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002912033,0.00052665273,0.00037400378,0.008182149,0.0010307512,0.0018410245,0.00046893538,0.00036865292,0.0021188557],"category_scores_gemma":[0.022615304,0.0001943065,0.00046289337,0.009446207,0.0010496251,0.0024846618,0.0012640962,0.00039287723,0.00058357016],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009989757,0.00019017445,0.30187848,0.012452621,0.00049176964,0.0066957017,0.15960518,0.00041466398,0.11141509,0.012193262,0.014895121,0.37876895],"study_design_scores_gemma":[0.00008553475,0.0004748354,0.7955089,0.0014980661,0.000613367,0.005760135,0.048779808,0.002279212,0.011972397,0.0061918786,0.12656444,0.00027142762],"about_ca_topic_score_codex":0.0038865344,"about_ca_topic_score_gemma":0.004983842,"teacher_disagreement_score":0.008182149,"about_ca_system_score_codex":0.0012927135,"about_ca_system_score_gemma":0.0017817138,"threshold_uncertainty_score":0.015400469},"labels":[],"label_agreement":null},{"id":"W2188260722","doi":"","title":"Summaries with SumUM and its Expansion for Document Understanding Conference (DUC 2002)","year":2002,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Domain (mathematical analysis); Adaptation (eye); Information retrieval; Artificial intelligence; Natural language processing; Mathematics; Psychology","score_opus":0.060278130272391185,"score_gpt":0.269045953592615,"score_spread":0.20876782332022384,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2188260722","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17033005,0.020595849,0.35576463,0.0054697157,0.0051290737,0.0029904786,0.10056407,0.28627366,0.05288255],"genre_scores_gemma":[0.16618295,0.0028979194,0.5808156,0.00083888345,0.00062277657,0.00126944,0.19743009,0.0061635035,0.043778803],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99680483,0.001232642,0.00046454652,0.0004753816,0.0008974553,0.00012516427],"domain_scores_gemma":[0.98973656,0.0029298544,0.00050953694,0.0017796914,0.004427534,0.0006167774],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003653913,0.0018262523,0.0013462905,0.0033315148,0.0010398568,0.002753656,0.0014381572,0.0011450965,0.0152605185],"category_scores_gemma":[0.015606208,0.00043202567,0.0009857477,0.0021130303,0.0003266481,0.003120446,0.002082282,0.001244717,0.00859956],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011796316,0.00030947698,0.0012344667,0.0022700054,0.0002168879,0.00035004318,0.0005348747,0.0045102644,0.023359245,0.0017372913,0.38786963,0.5764282],"study_design_scores_gemma":[0.00096635934,0.0019823054,0.0107328165,0.00036254234,0.000630131,0.0013846868,0.0009843572,0.09700686,0.113125645,0.004681914,0.76788634,0.0002560795],"about_ca_topic_score_codex":0.008928287,"about_ca_topic_score_gemma":0.013578342,"teacher_disagreement_score":0.0152605185,"about_ca_system_score_codex":0.0010551579,"about_ca_system_score_gemma":0.001416939,"threshold_uncertainty_score":0.051051497},"labels":[],"label_agreement":null},{"id":"W2188556664","doi":"10.1109/wimob.2015.7347988","title":"Microtext normalization using probably-phonetically-similar word discovery","year":2015,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Lakehead University","funders":"","keywords":"Spelling; Normalization (sociology); Computer science; Natural language processing; Rendering (computer graphics); Artificial intelligence; Speech recognition; Word (group theory); Pattern matching; Linguistics","score_opus":0.030832352929619774,"score_gpt":0.27863224947430376,"score_spread":0.24779989654468398,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2188556664","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04567664,0.00044917624,0.9295561,0.00039672037,0.00025731517,0.00022632639,0.0010832258,0.019227408,0.0031271325],"genre_scores_gemma":[0.26872426,0.00041658676,0.71545786,0.00028626536,0.00028889964,0.00024526147,0.004248458,0.0016826579,0.008649793],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99684453,0.00036921515,0.00037549846,0.0013160686,0.00089123106,0.00020343905],"domain_scores_gemma":[0.9940002,0.002138077,0.0007255021,0.0015437423,0.0014134793,0.00017898127],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016135535,0.0011828081,0.0018600879,0.0049072355,0.0014307437,0.0025503763,0.0023430432,0.0011689228,0.0065936716],"category_scores_gemma":[0.0098733185,0.0006267835,0.001515315,0.004115351,0.0011296964,0.0054634134,0.0028581822,0.0018499725,0.005760282],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004884188,0.0002102499,0.006489254,0.00028483756,0.00012454519,0.00038479982,0.00050527375,0.008527746,0.046488598,0.007937069,0.008591734,0.91996753],"study_design_scores_gemma":[0.00012351952,0.00028882272,0.009959786,0.000082820705,0.00029181887,0.0026958198,0.0011223257,0.6600242,0.20285717,0.07570058,0.046619277,0.00023390494],"about_ca_topic_score_codex":0.003001017,"about_ca_topic_score_gemma":0.005123187,"teacher_disagreement_score":0.0065936716,"about_ca_system_score_codex":0.000822705,"about_ca_system_score_gemma":0.0019093788,"threshold_uncertainty_score":0.02205801},"labels":[],"label_agreement":null},{"id":"W2188891439","doi":"","title":"Summarizing through sense concentration and Contextual Exploration rules: the CHORAL system at TAC 2009","year":2009,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Automatic summarization; Computer science; Task (project management); Set (abstract data type); Feature (linguistics); Natural language processing; Word (group theory); Choir; Artificial intelligence; Sense (electronics); Information retrieval; Linguistics; Engineering; Psychology","score_opus":0.012358840934423233,"score_gpt":0.26066245542116306,"score_spread":0.24830361448673982,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2188891439","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.34583715,0.0017378805,0.3221632,0.0030528195,0.00093490066,0.001791664,0.019975118,0.2815014,0.02300584],"genre_scores_gemma":[0.47147295,0.00032238086,0.45273197,0.0004991012,0.00036186934,0.0010932307,0.051472686,0.006576331,0.015469491],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99661607,0.0013798026,0.00032048876,0.0008584564,0.00067798444,0.00014709866],"domain_scores_gemma":[0.9922111,0.0034844065,0.00041940415,0.0015827694,0.001761907,0.0005403934],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004273728,0.0011634073,0.0011406221,0.0020953913,0.0015201407,0.002378677,0.0019109014,0.0013620439,0.0057583246],"category_scores_gemma":[0.013126244,0.00056576054,0.00051883585,0.0012801602,0.00068710453,0.0038426407,0.0023317137,0.0013296505,0.0038549916],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0038821755,0.0009931518,0.0044129537,0.001759686,0.0004373143,0.0009298983,0.006586139,0.014799983,0.052142404,0.0037468253,0.18919034,0.72111917],"study_design_scores_gemma":[0.0024159404,0.0026520973,0.014878784,0.0002083495,0.00060370716,0.0007206329,0.0060820254,0.5274605,0.1284984,0.014287152,0.30158186,0.0006105029],"about_ca_topic_score_codex":0.0060276086,"about_ca_topic_score_gemma":0.014457055,"teacher_disagreement_score":0.0060276086,"about_ca_system_score_codex":0.0008637683,"about_ca_system_score_gemma":0.0013921433,"threshold_uncertainty_score":0.022601902},"labels":[],"label_agreement":null},{"id":"W2189227176","doi":"","title":"HITS' Cross-lingual Entity Linking System at TAC 2011: One Model for All Languages","year":2011,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Entity linking; Natural language processing; Task (project management); Artificial intelligence; Context (archaeology); Knowledge base; Cluster analysis; Information retrieval","score_opus":0.02292574166370594,"score_gpt":0.29673203503062967,"score_spread":0.27380629336692375,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2189227176","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2579248,0.0016002513,0.43263716,0.0020476538,0.00065815286,0.0009461471,0.01851652,0.26200876,0.023660475],"genre_scores_gemma":[0.59748685,0.00050655333,0.30839965,0.00091700815,0.00015663987,0.0005514917,0.06356571,0.0075926688,0.02082344],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9971117,0.00074600434,0.00022324963,0.00097496173,0.00064545166,0.00029871654],"domain_scores_gemma":[0.99635184,0.0009020085,0.00017556763,0.0014152244,0.0008934148,0.00026201096],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0040398957,0.0018768858,0.0017388792,0.002332365,0.0019154869,0.0030770032,0.003225604,0.0027123208,0.008829868],"category_scores_gemma":[0.0076659615,0.0009734106,0.0021642952,0.0022477156,0.0005414573,0.00873137,0.0042555467,0.0025338093,0.008447884],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020464363,0.0016512866,0.017703028,0.0014001615,0.002017688,0.0021912805,0.0021943254,0.09549827,0.045712084,0.011618651,0.22090219,0.5970646],"study_design_scores_gemma":[0.00024384254,0.0008277283,0.009247588,0.00010458692,0.0007649098,0.0017474672,0.001321526,0.81882703,0.0619814,0.013035606,0.09146157,0.00043672806],"about_ca_topic_score_codex":0.018238368,"about_ca_topic_score_gemma":0.027152618,"teacher_disagreement_score":0.018238368,"about_ca_system_score_codex":0.00126612,"about_ca_system_score_gemma":0.0025929199,"threshold_uncertainty_score":0.03626442},"labels":[],"label_agreement":null},{"id":"W2193575455","doi":"","title":"Opgaveskrivning kort og godt","year":2009,"lang":"da","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"World Federation of Science Journalists","funders":"","keywords":"Political science","score_opus":0.013045933758231466,"score_gpt":0.2839635371152329,"score_spread":0.27091760335700144,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2193575455","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08821034,0.0071352273,0.37152833,0.014562184,0.010188432,0.00039902123,0.015845457,0.02338632,0.46874458],"genre_scores_gemma":[0.3937478,0.0048296154,0.24880047,0.0035793285,0.0013737204,0.00022600219,0.015921615,0.014241856,0.31727955],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9976616,0.0004436415,0.00015831618,0.00072520005,0.0007603381,0.00025092976],"domain_scores_gemma":[0.99855083,0.00044948002,0.00008447912,0.00036082923,0.0004479459,0.000106378015],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012905047,0.001154913,0.001012904,0.0012752679,0.0015228435,0.0061595915,0.0013234427,0.0016286465,0.07438458],"category_scores_gemma":[0.004766243,0.0008670527,0.001361422,0.0006949056,0.0018081521,0.0059466194,0.004440995,0.0034892655,0.039330505],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00083250005,0.00040446888,0.0031630686,0.0018592335,0.00021214827,0.0026163147,0.0065682437,0.0037262575,0.03368989,0.12588984,0.15609778,0.6649403],"study_design_scores_gemma":[0.000058873025,0.00013588589,0.0016200494,0.0004965349,0.00015512796,0.0016830609,0.002685182,0.0052739712,0.021652441,0.084752366,0.8813523,0.00013431968],"about_ca_topic_score_codex":0.0031035584,"about_ca_topic_score_gemma":0.0041509815,"teacher_disagreement_score":0.07438458,"about_ca_system_score_codex":0.0008102703,"about_ca_system_score_gemma":0.001591942,"threshold_uncertainty_score":0.24884117},"labels":[],"label_agreement":null},{"id":"W2200913422","doi":"10.33011/lilt.v9i.1321","title":"Frege in Space: A Program for Compositional Distributional Semantics","year":2014,"lang":"en","type":"article","venue":"Linguistic Issues in Language Technology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":245,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Atomic Energy of Canada Limited","keywords":"Principle of compositionality; Meaning (existential); Computer science; Semantics (computer science); Distributional semantics; Lexicon; Lexicology; Linguistics; Syntax; Formal semantics (linguistics); Computational semantics; Lexical semantics; Natural language processing; Cognitive semantics; Artificial intelligence; Space (punctuation); Function (biology); Operational semantics; Lexical item; Programming language; Cognition; Epistemology; Psychology; Philosophy","score_opus":0.005157375645620031,"score_gpt":0.31247372378189037,"score_spread":0.30731634813627035,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2200913422","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0013649939,0.00006410951,0.98886544,0.00032354082,0.000050780232,0.00005170939,0.00021318422,0.0040743747,0.00499194],"genre_scores_gemma":[0.06729056,0.00023603046,0.92062175,0.00032409528,0.00013127041,0.00045845544,0.0006418788,0.003126602,0.0071693095],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99883014,0.00046224645,0.00007101303,0.0002782278,0.0002633,0.0000951745],"domain_scores_gemma":[0.99841905,0.0011792061,0.000049924824,0.00018323881,0.00012525916,0.000043457254],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027771469,0.0011221439,0.0008740491,0.002436698,0.0017778075,0.003280514,0.0021768888,0.0014728697,0.02430371],"category_scores_gemma":[0.008366975,0.00076527154,0.0028792117,0.0018276185,0.0031677508,0.007947937,0.00470953,0.002429848,0.0044736755],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008765875,0.000038967508,0.00036128925,0.00015357428,0.000038554204,0.00014271718,0.00081111345,0.0115882335,0.00081008306,0.9061144,0.00917208,0.07068125],"study_design_scores_gemma":[0.00003324704,0.000017003484,0.00006825338,0.00004579401,0.00001723505,0.000086048465,0.00015501639,0.079719044,0.0010905195,0.87708825,0.041656982,0.000022514574],"about_ca_topic_score_codex":0.0025990163,"about_ca_topic_score_gemma":0.0042625493,"teacher_disagreement_score":0.02430371,"about_ca_system_score_codex":0.0014510778,"about_ca_system_score_gemma":0.0014329206,"threshold_uncertainty_score":0.081304014},"labels":[],"label_agreement":null},{"id":"W2205402206","doi":"10.1007/978-3-642-54906-9_37","title":"An Investigation on the Influence of Genres and Textual Organisation on the Use of Discourse Relations","year":2014,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Notice; Newspaper; Linguistics; Natural language processing; Genre analysis; Sociology; Political science; Media studies","score_opus":0.029410020510172784,"score_gpt":0.26562836702912523,"score_spread":0.23621834651895246,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2205402206","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.985415,0.0008060943,0.0010839584,0.00014940619,0.000022340178,0.000054643082,0.00033224004,0.000048254507,0.012088124],"genre_scores_gemma":[0.99500847,0.00041406878,0.0018209403,0.000031887164,0.00003759556,0.00003842235,0.00037989297,0.00008398257,0.002184645],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.995329,0.003103564,0.00024066442,0.0004253069,0.0007423811,0.00015907934],"domain_scores_gemma":[0.8454857,0.14121729,0.00575955,0.0024177616,0.0034093384,0.0017104209],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004895391,0.00032780893,0.0003789183,0.0023186645,0.0008518614,0.0041242484,0.0005473329,0.00042828542,0.005791786],"category_scores_gemma":[0.05386444,0.00030788773,0.00036127583,0.0022018054,0.0008071174,0.0019983058,0.0010606474,0.00075004157,0.0009320874],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.005924872,0.0020612192,0.41786674,0.0022744706,0.00041930837,0.0014798979,0.08989622,0.0011250527,0.12790918,0.0068510524,0.0038016615,0.34039032],"study_design_scores_gemma":[0.00017618577,0.0021109811,0.9048689,0.00040844738,0.00072109443,0.0016678524,0.033093642,0.010630059,0.024581071,0.0043487437,0.017267298,0.00012569127],"about_ca_topic_score_codex":0.0021520108,"about_ca_topic_score_gemma":0.0022128378,"teacher_disagreement_score":0.005791786,"about_ca_system_score_codex":0.00059955043,"about_ca_system_score_gemma":0.00052318367,"threshold_uncertainty_score":0.025889575},"labels":[],"label_agreement":null},{"id":"W2209426633","doi":"","title":"Port4NooJ: Portuguese Linguistic Module and Bilingual Resources for Machine Translation","year":2008,"lang":"en","type":"article","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Fundação para a Ciência e a Tecnologia; Division of Mathematical Sciences; York University","keywords":"Machine translation; Linguistics; Computer science; Portuguese; Natural language processing; Artificial intelligence; Brazilian Portuguese; Translation (biology); Philosophy","score_opus":0.015852063331014264,"score_gpt":0.24372891084207612,"score_spread":0.22787684751106185,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2209426633","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03204581,0.0013560125,0.62857366,0.0012572034,0.0009789491,0.0016078589,0.06658365,0.16419694,0.103399955],"genre_scores_gemma":[0.16879806,0.0010318274,0.59938467,0.00062135415,0.00042534948,0.0014010968,0.14197998,0.042406715,0.043951035],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9983525,0.0006323642,0.00022280193,0.00036762268,0.0002498981,0.0001747857],"domain_scores_gemma":[0.9978644,0.0006889247,0.000117786105,0.00066267105,0.0005127479,0.00015357576],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019350501,0.0021354326,0.0012131602,0.0034070816,0.0017846803,0.0033291408,0.0020634984,0.0012942682,0.06543859],"category_scores_gemma":[0.007718636,0.0012283834,0.0012089135,0.0038175303,0.0006223856,0.0050303824,0.003216261,0.0013515166,0.03408871],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0033412848,0.00051067374,0.0016989405,0.0031773825,0.00021106798,0.0013433818,0.0016690101,0.005238705,0.037549563,0.047936797,0.26152444,0.63579875],"study_design_scores_gemma":[0.00072360004,0.0003843436,0.003185005,0.0006853822,0.00034756475,0.0012372645,0.00069316244,0.058377232,0.09690821,0.04388533,0.7932825,0.00029040512],"about_ca_topic_score_codex":0.005738318,"about_ca_topic_score_gemma":0.0059906906,"teacher_disagreement_score":0.06543859,"about_ca_system_score_codex":0.00095290504,"about_ca_system_score_gemma":0.0027928597,"threshold_uncertainty_score":0.21891391},"labels":[],"label_agreement":null},{"id":"W2215310981","doi":"","title":"Fuzzy Coreference Resolution for Summarization","year":2003,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":39,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Coreference; Automatic summarization; Computer science; Resolution (logic); Artificial intelligence; Natural language processing; Task (project management); Fuzzy logic; Word (group theory); Noun; Linguistics","score_opus":0.023668400337138653,"score_gpt":0.27660020599297225,"score_spread":0.2529318056558336,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2215310981","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0049706004,0.00091634964,0.99094266,0.00018602317,0.000046850386,0.000083401486,0.00010017488,0.0009821574,0.0017717849],"genre_scores_gemma":[0.12898509,0.0006147513,0.86664045,0.00016281864,0.00011380989,0.00018436454,0.00056004204,0.00016848989,0.00257018],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99604154,0.0017221947,0.0002558398,0.0006693039,0.0011685168,0.00014263144],"domain_scores_gemma":[0.996725,0.0017310788,0.00023997221,0.0005137483,0.0007315107,0.000058536454],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034125166,0.00087654864,0.0010756298,0.0028892183,0.0015097132,0.0016282053,0.0019040131,0.0016314619,0.0040949946],"category_scores_gemma":[0.009284047,0.00040224107,0.0009936814,0.0024533407,0.0011641945,0.0028961655,0.0016112017,0.0015458114,0.001743139],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00031686688,0.000096675256,0.00046659017,0.00087967806,0.00021916558,0.00029646058,0.0014290409,0.07145125,0.052542534,0.13086456,0.014932631,0.7265046],"study_design_scores_gemma":[0.00006848816,0.00019241878,0.0006888975,0.00011146492,0.00016959342,0.00048534796,0.00033896792,0.6953162,0.05432243,0.21211512,0.036073882,0.000117146585],"about_ca_topic_score_codex":0.0015418057,"about_ca_topic_score_gemma":0.00185871,"teacher_disagreement_score":0.0040949946,"about_ca_system_score_codex":0.0011252374,"about_ca_system_score_gemma":0.00073597673,"threshold_uncertainty_score":0.018047333},"labels":[],"label_agreement":null},{"id":"W2218357198","doi":"","title":"On the Existence of Small Clauses in Japanese","year":2012,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Linguistics; Philosophy","score_opus":0.035563956883188926,"score_gpt":0.2776097059029013,"score_spread":0.24204574901971238,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2218357198","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.82263273,0.0034136702,0.13790444,0.0031073866,0.0001890492,0.00011620966,0.0009666519,0.00033810022,0.031331755],"genre_scores_gemma":[0.9723528,0.0006500121,0.023822233,0.00022691078,0.000104056984,0.000041311472,0.0006347741,0.000108336586,0.0020595032],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99915147,0.00022818774,0.0001195075,0.00026180147,0.00011959301,0.000119545344],"domain_scores_gemma":[0.98450863,0.012506098,0.00082503376,0.00063116226,0.0012071581,0.00032184637],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016052398,0.000347786,0.00050817756,0.0016933793,0.0026671865,0.0018950603,0.001074407,0.0009799943,0.0055358815],"category_scores_gemma":[0.00938791,0.00089640915,0.0006395162,0.0017744217,0.0029272037,0.0077189207,0.0020241588,0.0013090232,0.0002827486],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013073608,0.00022911027,0.042617504,0.0016414336,0.00013761707,0.005536158,0.024459723,0.003847951,0.030015837,0.79687923,0.0071365805,0.08619153],"study_design_scores_gemma":[0.00028875726,0.0004533886,0.08520253,0.00057192205,0.0018001792,0.003399599,0.016662344,0.06609986,0.031559058,0.7535041,0.040162597,0.00029571602],"about_ca_topic_score_codex":0.009465698,"about_ca_topic_score_gemma":0.011703792,"teacher_disagreement_score":0.009465698,"about_ca_system_score_codex":0.00081581785,"about_ca_system_score_gemma":0.00094601774,"threshold_uncertainty_score":0.01882118},"labels":[],"label_agreement":null},{"id":"W2220593735","doi":"10.22230/src.2014v5n2a159","title":"Documenting Subjective Interpretations of Illustrated Book Covers for Lewis Carroll’s Alice’s Adventures in Wonderland","year":2014,"lang":"en","type":"article","venue":"Scholarly and Research Communication","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"University of British Columbia; Mount Royal University","funders":"","keywords":"Adventure; Alice (programming language); Interpretation (philosophy); Focus (optics); Computer science; Test (biology); Order (exchange); Psychology; Art history; Artificial intelligence; History; Programming language","score_opus":0.02888699265777485,"score_gpt":0.3663856584903031,"score_spread":0.33749866583252824,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2220593735","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8488123,0.00031927606,0.08093259,0.00066908426,0.00007473814,0.00019839886,0.0012344916,0.00036829757,0.0673908],"genre_scores_gemma":[0.9660523,0.00009221404,0.027738085,0.00003232211,0.000015038805,0.000086198874,0.00073681766,0.000120474695,0.00512649],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.99824893,0.0008062776,0.00010003549,0.00026512088,0.00048454455,0.000094992756],"domain_scores_gemma":[0.9901894,0.0059172674,0.0012099219,0.0008498321,0.0015075784,0.00032603325],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019076739,0.00034312284,0.0001743346,0.003242448,0.0010597126,0.0032050766,0.00046120168,0.0005410381,0.0051200553],"category_scores_gemma":[0.012964173,0.00022674917,0.00019736521,0.0021398785,0.0023384297,0.00399441,0.0021558162,0.00085035415,0.0008121108],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007014695,0.00012182646,0.04634155,0.0008803392,0.000047691592,0.0017737431,0.52602595,0.001321053,0.04031496,0.10268548,0.008931688,0.27085423],"study_design_scores_gemma":[0.00004728852,0.00026698888,0.15054327,0.0005150111,0.00007269569,0.0014786784,0.5596181,0.015606249,0.030913547,0.065694705,0.17502889,0.00021454315],"about_ca_topic_score_codex":0.0030893805,"about_ca_topic_score_gemma":0.011019534,"teacher_disagreement_score":0.0051200553,"about_ca_system_score_codex":0.0017313459,"about_ca_system_score_gemma":0.00060809765,"threshold_uncertainty_score":0.017128348},"labels":[],"label_agreement":null},{"id":"W2221532354","doi":"","title":"Surmonter l'interférence culturelle et linguistique à l'aide de CALL","year":2006,"lang":"fr","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université TÉLUQ","funders":"","keywords":"Computer science; Linguistics; Process (computing); Cognitive science; Artificial intelligence; Psychology; Philosophy; Programming language","score_opus":0.016712321925943966,"score_gpt":0.26651480990857884,"score_spread":0.24980248798263488,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2221532354","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.38860327,0.0049866275,0.425867,0.013862215,0.0009472635,0.00020220672,0.0001230653,0.0019271327,0.1634812],"genre_scores_gemma":[0.90622026,0.0017880817,0.070618436,0.0012432989,0.00036748426,0.00010226756,0.00011218876,0.00021738012,0.01933063],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99289966,0.0031058928,0.00023845739,0.00075309124,0.0025136413,0.0004891652],"domain_scores_gemma":[0.9937729,0.0036042163,0.000467716,0.00066907,0.0011992934,0.0002868778],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004196846,0.0008661357,0.0005374528,0.0007250824,0.0018054199,0.0063913544,0.00096797873,0.0020315854,0.0040962007],"category_scores_gemma":[0.011920648,0.0003480681,0.00058750465,0.00063424633,0.0028984768,0.0047627026,0.004848648,0.0025802574,0.0013043532],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000627567,0.0004838966,0.028636752,0.001115188,0.00021930775,0.0016207177,0.029343545,0.008506823,0.087061204,0.2368851,0.0077753654,0.59772444],"study_design_scores_gemma":[0.00017353079,0.0010213143,0.050900098,0.00058714114,0.00060168933,0.0061285314,0.03149263,0.12665325,0.15413049,0.1196019,0.5081461,0.00056330353],"about_ca_topic_score_codex":0.010020067,"about_ca_topic_score_gemma":0.008642135,"teacher_disagreement_score":0.010020067,"about_ca_system_score_codex":0.0016465833,"about_ca_system_score_gemma":0.0026229213,"threshold_uncertainty_score":0.02219534},"labels":[],"label_agreement":null},{"id":"W2222627577","doi":"","title":"A probabilistic model of early argument structure acquisition","year":2008,"lang":"en","type":"dissertation","venue":"eScholarship (California Digital Library)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"University of Pennsylvania","keywords":"Computer science; Artificial intelligence; Natural language processing; Probabilistic logic; Feature (linguistics); Verb; Natural language; Argument (complex analysis); Object language; Language acquisition; Linguistics","score_opus":0.010703370443055823,"score_gpt":0.22684718007755367,"score_spread":0.21614380963449784,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2222627577","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.050219454,0.00024717333,0.94054735,0.0011836,0.000021393407,0.000060249175,0.00024935274,0.0003386149,0.007132819],"genre_scores_gemma":[0.7327168,0.00071484386,0.2519885,0.00028601568,0.000095646865,0.00039214166,0.00066795235,0.00022758271,0.012910563],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99842334,0.00050351385,0.00007628051,0.0004224146,0.00040852706,0.0001659684],"domain_scores_gemma":[0.9916015,0.0057747494,0.0007350149,0.0009177852,0.0006781343,0.00029277973],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035283575,0.0005543153,0.0010026557,0.0020323473,0.0007927121,0.0026174297,0.0032999215,0.0020972765,0.0052540894],"category_scores_gemma":[0.01714539,0.0010544478,0.0017322864,0.0017873292,0.002583691,0.009188231,0.0019044672,0.0031237316,0.0012804308],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001082752,0.00011031496,0.0065711765,0.00013242467,0.000089851856,0.0003874837,0.0008034448,0.2551112,0.0038284687,0.68008995,0.0021665927,0.050600905],"study_design_scores_gemma":[0.000016352978,0.000031641393,0.0011623446,0.000014411784,0.000018146313,0.00022706017,0.000034015447,0.6936744,0.00042488097,0.30325896,0.001113133,0.000024623476],"about_ca_topic_score_codex":0.002886416,"about_ca_topic_score_gemma":0.002632832,"teacher_disagreement_score":0.0052540894,"about_ca_system_score_codex":0.0016890093,"about_ca_system_score_gemma":0.0011191,"threshold_uncertainty_score":0.01865995},"labels":[],"label_agreement":null},{"id":"W2224819922","doi":"","title":"Segmentation des corpus chronologiques : 143 ans de discours gouvernemental au Québec","year":2010,"lang":"fr","type":"article","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Humanities; Political science; Art","score_opus":0.01085560157810038,"score_gpt":0.2507190383187914,"score_spread":0.239863436740691,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2224819922","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.591586,0.010609687,0.027036292,0.0015326713,0.00049422606,0.00072419085,0.28753835,0.0017879466,0.07869069],"genre_scores_gemma":[0.6759805,0.003461105,0.034797244,0.00022390806,0.00018047041,0.0008742236,0.21630417,0.0005637553,0.0676146],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993389,0.000081683844,0.000064934626,0.0002039454,0.00022109642,0.000089430934],"domain_scores_gemma":[0.9970578,0.00062081974,0.00019991393,0.00013218368,0.0017916777,0.00019768211],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00070723484,0.0003960368,0.0003637603,0.007872694,0.0019930948,0.0012317847,0.0005019801,0.00045336675,0.008386272],"category_scores_gemma":[0.0030093216,0.00024314683,0.0002505171,0.009281854,0.0006901655,0.0003916757,0.00041273297,0.00041106011,0.0015487162],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015355819,0.00022176579,0.11261998,0.003232285,0.00024282894,0.0030629039,0.031003445,0.008818964,0.049187675,0.015735017,0.2643132,0.5100264],"study_design_scores_gemma":[0.00004457933,0.000049481623,0.49888223,0.00025339265,0.00006588372,0.00040277006,0.004532006,0.0040573166,0.00522055,0.00051707233,0.48591644,0.000058348433],"about_ca_topic_score_codex":0.91524523,"about_ca_topic_score_gemma":0.9625329,"teacher_disagreement_score":0.084754765,"about_ca_system_score_codex":0.008765758,"about_ca_system_score_gemma":0.013414321,"threshold_uncertainty_score":0.17050773},"labels":[],"label_agreement":null},{"id":"W2224941941","doi":"","title":"LexisNexis Skills Series: Drafting, 2nd Edition [Book Review]","year":2009,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Series (stratigraphy); Computer science; History; Geology","score_opus":0.005547453191916548,"score_gpt":0.26913268403917917,"score_spread":0.2635852308472626,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2224941941","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0006725943,0.40407124,0.005355064,0.016478779,0.028001957,0.0003376414,0.003380406,0.001656106,0.5400463],"genre_scores_gemma":[0.0023060278,0.24475765,0.004203586,0.0035058875,0.0046821125,0.0002630636,0.0042710574,0.00048877095,0.7355219],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9990262,0.00008789073,0.000090051835,0.00009065854,0.00065415737,0.00005110999],"domain_scores_gemma":[0.9971058,0.000751647,0.00028356173,0.00016279265,0.0014627451,0.00023346987],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010422465,0.0010202826,0.0012540264,0.0053783604,0.0008110591,0.004155486,0.00132304,0.00118447,0.17894207],"category_scores_gemma":[0.0062498804,0.00052167085,0.0005475362,0.006838296,0.0008941516,0.0040825987,0.0010521219,0.0019771191,0.14859363],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000004788199,0.000006401101,0.000025423959,0.00037196768,0.0000023893954,0.000016004715,0.000021843494,0.000040753213,0.00010817559,0.0013341989,0.8886801,0.10938798],"study_design_scores_gemma":[0.000002320142,0.000006577425,0.00017201413,0.00040011384,0.0000027527126,0.00008930096,0.000019528698,0.000017441156,0.000055600838,0.00064806384,0.9985825,0.0000037392324],"about_ca_topic_score_codex":0.005902624,"about_ca_topic_score_gemma":0.01109318,"teacher_disagreement_score":0.17894207,"about_ca_system_score_codex":0.0018827755,"about_ca_system_score_gemma":0.0053435676,"threshold_uncertainty_score":0.5986209},"labels":[],"label_agreement":null},{"id":"W2227591714","doi":"","title":"Tag Generalization For Facet-Based Search","year":2013,"lang":"en","type":"article","venue":"Library and Archives Canada (Government of Canada)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Generalization; Computer science; Facet (psychology); Artificial intelligence; Mathematics; Psychology","score_opus":0.004339373332833002,"score_gpt":0.16598899010582496,"score_spread":0.16164961677299197,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2227591714","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015285686,0.0010395643,0.96477413,0.00022075092,0.00005940003,0.00046304334,0.0018042778,0.009626735,0.00672634],"genre_scores_gemma":[0.19904195,0.00075660745,0.7803119,0.00039840475,0.00007546282,0.00042659152,0.009535375,0.000903192,0.008550638],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99803954,0.0004166532,0.00017515468,0.00049307756,0.00070910424,0.00016650074],"domain_scores_gemma":[0.99712175,0.00091728265,0.00016133531,0.0013619914,0.000351625,0.00008606889],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016609408,0.0009118915,0.00146103,0.003835813,0.0011758236,0.001652277,0.0022601967,0.0013028677,0.008203492],"category_scores_gemma":[0.005911234,0.00047216823,0.0018846896,0.004750955,0.0008635704,0.0039277007,0.002528666,0.0010068393,0.0042091436],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00038241502,0.00033006028,0.0052747536,0.000616777,0.00022117989,0.00027287257,0.00070577435,0.068145566,0.013592849,0.05739179,0.029935734,0.8231302],"study_design_scores_gemma":[0.000042448493,0.00010800133,0.0015173315,0.00008330612,0.00007402761,0.00043291517,0.00023714828,0.8318769,0.0095342165,0.115702435,0.040324952,0.0000663588],"about_ca_topic_score_codex":0.015961945,"about_ca_topic_score_gemma":0.028239818,"teacher_disagreement_score":0.015961945,"about_ca_system_score_codex":0.0014420418,"about_ca_system_score_gemma":0.0013462533,"threshold_uncertainty_score":0.031738043},"labels":[],"label_agreement":null},{"id":"W2229833550","doi":"10.18653/v1/n16-1101","title":"Multi-Way, Multilingual Neural Machine Translation with a Shared Attention Mechanism","year":2016,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":119,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; Canadian Institute for Advanced Research","funders":"Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs; Samsung; Compute Canada; Türkiye Bilimsel ve Teknolojik Araştırma Kurumu; Canadian Institute for Advanced Research","keywords":"Machine translation; Computer science; Translation (biology); Mechanism (biology); Artificial intelligence; Natural language processing; Language model","score_opus":0.03113405826540071,"score_gpt":0.2980237055648754,"score_spread":0.2668896472994747,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2229833550","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.047919262,0.00059932104,0.94091845,0.0005808456,0.00019374819,0.00005675868,0.00027318287,0.0036490671,0.0058092894],"genre_scores_gemma":[0.66158915,0.00035448678,0.32699904,0.0004400504,0.00013759577,0.00016921519,0.0010747701,0.00030556752,0.008930196],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992656,0.00025215093,0.00004324037,0.00023374853,0.00012904403,0.0000761866],"domain_scores_gemma":[0.99931455,0.0001668205,0.00007078297,0.00023267439,0.00017420757,0.000041062496],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009917917,0.0008701642,0.0006469899,0.00048455916,0.000488079,0.0010523695,0.0014361537,0.001160408,0.0032005126],"category_scores_gemma":[0.0024211844,0.00037903787,0.00091510185,0.00096618064,0.00055038993,0.0022970266,0.0018729361,0.0014314763,0.0015586235],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00054769154,0.0003881154,0.0024950902,0.00040198286,0.0005518103,0.00057518436,0.0004738856,0.31195316,0.052693795,0.03344184,0.012925345,0.58355206],"study_design_scores_gemma":[0.000031757965,0.00012838593,0.00047404866,0.000014728696,0.00007387674,0.00014600126,0.000031605232,0.9650648,0.014285608,0.015461216,0.0042584143,0.00002956565],"about_ca_topic_score_codex":0.0027195055,"about_ca_topic_score_gemma":0.005556566,"teacher_disagreement_score":0.0032005126,"about_ca_system_score_codex":0.00052970333,"about_ca_system_score_gemma":0.0009339094,"threshold_uncertainty_score":0.010706723},"labels":[],"label_agreement":null},{"id":"W2235881396","doi":"","title":"Reference grammars for speakers of minority languages","year":2012,"lang":"en","type":"article","venue":"ScholarSpace (University of Hawaii at Manoa)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Rule-based machine translation; Computer science; Linguistics; Grammar; Natural language processing; Artificial intelligence; Phrase structure grammar; Spoken language; Context-free grammar","score_opus":0.01790095041037173,"score_gpt":0.25234709245877046,"score_spread":0.23444614204839873,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2235881396","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.092170484,0.00045806865,0.7936249,0.0017384562,0.0001474938,0.00019068904,0.0008507057,0.0026840323,0.108135164],"genre_scores_gemma":[0.5929979,0.0004159012,0.35149318,0.00035479522,0.00010707156,0.00026280174,0.0016548873,0.0013337788,0.051379662],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.998582,0.00056608993,0.00013360666,0.00024034135,0.00039515397,0.000082901475],"domain_scores_gemma":[0.9962603,0.0016230141,0.00027770523,0.0010781224,0.00065128575,0.00010969548],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014616959,0.0002421601,0.00030831248,0.0013224417,0.0012681029,0.0025064144,0.0010708756,0.00088931964,0.009597438],"category_scores_gemma":[0.006683519,0.00019142199,0.00042975481,0.0015648014,0.0017550263,0.00311674,0.0018783422,0.0008753622,0.0019251838],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000043470074,0.00002356187,0.0026618203,0.000096321055,0.000009622011,0.0005037246,0.011836163,0.0037887932,0.0048296493,0.88136166,0.00822402,0.08662128],"study_design_scores_gemma":[0.0000318431,0.000065149965,0.0027430875,0.000191438,0.00003569671,0.0012749955,0.006276819,0.043786824,0.009206972,0.4779474,0.4583624,0.00007734226],"about_ca_topic_score_codex":0.00816411,"about_ca_topic_score_gemma":0.015193243,"teacher_disagreement_score":0.009597438,"about_ca_system_score_codex":0.0018006825,"about_ca_system_score_gemma":0.0018094262,"threshold_uncertainty_score":0.032106698},"labels":[],"label_agreement":null},{"id":"W2238944463","doi":"10.6084/m9.figshare.1293600","title":"#MLA5 Twitter Archive, 8-11 January 2015","year":2015,"lang":"en","type":"dataset","venue":"City Research Online (City University London)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Thursday; Computer science; Convention; World Wide Web; Information retrieval; Database; Library science; Political science","score_opus":0.07967748673015654,"score_gpt":0.36653765044196335,"score_spread":0.2868601637118068,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2238944463","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00038293557,0.00006694208,0.00013009216,0.00011197097,0.000121447745,0.000029256347,0.99492043,0.0008295455,0.0034074055],"genre_scores_gemma":[0.00088939996,0.00006687028,0.00032195562,0.00006567371,0.000025635916,0.000100428224,0.9943198,0.00021454027,0.003995773],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9989285,0.000109081295,0.00015815692,0.00020081496,0.00042216026,0.00018127715],"domain_scores_gemma":[0.9975356,0.0002693849,0.00020054758,0.0005873922,0.001123551,0.00028358688],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008416473,0.0010855461,0.0008893302,0.0038051126,0.0012735218,0.0030525997,0.0011548542,0.0010419665,0.08174304],"category_scores_gemma":[0.00610392,0.0005266216,0.00057973963,0.0049558827,0.00031678402,0.002497262,0.0025359085,0.0011185437,0.19282909],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000038617396,0.0000056779404,0.00055953476,0.00017777493,0.0000057887187,0.000015524276,0.00006034279,0.00006246536,0.0001624926,0.00032063728,0.9952282,0.0033629863],"study_design_scores_gemma":[0.00002904918,0.0000126696295,0.006181917,0.0001683048,0.000008087803,0.000024333844,0.00024010001,0.00032570495,0.00044725195,0.0004967663,0.9920397,0.000026153151],"about_ca_topic_score_codex":0.051483754,"about_ca_topic_score_gemma":0.096888535,"teacher_disagreement_score":0.08174304,"about_ca_system_score_codex":0.0017205554,"about_ca_system_score_gemma":0.002370695,"threshold_uncertainty_score":0.2734577},"labels":[],"label_agreement":null},{"id":"W2240814959","doi":"","title":"From Wordpress CMS to Wordpress LMS: A brilliant Idea","year":2011,"lang":"en","type":"article","venue":"EdMedia: World Conference on Educational Media and Technology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"World Wide Web; Computer science; Multimedia","score_opus":0.03682257098635125,"score_gpt":0.2775202108058185,"score_spread":0.24069763981946726,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2240814959","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013715131,0.0027122637,0.75231826,0.1050309,0.010286987,0.0006834258,0.0016813483,0.024763377,0.08880829],"genre_scores_gemma":[0.27822235,0.005491855,0.49473366,0.02056483,0.007099709,0.0017045235,0.0029396988,0.01662406,0.17261934],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99334395,0.0023516952,0.00031317925,0.0012197037,0.0023641284,0.0004072732],"domain_scores_gemma":[0.9804135,0.005027151,0.0003095766,0.007922045,0.0052141165,0.0011135414],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007111533,0.001254405,0.0010748381,0.0023942168,0.0024535567,0.009712521,0.005939093,0.006495247,0.032873053],"category_scores_gemma":[0.03308378,0.0014390848,0.0009903235,0.0020363561,0.005236797,0.046210684,0.007995839,0.010241852,0.015822304],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005669408,0.0003295899,0.00096851407,0.00056867197,0.00005405169,0.0004154001,0.0014348979,0.0013439552,0.012303776,0.5355455,0.17719893,0.26926973],"study_design_scores_gemma":[0.00018902034,0.00041560948,0.00092538225,0.0003773878,0.000109581815,0.0009526533,0.0014727148,0.022904383,0.030917302,0.2893019,0.65223974,0.00019431335],"about_ca_topic_score_codex":0.002096185,"about_ca_topic_score_gemma":0.0015297397,"teacher_disagreement_score":0.032873053,"about_ca_system_score_codex":0.0012498017,"about_ca_system_score_gemma":0.002185018,"threshold_uncertainty_score":0.109971344},"labels":[],"label_agreement":null},{"id":"W2241605029","doi":"10.18192/olbiwp.v7i0.1362","title":"La place de la compétence paraphrastique dans le Cadre européen commun de référence pour les langues","year":2015,"lang":"fr","type":"article","venue":"OLBI Journal","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"Mount Saint Vincent University","funders":"","keywords":"Humanities; Philosophy; Political science","score_opus":0.030622494043463247,"score_gpt":0.3050135326744553,"score_spread":0.27439103863099207,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2241605029","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5224644,0.0018190782,0.3170743,0.002394441,0.00012684005,0.00028308414,0.0010015867,0.0011858294,0.15365048],"genre_scores_gemma":[0.9451003,0.0004323195,0.04255064,0.00016841966,0.000027854636,0.0000764595,0.0005085082,0.00027862407,0.010856987],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9952154,0.002554814,0.00041833986,0.0007286294,0.0007663042,0.0003164999],"domain_scores_gemma":[0.9892881,0.0049819155,0.0008794081,0.0014508332,0.0031019321,0.00029776004],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004348821,0.00063972955,0.00040556703,0.0028732193,0.0016692741,0.00463891,0.0006718484,0.0011402235,0.0061081247],"category_scores_gemma":[0.013216713,0.00047405466,0.0005345464,0.0019528079,0.0030939318,0.0061476436,0.002942428,0.0015965399,0.0017664286],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002805016,0.00007352827,0.04789519,0.000977576,0.0001303365,0.0015602473,0.17065746,0.0014347533,0.06338728,0.4766905,0.0048469175,0.23206568],"study_design_scores_gemma":[0.000064445274,0.00062165514,0.14304765,0.0014108928,0.00033969554,0.008832465,0.13280466,0.016317083,0.07005172,0.10780689,0.51825535,0.00044743958],"about_ca_topic_score_codex":0.016587576,"about_ca_topic_score_gemma":0.016055943,"teacher_disagreement_score":0.016587576,"about_ca_system_score_codex":0.002086147,"about_ca_system_score_gemma":0.002313467,"threshold_uncertainty_score":0.03298205},"labels":[],"label_agreement":null},{"id":"W2250104175","doi":"10.33011/lilt.v8i.1305","title":"Learning to Classify Documents According to Formal and Informal Style","year":2012,"lang":"en","type":"article","venue":"Linguistic Issues in Language Technology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; University of Ottawa","keywords":"Computer science; Artificial intelligence; Naive Bayes classifier; Classifier (UML); Natural language processing; Style (visual arts); Sentence; Support vector machine; Machine learning; Computational linguistics","score_opus":0.0067689710536718095,"score_gpt":0.307116565836935,"score_spread":0.30034759478326317,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2250104175","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.45153674,0.0024445076,0.51855314,0.0021332914,0.0005416019,0.001361643,0.006031853,0.0035009743,0.01389622],"genre_scores_gemma":[0.5958941,0.0011404823,0.38811442,0.00037799941,0.0006093243,0.00063814264,0.008517098,0.00015592536,0.004552459],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99813086,0.0004652395,0.00027185486,0.00048935704,0.00049471855,0.00014795401],"domain_scores_gemma":[0.990035,0.005702756,0.0011156508,0.0007837943,0.0020097862,0.00035307292],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027030588,0.0010326697,0.0009098883,0.0065593505,0.00080241886,0.0029564856,0.0009218964,0.0011185884,0.002424589],"category_scores_gemma":[0.012971674,0.00029887998,0.0010464555,0.0032438722,0.00056099036,0.0042603016,0.0007647794,0.0013415994,0.0023638254],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003767578,0.00060393655,0.08189279,0.00059451326,0.00020905993,0.0002207096,0.0011606838,0.006937195,0.012241513,0.0050894674,0.013610537,0.87706286],"study_design_scores_gemma":[0.00032937166,0.0013249003,0.13423073,0.0007679242,0.0005874664,0.0017272675,0.0054859645,0.6720929,0.038965806,0.08714084,0.05700776,0.00033906274],"about_ca_topic_score_codex":0.0015060517,"about_ca_topic_score_gemma":0.0022221676,"teacher_disagreement_score":0.0065593505,"about_ca_system_score_codex":0.00075685466,"about_ca_system_score_gemma":0.00086360984,"threshold_uncertainty_score":0.01429528},"labels":[],"label_agreement":null},{"id":"W2250176460","doi":"10.18653/v1/w15-22","title":"Proceedings of the 14th International Conference on Parsing Technologies","year":2015,"lang":"en","type":"paratext","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Parsing; Computer science; Programming language","score_opus":0.03956609107596272,"score_gpt":0.30929038283822574,"score_spread":0.26972429176226304,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2250176460","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004914313,0.07852471,0.39700243,0.031619184,0.06040554,0.0010492203,0.012178826,0.035920482,0.37838528],"genre_scores_gemma":[0.023882013,0.050744485,0.2527112,0.013161042,0.011921642,0.0010604404,0.058541007,0.011895965,0.5760822],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99508977,0.0013276658,0.0006035572,0.0011002716,0.0015255188,0.00035325447],"domain_scores_gemma":[0.995006,0.0014131861,0.00014425072,0.001316823,0.0016436905,0.00047609708],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004392933,0.0021558756,0.0018201831,0.002557295,0.0015393601,0.009174203,0.0030448907,0.0032228534,0.15050222],"category_scores_gemma":[0.009451884,0.0010538754,0.0018052807,0.0033504872,0.0016028121,0.012832723,0.004132294,0.0055359495,0.10866599],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016219995,0.000103999606,0.0003078895,0.00047918197,0.000069691174,0.00020485603,0.0002645922,0.00034929733,0.00286104,0.018826686,0.7222961,0.25407436],"study_design_scores_gemma":[0.000012913024,0.000020386653,0.00023924539,0.00018096273,0.00002405257,0.00016052666,0.000075345175,0.00082812464,0.00071795215,0.008068224,0.98965365,0.000018560984],"about_ca_topic_score_codex":0.0034468623,"about_ca_topic_score_gemma":0.0034397217,"teacher_disagreement_score":0.15050222,"about_ca_system_score_codex":0.0016176138,"about_ca_system_score_gemma":0.002716875,"threshold_uncertainty_score":0.5034801},"labels":[],"label_agreement":null},{"id":"W2250181863","doi":"","title":"Poly-co: a multilayer perceptron approach for coreference detection","year":2011,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Coreference; Computer science; Artificial intelligence; Classifier (UML); Perceptron; Natural language processing; Pattern recognition (psychology); Task (project management); Pipeline (software); Similarity (geometry); Machine learning; Multilayer perceptron; Resolution (logic); Artificial neural network","score_opus":0.05520778678818637,"score_gpt":0.28798700781334413,"score_spread":0.23277922102515775,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2250181863","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0069705504,0.000494756,0.96552604,0.0002614359,0.0001614272,0.00020462924,0.0010345618,0.020811312,0.00453529],"genre_scores_gemma":[0.17687286,0.00038752792,0.796525,0.000683148,0.00022687332,0.00070298504,0.0050368765,0.0018147416,0.017749997],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99802303,0.00058256724,0.000106154104,0.0005547393,0.00050716125,0.00022632627],"domain_scores_gemma":[0.9978205,0.000765663,0.00018120822,0.0005409182,0.0005593462,0.0001325012],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003126425,0.0015053069,0.0012272268,0.0025110764,0.0013313796,0.0021625205,0.0033519284,0.0020125217,0.010072137],"category_scores_gemma":[0.004853814,0.000875112,0.0008637354,0.0019079383,0.00060833007,0.0034081424,0.003018113,0.0025186145,0.0053591486],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011844639,0.0005173844,0.0025431274,0.0005219498,0.0004889825,0.00027513766,0.00025026355,0.039577406,0.026442377,0.011087641,0.05760765,0.8595037],"study_design_scores_gemma":[0.0000676762,0.00013531414,0.0010267288,0.000036655623,0.000095677424,0.00014457983,0.00005163645,0.94297284,0.017751504,0.013813618,0.023849612,0.000054142976],"about_ca_topic_score_codex":0.005862859,"about_ca_topic_score_gemma":0.013103668,"teacher_disagreement_score":0.010072137,"about_ca_system_score_codex":0.0009795774,"about_ca_system_score_gemma":0.001580465,"threshold_uncertainty_score":0.033694685},"labels":[],"label_agreement":null},{"id":"W2250225327","doi":"10.18653/v1/d13-1043","title":"Effectiveness and Efficiency of Open Relation Extraction","year":2013,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":79,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Relationship extraction; Relation (database); Artificial intelligence; Natural language processing; Range (aeronautics); Extraction (chemistry); Binary classification; Binary relation; Machine learning; Information extraction; Data mining; Support vector machine; Mathematics; Engineering","score_opus":0.010791339979436447,"score_gpt":0.30463850747660665,"score_spread":0.2938471674971702,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2250225327","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.40564442,0.032426476,0.3801418,0.008377211,0.0014279191,0.0012721851,0.020874504,0.064446725,0.08538873],"genre_scores_gemma":[0.47063878,0.005346653,0.4624052,0.0009534825,0.00049595494,0.00027684364,0.041440494,0.0021847456,0.016257927],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9850246,0.005244468,0.0018582485,0.0035555558,0.003519076,0.000798119],"domain_scores_gemma":[0.9516119,0.03301656,0.0017285192,0.010772454,0.0021683422,0.00070235704],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010101905,0.0025387867,0.0032770357,0.008425023,0.0024202839,0.007331897,0.0047889976,0.0044175726,0.009658114],"category_scores_gemma":[0.03746496,0.0010563855,0.0025099553,0.00785616,0.0025790355,0.01870468,0.007235856,0.0028484815,0.008322517],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015501832,0.0005880797,0.006852056,0.0017884426,0.0002539952,0.0002476575,0.00071555475,0.007685426,0.012886258,0.0112494705,0.037035998,0.9191469],"study_design_scores_gemma":[0.0007536799,0.0012120458,0.027775906,0.0009160613,0.0011471506,0.0039988477,0.0047443537,0.6020425,0.12519091,0.11259433,0.119184285,0.00043994538],"about_ca_topic_score_codex":0.0044394853,"about_ca_topic_score_gemma":0.007376331,"teacher_disagreement_score":0.010101905,"about_ca_system_score_codex":0.0017525919,"about_ca_system_score_gemma":0.0026553052,"threshold_uncertainty_score":0.053424597},"labels":[],"label_agreement":null},{"id":"W2250257730","doi":"10.18653/v1/w15-3056","title":"Improving evaluation and optimization of MT systems against MEANT","year":2015,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"Advanced Research Projects Agency; Defense Advanced Research Projects Agency; European Commission","keywords":"Computer science","score_opus":0.02890033713408698,"score_gpt":0.28656541010485465,"score_spread":0.2576650729707677,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2250257730","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.61101377,0.0030956075,0.34555537,0.0006014343,0.0003851089,0.0002586483,0.00088085537,0.021510277,0.016698973],"genre_scores_gemma":[0.89582527,0.0001569762,0.099503346,0.00016205494,0.000042054886,0.00010754372,0.0013227705,0.00057355745,0.0023064392],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9972487,0.0012077622,0.000270951,0.00057103665,0.00053359295,0.00016794748],"domain_scores_gemma":[0.99521977,0.002512294,0.00026804843,0.000829082,0.0010288103,0.0001420539],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027916818,0.0009147387,0.0008267062,0.0007900003,0.0007157268,0.0011733153,0.0009687907,0.00096284854,0.0015963521],"category_scores_gemma":[0.014268622,0.00036513648,0.00043758066,0.00072667864,0.0005660243,0.001344587,0.0012812555,0.0009268823,0.00096010906],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015796545,0.00044379782,0.017138176,0.0011510033,0.00036545002,0.00059161504,0.00058032904,0.29111457,0.15224877,0.007993255,0.011650508,0.5151429],"study_design_scores_gemma":[0.000112899055,0.00074791606,0.008663564,0.000057804922,0.00017963462,0.000511627,0.00017760413,0.79207635,0.1799973,0.006722424,0.010641467,0.00011146398],"about_ca_topic_score_codex":0.002146474,"about_ca_topic_score_gemma":0.0048941183,"teacher_disagreement_score":0.0027916818,"about_ca_system_score_codex":0.0010950393,"about_ca_system_score_gemma":0.0009372423,"threshold_uncertainty_score":0.014764011},"labels":[],"label_agreement":null},{"id":"W2250335667","doi":"","title":"Events are Not Simple: Identity, Non-Identity, and Quasi-Identity","year":2013,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":63,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Coreference; Identity (music); Computer science; Annotation; Event (particle physics); Natural language processing; Simple (philosophy); Artificial intelligence; Information retrieval; Resolution (logic); Epistemology; Philosophy","score_opus":0.018982804843620842,"score_gpt":0.30834303601066004,"score_spread":0.2893602311670392,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2250335667","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.47528455,0.0018360215,0.3898353,0.005258196,0.0005047832,0.00024651,0.00067278015,0.00059437903,0.12576745],"genre_scores_gemma":[0.97004974,0.0001885941,0.026387349,0.00031625433,0.000092753115,0.00006357997,0.0003432998,0.000103731654,0.002454621],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9902464,0.00538116,0.00071536103,0.0021175377,0.0011374997,0.0004020381],"domain_scores_gemma":[0.95510644,0.03065827,0.00343445,0.0060799075,0.0041078376,0.00061317603],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007491659,0.00034961422,0.00053610164,0.0017009013,0.0042526727,0.0052460576,0.0012377936,0.0016687013,0.0087782955],"category_scores_gemma":[0.048184082,0.0006882679,0.0005880974,0.0019654778,0.008338771,0.016688216,0.0057284893,0.0025178588,0.0009698351],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004798271,0.000108627195,0.019067783,0.0003485361,0.00006307103,0.0009204104,0.017220119,0.0015354292,0.002879008,0.8699845,0.005524976,0.08186773],"study_design_scores_gemma":[0.000043578577,0.00008001847,0.012909974,0.00020260713,0.0000672362,0.0013131478,0.008940414,0.019489631,0.0042871092,0.9214895,0.031086138,0.000090746944],"about_ca_topic_score_codex":0.0024655317,"about_ca_topic_score_gemma":0.0019833446,"teacher_disagreement_score":0.0087782955,"about_ca_system_score_codex":0.0015049034,"about_ca_system_score_gemma":0.0011234921,"threshold_uncertainty_score":0.03962016},"labels":[],"label_agreement":null},{"id":"W2250341893","doi":"","title":"Mapping Source to Target Strings without Alignment by Analogical Learning: A Case Study with Transliteration","year":2013,"lang":"en","type":"article","venue":"Meeting of the Association for Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Transliteration; Computer science; Natural language processing; Phrase; Analogy; Artificial intelligence; Machine translation; Task (project management); Translation (biology); Rule-based machine translation; Machine learning; Linguistics; Engineering","score_opus":0.010253602251457633,"score_gpt":0.2590147617676204,"score_spread":0.24876115951616276,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2250341893","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9000833,0.0009240702,0.081973985,0.0016989292,0.000062237625,0.00037674367,0.00063251227,0.0011285197,0.013119727],"genre_scores_gemma":[0.9013769,0.00035129217,0.093037456,0.00028616586,0.000048210146,0.00016554572,0.0005800484,0.0003292841,0.0038250575],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99547195,0.0030847236,0.0003358724,0.0004429277,0.0004989469,0.00016557188],"domain_scores_gemma":[0.97325486,0.021031415,0.00072091556,0.0035806003,0.0011937316,0.00021840126],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003982951,0.00083069113,0.0009803104,0.0010277838,0.0015777225,0.0018196797,0.0020147236,0.003441906,0.0040131756],"category_scores_gemma":[0.028795637,0.00042885123,0.00079471373,0.002504324,0.0016085582,0.004666582,0.0019127923,0.0020967417,0.0014836256],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0026630391,0.0055127256,0.035652332,0.0036493833,0.00040933097,0.037164845,0.03544879,0.09575075,0.03850665,0.030721594,0.015833003,0.69868755],"study_design_scores_gemma":[0.0017565096,0.0043617054,0.025584053,0.00041311592,0.0005218175,0.029713076,0.023825599,0.5814965,0.15688834,0.08422687,0.090884194,0.000328216],"about_ca_topic_score_codex":0.0037373707,"about_ca_topic_score_gemma":0.0053921933,"teacher_disagreement_score":0.0040131756,"about_ca_system_score_codex":0.0007223265,"about_ca_system_score_gemma":0.00054283394,"threshold_uncertainty_score":0.021064103},"labels":[],"label_agreement":null},{"id":"W2250351693","doi":"","title":"Comparing Word Relatedness Measures Based on Google $n$-grams","year":2012,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Computer science; Word (group theory); Natural language processing; n-gram; Artificial intelligence; Text corpus; Point (geometry); Language model; Information retrieval; Mathematics","score_opus":0.03840143162589264,"score_gpt":0.2762720713499136,"score_spread":0.23787063972402095,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2250351693","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9059611,0.006536846,0.05528564,0.00039253145,0.00042456453,0.0004602387,0.009414127,0.0043239924,0.017200982],"genre_scores_gemma":[0.84215546,0.001595592,0.12536931,0.00012950136,0.0001877475,0.0005021362,0.026385177,0.0005402806,0.0031347524],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9943321,0.0018435016,0.00075124,0.0007467118,0.0020820794,0.00024440474],"domain_scores_gemma":[0.9832844,0.010036053,0.0015465257,0.001746965,0.002956223,0.00042974742],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004111827,0.001121981,0.001224737,0.017834717,0.0009508101,0.0019277884,0.0009812971,0.0012412186,0.0018470443],"category_scores_gemma":[0.031870507,0.00028442222,0.0009868151,0.012659202,0.0007324032,0.004848326,0.0019654704,0.00080520695,0.0018132464],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0032578674,0.0011867784,0.14583658,0.003200075,0.0018774373,0.0006891497,0.0025408159,0.034111135,0.028961444,0.014843179,0.042147886,0.72134763],"study_design_scores_gemma":[0.00040148484,0.0024600094,0.35649735,0.0004764668,0.00089901534,0.0033918521,0.00510401,0.4886434,0.050588757,0.043758553,0.047109317,0.0006696945],"about_ca_topic_score_codex":0.0050281347,"about_ca_topic_score_gemma":0.009637434,"teacher_disagreement_score":0.017834717,"about_ca_system_score_codex":0.0010021639,"about_ca_system_score_gemma":0.0007967303,"threshold_uncertainty_score":0.021745682},"labels":[],"label_agreement":null},{"id":"W2250484029","doi":"10.3115/v1/p14-2087","title":"Applying a Naive Bayes Similarity Measure to Word Sense Disambiguation","year":2014,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Word-sense disambiguation; Naive Bayes classifier; Bayes' theorem; Computer science; Word (group theory); Artificial intelligence; Similarity (geometry); Simple (philosophy); Natural language processing; Measure (data warehouse); Semantic similarity; Mathematics; Bayesian probability; Data mining; WordNet","score_opus":0.01600109139405482,"score_gpt":0.2691007630015634,"score_spread":0.25309967160750857,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2250484029","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02083144,0.0006546176,0.9727797,0.00024292679,0.0002382134,0.00018553529,0.00016468647,0.0024262525,0.0024767346],"genre_scores_gemma":[0.36364937,0.0004995556,0.6300286,0.0005492767,0.00033786558,0.0002734023,0.0009963509,0.00029561398,0.0033699558],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.98971117,0.003170425,0.00095005677,0.0022257045,0.0036333937,0.0003091651],"domain_scores_gemma":[0.9901002,0.004491853,0.00065025635,0.0018109109,0.0026859771,0.0002607959],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005809041,0.0011928617,0.0021972803,0.0060425056,0.0019506912,0.00321794,0.0023708222,0.0019131502,0.002801076],"category_scores_gemma":[0.02638095,0.0006841645,0.0011351461,0.0045441114,0.0015966546,0.0070080943,0.0030439384,0.001620413,0.002638814],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00051969005,0.00043830118,0.0060337074,0.00042893022,0.0003509849,0.00028960773,0.00041382917,0.05102366,0.014033775,0.027804174,0.008959064,0.8897043],"study_design_scores_gemma":[0.00006085418,0.00016069222,0.0015313951,0.00005568149,0.000094104435,0.00044593922,0.00012368578,0.89340156,0.011515773,0.088068135,0.004451956,0.00009026747],"about_ca_topic_score_codex":0.0043020467,"about_ca_topic_score_gemma":0.006790054,"teacher_disagreement_score":0.0060425056,"about_ca_system_score_codex":0.0013936037,"about_ca_system_score_gemma":0.0023375778,"threshold_uncertainty_score":0.030721486},"labels":[],"label_agreement":null},{"id":"W2250484191","doi":"10.3115/v1/w14-0904","title":"Time after Time: Representing Time in Literary Texts","year":2014,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Representation (politics); Narrative; Computer science; Meaning (existential); Simple (philosophy); Perception; Linguistics; Epistemology; Philosophy; Politics","score_opus":0.0032520082641256083,"score_gpt":0.22774450065395413,"score_spread":0.22449249238982852,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2250484191","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05109816,0.001285585,0.9296181,0.0014515547,0.00034055547,0.00012116413,0.0018242535,0.0020277253,0.012232844],"genre_scores_gemma":[0.59825504,0.0015617892,0.38988206,0.00021826771,0.00021635962,0.00020031053,0.0022158264,0.0004324257,0.007017921],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9992349,0.00033409533,0.00006772273,0.00022121979,0.0000957662,0.000046220204],"domain_scores_gemma":[0.99641466,0.0021343671,0.00047487157,0.0004964221,0.00032157596,0.00015812613],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012935916,0.00064151606,0.00036663184,0.0018538719,0.0009320505,0.003588489,0.0011409634,0.0012464391,0.0037865744],"category_scores_gemma":[0.009278764,0.0003395846,0.0007155553,0.0024437087,0.0017409066,0.0087044565,0.0014846475,0.0011472546,0.00093690347],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005335494,0.000059887763,0.004009429,0.00036219042,0.00004638739,0.00062378094,0.0058433535,0.07747548,0.0040607452,0.78848654,0.007000962,0.11149776],"study_design_scores_gemma":[0.000036101133,0.00007444524,0.001066827,0.00015678328,0.000068351226,0.0003093239,0.0012466912,0.341144,0.003355147,0.5751225,0.07735682,0.00006302307],"about_ca_topic_score_codex":0.008369892,"about_ca_topic_score_gemma":0.0069265803,"teacher_disagreement_score":0.008369892,"about_ca_system_score_codex":0.0013004564,"about_ca_system_score_gemma":0.00091870676,"threshold_uncertainty_score":0.016642332},"labels":[],"label_agreement":null},{"id":"W2250491265","doi":"10.63317/24i436sxdny8","title":"Texto4Science: a Quebec French Database of Annotated Short Text Messages","year":2012,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":46,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Period (music); Computer science; Phenomenon; Database; World Wide Web; Library science; Art","score_opus":0.019129728850742947,"score_gpt":0.30003272760387034,"score_spread":0.2809029987531274,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2250491265","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019912133,0.0017184617,0.0055888854,0.0006478563,0.00013314463,0.00040825765,0.95054615,0.012313133,0.008731963],"genre_scores_gemma":[0.027272396,0.00073380995,0.00991478,0.00014923872,0.00004424195,0.00038335976,0.95301527,0.00081873895,0.0076681236],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9987447,0.00022128677,0.000107075895,0.0003134567,0.0004721964,0.0001412763],"domain_scores_gemma":[0.9953107,0.001097274,0.00025112374,0.00049899734,0.002501062,0.000340784],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00091832224,0.002292968,0.0010971919,0.011402843,0.0021509174,0.002004364,0.002240851,0.001859555,0.02506322],"category_scores_gemma":[0.007874903,0.0005140699,0.00081214024,0.010032486,0.00066857424,0.0019717407,0.0009749375,0.0010406406,0.013917147],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012386917,0.00022179206,0.0063698147,0.0028840615,0.00019121093,0.00096055755,0.00087797415,0.0034319623,0.01041595,0.0041458425,0.8719248,0.09733732],"study_design_scores_gemma":[0.00042913135,0.00012487684,0.03491121,0.00051133335,0.0002308656,0.0005250852,0.0011648558,0.022093173,0.010928782,0.0022274882,0.9266265,0.00022680176],"about_ca_topic_score_codex":0.6333551,"about_ca_topic_score_gemma":0.62965316,"teacher_disagreement_score":0.36664492,"about_ca_system_score_codex":0.005495195,"about_ca_system_score_gemma":0.011580894,"threshold_uncertainty_score":0.7376083},"labels":[],"label_agreement":null},{"id":"W2250536124","doi":"","title":"Discovering frames in specialized domains","year":2014,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"FrameNet; Computer science; Complement (music); Field (mathematics); Artificial intelligence; Natural language processing; Information retrieval; Parsing; Mathematics","score_opus":0.011399440839449463,"score_gpt":0.2946918213716688,"score_spread":0.2832923805322194,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2250536124","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.41011974,0.0020587407,0.55894136,0.00146152,0.00015509484,0.0006675455,0.005838694,0.0066876737,0.014069563],"genre_scores_gemma":[0.70848316,0.00092716084,0.2723902,0.00024068086,0.00010541125,0.00019827289,0.012108729,0.0006971947,0.004849269],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9961202,0.0012581941,0.00028523305,0.0010352691,0.00081068795,0.00049039867],"domain_scores_gemma":[0.9907265,0.0055450033,0.00043171743,0.0015315334,0.0013207643,0.00044454992],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026232232,0.0013346843,0.0015756583,0.0066907303,0.0016889048,0.0038236028,0.002301669,0.0020316483,0.007951257],"category_scores_gemma":[0.014247846,0.00070108986,0.001619433,0.003815797,0.0011489973,0.00955022,0.0029345094,0.0018331743,0.0022174825],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002561462,0.0012876819,0.031422596,0.001002252,0.00045900972,0.0018488746,0.0021206883,0.032114815,0.030841712,0.07739259,0.023610396,0.7953379],"study_design_scores_gemma":[0.0002988354,0.0005231222,0.009576034,0.00029668122,0.00070761156,0.00082268304,0.0042038662,0.7242944,0.058071848,0.17430757,0.02678604,0.000111278074],"about_ca_topic_score_codex":0.013325788,"about_ca_topic_score_gemma":0.020223152,"teacher_disagreement_score":0.013325788,"about_ca_system_score_codex":0.001808833,"about_ca_system_score_gemma":0.0026336159,"threshold_uncertainty_score":0.026599586},"labels":[],"label_agreement":null},{"id":"W2250593090","doi":"10.63317/57vi3dv2nbm9","title":"Semantic Relations Established by Specialized Processes Expressed by Nouns and Verbs: Identification in a Corpus by means of Syntactico-semantic Annotation","year":2012,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Natural language processing; Noun; Identification (biology); Annotation; Verb; Context (archaeology); Artificial intelligence; Linguistics; Process (computing); Semantics (computer science); History; Philosophy","score_opus":0.009210122422158335,"score_gpt":0.25576971547144556,"score_spread":0.24655959304928723,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2250593090","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7081561,0.0023519732,0.26236343,0.0007470111,0.00015561857,0.0011868484,0.007425985,0.00070908316,0.016903922],"genre_scores_gemma":[0.6978137,0.0010931662,0.2796804,0.00013156263,0.00008031524,0.0030353481,0.015105972,0.0005300945,0.0025295147],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99195683,0.0029904789,0.0012761371,0.001831046,0.0017456945,0.00019968959],"domain_scores_gemma":[0.97177255,0.019416004,0.002642345,0.0031892024,0.002653389,0.0003265511],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0053273314,0.0006083336,0.0008513137,0.011653917,0.0033596552,0.0032879363,0.0009876592,0.0011630567,0.0022783373],"category_scores_gemma":[0.021637343,0.00054691394,0.0006695556,0.014404835,0.00410775,0.0052301735,0.0033764679,0.0017794025,0.0005663942],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010111319,0.00067181,0.050308637,0.0059443964,0.00026093144,0.004419655,0.1945404,0.004069898,0.19438489,0.1409808,0.008069003,0.39533848],"study_design_scores_gemma":[0.00030145934,0.0003807425,0.23717636,0.0018874734,0.0008898091,0.0058520064,0.10554278,0.06238604,0.09295329,0.12585965,0.3661885,0.00058190105],"about_ca_topic_score_codex":0.004154551,"about_ca_topic_score_gemma":0.005171385,"teacher_disagreement_score":0.011653917,"about_ca_system_score_codex":0.0018201552,"about_ca_system_score_gemma":0.0028128403,"threshold_uncertainty_score":0.028173923},"labels":[],"label_agreement":null},{"id":"W2250595697","doi":"","title":"Building NLP resources for Dzongkha: A Tagset and A Tagged Corpus","year":2010,"lang":"en","type":"article","venue":"Publication Server of the Institute for German Language (Institute for German Language)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"National University of Computer and Emerging Sciences; Universität Stuttgart; International Development Research Centre","keywords":"Treebank; Lexicon; Computer science; Natural language processing; Artificial intelligence; Set (abstract data type); Security token; Part of speech; Word (group theory); Training set; Speech recognition; Parsing; Linguistics","score_opus":0.010980096040104602,"score_gpt":0.2980911649840398,"score_spread":0.28711106894393523,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2250595697","genre_codex":"methods","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07072034,0.00077772076,0.66469455,0.001083105,0.0005249132,0.0029558884,0.18683584,0.042493965,0.029913705],"genre_scores_gemma":[0.16775393,0.0004885275,0.53266305,0.0002221553,0.00006915346,0.0037790341,0.28088045,0.0038009575,0.010342748],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9986765,0.00026154178,0.0002031665,0.00048030133,0.00027175533,0.0001066952],"domain_scores_gemma":[0.9965204,0.0013783352,0.0003035862,0.000936709,0.0006802716,0.00018067361],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019479712,0.0009558507,0.0009596285,0.005148655,0.0017277219,0.0019884962,0.0015947645,0.0008563445,0.016909545],"category_scores_gemma":[0.006235843,0.0012903743,0.00079842407,0.0036443572,0.00079865265,0.00480283,0.0032437537,0.0014976384,0.015352491],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014616331,0.00063127925,0.021675523,0.0041582454,0.00022768874,0.0041985703,0.0064889495,0.027965486,0.08580084,0.06776998,0.20055641,0.5790655],"study_design_scores_gemma":[0.00030569857,0.000208773,0.028442996,0.0007204225,0.00023956908,0.0024863181,0.0043350705,0.12937963,0.060858086,0.033712097,0.7389076,0.00040369976],"about_ca_topic_score_codex":0.010051638,"about_ca_topic_score_gemma":0.013884499,"teacher_disagreement_score":0.016909545,"about_ca_system_score_codex":0.0017475545,"about_ca_system_score_gemma":0.003854633,"threshold_uncertainty_score":0.056568027},"labels":[],"label_agreement":null},{"id":"W2250607250","doi":"","title":"Using a Weighted Semantic Network for Lexical Semantic Relatedness","year":2013,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Semantic similarity; Semantic compression; Computer science; Natural language processing; Semantic computing; Artificial intelligence; Semantic equivalence; Lexicon; Semantic network; Semantics (computer science); Semantic technology; Semantic Web","score_opus":0.02609649029556489,"score_gpt":0.28950165590256705,"score_spread":0.26340516560700217,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2250607250","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12652175,0.000964027,0.8589867,0.0003474436,0.00012463806,0.00033980023,0.0017871343,0.0010162586,0.009912233],"genre_scores_gemma":[0.68557096,0.00071118254,0.30842087,0.00008943184,0.000082064995,0.0005313145,0.0026509394,0.00013764405,0.0018055505],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9971909,0.0010518255,0.00025426492,0.0005334005,0.00086038385,0.000109225],"domain_scores_gemma":[0.99387336,0.0036342635,0.0007784138,0.0005367246,0.0010277212,0.00014955318],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024609566,0.00096841,0.00078404706,0.011524504,0.00095062715,0.0020495153,0.0009644061,0.0011807745,0.002474953],"category_scores_gemma":[0.016812772,0.00029152955,0.00063588296,0.008076514,0.0009411819,0.007385952,0.0018296535,0.000724956,0.0007026597],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00092113484,0.0004793733,0.0495396,0.0010406129,0.00093337684,0.00051513826,0.001281808,0.11642156,0.028991636,0.12300509,0.0072454,0.66962534],"study_design_scores_gemma":[0.00005344939,0.00022388149,0.015002617,0.00013731488,0.00034640194,0.0006375583,0.00046112016,0.7704057,0.010493801,0.18959579,0.012502532,0.00013985249],"about_ca_topic_score_codex":0.0028509817,"about_ca_topic_score_gemma":0.0040276526,"teacher_disagreement_score":0.011524504,"about_ca_system_score_codex":0.0010715041,"about_ca_system_score_gemma":0.0006579916,"threshold_uncertainty_score":0.013014913},"labels":[],"label_agreement":null},{"id":"W2250616809","doi":"","title":"Paraphrasing for Style","year":2012,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":97,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Paraphrase; Style (visual arts); Computer science; Task (project management); Natural language processing; Writing style; Artificial intelligence; Testbed; Domain (mathematical analysis); Linguistics; World Wide Web; Literature; Art; Philosophy","score_opus":0.01955040338063642,"score_gpt":0.29193972365145227,"score_spread":0.27238932027081586,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2250616809","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11104705,0.0024121308,0.8112314,0.0017108418,0.00046056387,0.0008829281,0.008564486,0.015701098,0.04798948],"genre_scores_gemma":[0.5070611,0.000846389,0.4671553,0.00053990446,0.00021569985,0.00033945584,0.008667965,0.0011100816,0.014063979],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9974533,0.0008725878,0.0002367992,0.00071995845,0.00062517583,0.00009218495],"domain_scores_gemma":[0.9886091,0.0045196638,0.0008666077,0.0039149495,0.0018967058,0.0001930435],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016449718,0.000977716,0.0006096727,0.0019135481,0.0005281646,0.0020316753,0.0010814308,0.0007675268,0.013521231],"category_scores_gemma":[0.01483523,0.00024391981,0.00077166373,0.0018123585,0.0005511884,0.0038629838,0.0015825175,0.001673127,0.007054086],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00032049612,0.00028565415,0.0045050783,0.0010738384,0.00012357428,0.0002659526,0.0009732355,0.0056640655,0.07294145,0.017744003,0.026497839,0.8696048],"study_design_scores_gemma":[0.0001259762,0.0014436276,0.024475604,0.00046599447,0.0003218041,0.0046305847,0.0016503523,0.4919774,0.18545128,0.098418675,0.19082047,0.00021831726],"about_ca_topic_score_codex":0.00058612955,"about_ca_topic_score_gemma":0.0012367599,"teacher_disagreement_score":0.013521231,"about_ca_system_score_codex":0.0004887426,"about_ca_system_score_gemma":0.0005425975,"threshold_uncertainty_score":0.04523301},"labels":[],"label_agreement":null},{"id":"W2250657762","doi":"","title":"Clustering Semantically Equivalent Words into Cognate Sets in Multilingual Lists","year":2011,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":75,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Cognate; Computer science; Word (group theory); Natural language processing; Artificial intelligence; Cluster analysis; ENCODE; Similarity (geometry); Mathematics; Linguistics","score_opus":0.03354109724348471,"score_gpt":0.3110713321578363,"score_spread":0.2775302349143516,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2250657762","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.64403486,0.00055877946,0.34269756,0.0002769597,0.00008798077,0.000386093,0.0012392938,0.002006413,0.008712041],"genre_scores_gemma":[0.73973614,0.00022711993,0.25338927,0.00013422266,0.000058210244,0.00018568657,0.003948232,0.0001899702,0.0021311508],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99902,0.0002412279,0.00012158354,0.00031799206,0.00020858625,0.00009062799],"domain_scores_gemma":[0.99828017,0.0007126956,0.00023109985,0.00024732185,0.0004465655,0.00008209787],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005560224,0.00056956254,0.00065333216,0.004696778,0.0017381146,0.0018898902,0.0007348501,0.00072861055,0.0029114722],"category_scores_gemma":[0.0046028146,0.00021871387,0.00071705977,0.002789084,0.0007935276,0.002337492,0.0014924265,0.00056700513,0.0014448713],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015608803,0.00074532995,0.05212946,0.00065328873,0.00030476553,0.00074623927,0.004584885,0.016137972,0.07219601,0.022927362,0.005800176,0.8222137],"study_design_scores_gemma":[0.00021228874,0.0012099685,0.093428776,0.0003976263,0.00079322833,0.0032910132,0.010988983,0.54456246,0.10813563,0.19896387,0.037589733,0.00042642845],"about_ca_topic_score_codex":0.0027576801,"about_ca_topic_score_gemma":0.005118226,"teacher_disagreement_score":0.004696778,"about_ca_system_score_codex":0.00069851556,"about_ca_system_score_gemma":0.00078080635,"threshold_uncertainty_score":0.009739876},"labels":[],"label_agreement":null},{"id":"W2250672871","doi":"10.63317/5pfvmss38ob7","title":"Logic Based Methods for Terminological Assessment","year":2012,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Graphical user interface; Visualization; Graph; User interface; Programming language; Graph drawing; Artificial intelligence; Theoretical computer science; Information retrieval; Natural language processing","score_opus":0.07831892657256248,"score_gpt":0.46241662813405937,"score_spread":0.3840977015614969,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2250672871","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0004187098,0.0005042654,0.9928402,0.00021556787,0.000086653876,0.0000766237,0.00017410384,0.00060103723,0.005082783],"genre_scores_gemma":[0.03109836,0.0009878832,0.9612649,0.00029846045,0.0003456946,0.00043992198,0.0008977969,0.00040540867,0.004261616],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.98890287,0.0042625125,0.0013224698,0.0013270684,0.00388292,0.00030210393],"domain_scores_gemma":[0.9833567,0.011084574,0.0007171144,0.0020937736,0.0024851782,0.00026266617],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009500124,0.001528075,0.0011420649,0.00829749,0.0017019752,0.007489898,0.003762873,0.0017131168,0.019654127],"category_scores_gemma":[0.030448148,0.00079769926,0.0028352072,0.00563658,0.004162155,0.009678284,0.004245579,0.004330503,0.0076084463],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005203693,0.000052739273,0.0004467823,0.0005287709,0.00007001072,0.00012382347,0.0004557947,0.005542962,0.0011708483,0.80195093,0.007247368,0.18235801],"study_design_scores_gemma":[0.000023284625,0.000020056012,0.00011995018,0.00016672438,0.0000374357,0.00014284131,0.00012729334,0.040362787,0.0011296667,0.9196796,0.03815274,0.000037595484],"about_ca_topic_score_codex":0.0016830916,"about_ca_topic_score_gemma":0.0016234083,"teacher_disagreement_score":0.019654127,"about_ca_system_score_codex":0.0030171862,"about_ca_system_score_gemma":0.0021147684,"threshold_uncertainty_score":0.065749586},"labels":[],"label_agreement":null},{"id":"W2250674905","doi":"10.18653/v1/w15-2706","title":"Identification and Disambiguation of Lexical Cues of Rhetorical Relations across Different Text Genres","year":2015,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Natural language processing; Rhetorical question; Relation (database); Artificial intelligence; Set (abstract data type); Context (archaeology); Probabilistic logic; Graph; Identification (biology); Linguistics","score_opus":0.042444169908227955,"score_gpt":0.34298973256943394,"score_spread":0.30054556266120597,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2250674905","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8836672,0.002811253,0.09477474,0.0005556726,0.0002483174,0.00039006138,0.003480942,0.0019047586,0.012167149],"genre_scores_gemma":[0.9318649,0.0005360088,0.06376861,0.00005836646,0.0000749438,0.00009113465,0.0025712012,0.00019784589,0.0008369149],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.999094,0.0002968209,0.00008949928,0.00028135235,0.00017556529,0.00006285816],"domain_scores_gemma":[0.9928706,0.0046810643,0.000877169,0.00046984424,0.0008520845,0.00024922178],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012441,0.00053937297,0.000636813,0.0062330547,0.00065462646,0.0016198832,0.00043568373,0.0007091197,0.001844754],"category_scores_gemma":[0.010447166,0.0002863857,0.00043614226,0.002370361,0.00049573247,0.0027903586,0.0012575514,0.0008266762,0.0007686426],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0024465316,0.0005625442,0.055100486,0.0033186174,0.00020234565,0.0014822847,0.0060201464,0.0051199407,0.305873,0.016670747,0.0056504095,0.5975529],"study_design_scores_gemma":[0.0003239384,0.0014239963,0.42017815,0.000939196,0.0009809956,0.0030394017,0.011423656,0.29444262,0.15276629,0.057151776,0.056860395,0.00046951827],"about_ca_topic_score_codex":0.0011794952,"about_ca_topic_score_gemma":0.0022531727,"teacher_disagreement_score":0.0062330547,"about_ca_system_score_codex":0.00042243136,"about_ca_system_score_gemma":0.00068242155,"threshold_uncertainty_score":0.0065795183},"labels":[],"label_agreement":null},{"id":"W2250711464","doi":"","title":"Adapting SimpleNLG for Bilingual English-French Realisation","year":2013,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":38,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Realisation; Computer science; Linguistics; Adaptation (eye); Artificial intelligence; Natural language processing; Psychology; Physics; Philosophy; Astronomy","score_opus":0.0186436990998433,"score_gpt":0.27462152872427026,"score_spread":0.25597782962442694,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2250711464","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.034673426,0.0005581354,0.8419201,0.00049237476,0.0005382953,0.00042460705,0.0014038598,0.096017435,0.02397177],"genre_scores_gemma":[0.35089728,0.00040352196,0.60578924,0.000593903,0.00014817248,0.00028091445,0.0063610612,0.018322423,0.017203502],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99732816,0.0008361877,0.00022266418,0.00057092286,0.00078561174,0.00025643574],"domain_scores_gemma":[0.99737835,0.0007556601,0.0000896332,0.0010211125,0.0006504082,0.000104838],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002525523,0.0013311304,0.0009161247,0.0012272893,0.00054508616,0.002193549,0.001949494,0.0011566944,0.020567393],"category_scores_gemma":[0.0076531493,0.0007328853,0.0011347699,0.00064102345,0.0010443899,0.003962033,0.003937778,0.0016322986,0.007473256],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021743781,0.00038154627,0.004321398,0.0018915961,0.0002597323,0.0024984314,0.004644202,0.024221625,0.09897442,0.0787843,0.05233302,0.7295154],"study_design_scores_gemma":[0.00039472297,0.0004033095,0.002080922,0.00027765954,0.00016633069,0.0032864597,0.0011352925,0.15292653,0.13554832,0.044427034,0.65898824,0.00036511966],"about_ca_topic_score_codex":0.0037366287,"about_ca_topic_score_gemma":0.005088643,"teacher_disagreement_score":0.020567393,"about_ca_system_score_codex":0.0011775885,"about_ca_system_score_gemma":0.00090506783,"threshold_uncertainty_score":0.06880474},"labels":[],"label_agreement":null},{"id":"W2250713394","doi":"10.3115/v1/w15-0705","title":"GutenTag: an NLP-driven Tool for Digital Humanities Research in the Project Gutenberg Corpus","year":2015,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":49,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Guelph","funders":"Natural Sciences and Engineering Research Council of Canada; Mitacs; University of Guelph","keywords":"Disk formatting; Metadata; Computer science; Digital humanities; Variety (cybernetics); Corpus linguistics; Natural language processing; World Wide Web; Software; Digital library; Artificial intelligence; Linguistics; Information retrieval; Programming language","score_opus":0.2177205030712601,"score_gpt":0.41288166879962973,"score_spread":0.19516116572836964,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2250713394","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007844079,0.00049455947,0.7523372,0.0015364767,0.00044210977,0.0011247735,0.021619545,0.18626104,0.02834023],"genre_scores_gemma":[0.037561085,0.00050398154,0.87436885,0.00043437933,0.00015206463,0.0029002784,0.037943646,0.035392176,0.010743437],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9966338,0.0013968692,0.00038301238,0.0005989399,0.00085030816,0.00013708459],"domain_scores_gemma":[0.987927,0.00849448,0.0005205539,0.0017625251,0.00089040527,0.00040494336],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0067552933,0.00144547,0.0011090043,0.009005688,0.0027139785,0.005420538,0.0021360025,0.001546753,0.04315591],"category_scores_gemma":[0.022402182,0.0014759958,0.0012222514,0.005944128,0.002002345,0.00813828,0.0073901135,0.0029228204,0.012940977],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005903705,0.00018229637,0.0027923961,0.0030849962,0.00021536087,0.0017289382,0.01165598,0.005807428,0.024900142,0.117236175,0.3059878,0.5258181],"study_design_scores_gemma":[0.00018501184,0.00006146655,0.0017226796,0.00043437813,0.00005988604,0.001014855,0.0017402661,0.02356109,0.017422335,0.05693624,0.8966569,0.00020492444],"about_ca_topic_score_codex":0.0038525953,"about_ca_topic_score_gemma":0.0056221634,"teacher_disagreement_score":0.04315591,"about_ca_system_score_codex":0.0015652917,"about_ca_system_score_gemma":0.0040431228,"threshold_uncertainty_score":0.14437091},"labels":[],"label_agreement":null},{"id":"W2250730265","doi":"","title":"Hierarchical Topical Segmentation with Affinity Propagation","year":2014,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Segmentation; Affinity propagation; Tree (set theory); Paragraph; Hierarchical clustering; Cluster analysis; Artificial intelligence; Tree structure; Pattern recognition (psychology); Algorithm; Binary tree; Mathematics; Combinatorics; World Wide Web; Fuzzy clustering","score_opus":0.009472941188655253,"score_gpt":0.2555881451269402,"score_spread":0.24611520393828498,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2250730265","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0045189587,0.00022428551,0.9757756,0.0001025701,0.000078527955,0.0001388177,0.000458933,0.01705427,0.0016480751],"genre_scores_gemma":[0.064994246,0.00016192891,0.9201083,0.0002042802,0.00017296722,0.00034148595,0.0026690476,0.00326021,0.008087522],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99784434,0.0004502771,0.00012802922,0.00076451275,0.0006269979,0.0001858958],"domain_scores_gemma":[0.995514,0.0019445153,0.00035034516,0.00085409556,0.0011715848,0.00016546932],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019428428,0.0018690864,0.0013693887,0.0046781264,0.0016691914,0.0020333545,0.002816778,0.0026122811,0.011234937],"category_scores_gemma":[0.0069058132,0.0014024422,0.0019994846,0.003982007,0.0011721038,0.0033931,0.0025480082,0.002833882,0.008644501],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00044388112,0.00021801438,0.0018506781,0.0006118334,0.0003240666,0.00043223824,0.0015675286,0.056954227,0.06030707,0.017468989,0.043356497,0.8164649],"study_design_scores_gemma":[0.00007565459,0.000109004344,0.0010819078,0.00004137251,0.00009595438,0.00026170613,0.00018750441,0.90268564,0.033576887,0.033946514,0.027868535,0.00006931871],"about_ca_topic_score_codex":0.008458447,"about_ca_topic_score_gemma":0.014840235,"teacher_disagreement_score":0.011234937,"about_ca_system_score_codex":0.001299711,"about_ca_system_score_gemma":0.0014612519,"threshold_uncertainty_score":0.037584603},"labels":[],"label_agreement":null},{"id":"W2250762230","doi":"","title":"Similarity Patterns in Words (Invited talk)","year":2012,"lang":"en","type":"article","venue":"Conference of the European Chapter of the Association for Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Morpheme; Similarity (geometry); Sketch; Computer science; Natural language processing; Linguistics; Presentation (obstetrics); Artificial intelligence; Natural (archaeology); History; Philosophy","score_opus":0.03261942453279171,"score_gpt":0.2696888129044976,"score_spread":0.23706938837170588,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2250762230","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.044052918,0.108047485,0.26716393,0.16916035,0.14900604,0.0004688237,0.003259896,0.0035295596,0.25531104],"genre_scores_gemma":[0.2711781,0.044815585,0.08681218,0.023763303,0.06849557,0.0008296568,0.005058811,0.0021665266,0.4968802],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9994993,0.00015962112,0.000020723983,0.00017114023,0.00010961512,0.000039508624],"domain_scores_gemma":[0.99916744,0.00044055065,0.000028681547,0.000056201923,0.0001922169,0.000114902476],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012922598,0.0007004468,0.0005349911,0.0009560889,0.0011415869,0.0034217623,0.0005496513,0.0017102072,0.04288368],"category_scores_gemma":[0.0049307714,0.00037018926,0.0005431553,0.0008739899,0.00094015576,0.005061936,0.002041654,0.002052999,0.017480616],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002412828,0.00007969671,0.0009175376,0.0004797493,0.000058452173,0.00031250052,0.0016489684,0.00080791756,0.0045292913,0.115610965,0.6800881,0.19522563],"study_design_scores_gemma":[0.00005650673,0.00012380813,0.0020524242,0.00039618637,0.00004400249,0.00078844436,0.001318441,0.0018767968,0.0028226138,0.14315648,0.84729236,0.00007197009],"about_ca_topic_score_codex":0.00045924878,"about_ca_topic_score_gemma":0.0005679498,"teacher_disagreement_score":0.04288368,"about_ca_system_score_codex":0.0006163941,"about_ca_system_score_gemma":0.00033244735,"threshold_uncertainty_score":0.14346015},"labels":[],"label_agreement":null},{"id":"W2250803090","doi":"","title":"Method Mention Extraction from Scientific Research Papers","year":2012,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Computer science; Conditional random field; Natural language processing; Field (mathematics); Artificial intelligence; Focus (optics); Domain (mathematical analysis); Information retrieval","score_opus":0.07521854273098917,"score_gpt":0.4394912978280038,"score_spread":0.3642727550970146,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2250803090","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11057277,0.012703113,0.76849276,0.002426215,0.002330205,0.003037839,0.060268246,0.024194416,0.015974453],"genre_scores_gemma":[0.08750684,0.0033197256,0.8452184,0.00029559652,0.0006561654,0.0013133632,0.055866186,0.0009544857,0.0048691835],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9923504,0.0013041119,0.002106522,0.0013233329,0.002705158,0.00021055448],"domain_scores_gemma":[0.9404075,0.03643352,0.0060780845,0.0036417001,0.012783003,0.0006560723],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008062415,0.0018574365,0.0020033477,0.02497773,0.0015566716,0.003190626,0.0019576447,0.0018551836,0.0038157387],"category_scores_gemma":[0.03402984,0.0009980258,0.0017971565,0.014067681,0.00054064504,0.0034821767,0.0019801727,0.001914809,0.004135058],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00067586853,0.00022564399,0.020450985,0.013034015,0.00043764513,0.003767884,0.0027064255,0.004482973,0.1147539,0.00913504,0.04098339,0.7893462],"study_design_scores_gemma":[0.0002865392,0.00057063106,0.05027655,0.0021855508,0.0013811144,0.006806857,0.002757165,0.10501234,0.27697945,0.026324105,0.5269452,0.00047450213],"about_ca_topic_score_codex":0.0011631276,"about_ca_topic_score_gemma":0.0017824848,"teacher_disagreement_score":0.02497773,"about_ca_system_score_codex":0.0010220581,"about_ca_system_score_gemma":0.003640889,"threshold_uncertainty_score":0.0426386},"labels":[],"label_agreement":null},{"id":"W2250907725","doi":"","title":"Improved Reordering for Phrase-Based Translation using Sparse Features","year":2013,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":45,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Machine translation; Phrase; Artificial intelligence; Translation (biology); Natural language processing; Principle of maximum entropy; Entropy (arrow of time); Pattern recognition (psychology); Speech recognition","score_opus":0.026135768180464387,"score_gpt":0.27775514984921307,"score_spread":0.2516193816687487,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2250907725","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.050260846,0.0002098384,0.9396861,0.00020883104,0.000062850864,0.00008170458,0.00046871984,0.006478604,0.0025425002],"genre_scores_gemma":[0.5708197,0.00017557798,0.42100397,0.0001885821,0.00009849229,0.00017747999,0.002600628,0.0010510229,0.0038846354],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99935037,0.00027249168,0.0000451289,0.000116604,0.00016250505,0.00005288269],"domain_scores_gemma":[0.99834406,0.0008421981,0.00016363291,0.0003098312,0.00029250313,0.000047730195],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009885945,0.00083820865,0.00070444256,0.0008411358,0.0003461427,0.0005984822,0.0005667223,0.00065866666,0.004057811],"category_scores_gemma":[0.0048791184,0.00035023605,0.00047024738,0.00093248097,0.00034905504,0.0015668157,0.000739869,0.0010918248,0.0025075497],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005429991,0.00028199778,0.0025994405,0.00027070107,0.00008964097,0.00029970822,0.0002083893,0.23841828,0.08292039,0.011431781,0.010790778,0.65214586],"study_design_scores_gemma":[0.00003135562,0.00013525548,0.00077499077,0.000012386383,0.000018975064,0.00012000355,0.000022228085,0.96915495,0.01994896,0.0073085106,0.0024507446,0.000021539883],"about_ca_topic_score_codex":0.0016830724,"about_ca_topic_score_gemma":0.004573962,"teacher_disagreement_score":0.004057811,"about_ca_system_score_codex":0.00043979197,"about_ca_system_score_gemma":0.0006061601,"threshold_uncertainty_score":0.013574779},"labels":[],"label_agreement":null},{"id":"W2250951601","doi":"10.63317/2uw4887z6mj3","title":"Building a reference lexicon for countability in English","year":2014,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Lexicon; Noun; Computer science; WordNet; Natural language processing; Artificial intelligence; Set (abstract data type); Countable set; Linguistics; Class (philosophy); Proper noun; Noun phrase; Agreement; Mathematics; Combinatorics","score_opus":0.020295716354989182,"score_gpt":0.30264079827302937,"score_spread":0.2823450819180402,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2250951601","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020609982,0.0012898062,0.8821291,0.0016549786,0.0007116607,0.00071448967,0.027167423,0.02883277,0.036889784],"genre_scores_gemma":[0.23389623,0.0015116701,0.6649117,0.0007061991,0.00043279392,0.00097414956,0.07634224,0.008916162,0.012308968],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99558395,0.0010184137,0.0010527155,0.0011624604,0.0009279786,0.00025444175],"domain_scores_gemma":[0.9889621,0.003766036,0.0005096185,0.0014795088,0.005021525,0.0002611386],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002412981,0.0015050217,0.0019106532,0.013633872,0.0026509229,0.0060042017,0.0031687068,0.0017789385,0.024391126],"category_scores_gemma":[0.016628766,0.0015211543,0.0015045912,0.008986647,0.0013754979,0.016751088,0.004135419,0.0025283897,0.013623021],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002339322,0.0002950335,0.004579358,0.0031319042,0.00015826561,0.0014611494,0.003541562,0.0056394413,0.017992686,0.28674984,0.19277234,0.48344448],"study_design_scores_gemma":[0.000166847,0.00019736664,0.0039856327,0.00136052,0.00053294736,0.001983854,0.003513661,0.15728992,0.027556837,0.17523625,0.6278713,0.00030482892],"about_ca_topic_score_codex":0.013802549,"about_ca_topic_score_gemma":0.018902825,"teacher_disagreement_score":0.024391126,"about_ca_system_score_codex":0.0025402505,"about_ca_system_score_gemma":0.0041556084,"threshold_uncertainty_score":0.081596434},"labels":[],"label_agreement":null},{"id":"W2250959204","doi":"10.18653/v1/k15-2008","title":"The CLaC Discourse Parser at CoNLL-2015","year":2015,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Parsing; Task (project management); Natural language processing; Artificial intelligence; Set (abstract data type); Test set; Programming language","score_opus":0.02879135923037636,"score_gpt":0.34077442234351807,"score_spread":0.3119830631131417,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2250959204","genre_codex":"dataset","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015668144,0.004206307,0.15655725,0.010558469,0.011248441,0.003581671,0.48493442,0.23564547,0.07759985],"genre_scores_gemma":[0.02986099,0.00057690986,0.14070188,0.0019240415,0.00093367224,0.0025495999,0.77632195,0.018760948,0.02837003],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.982828,0.0060362713,0.0014250472,0.003929491,0.004559481,0.0012216935],"domain_scores_gemma":[0.97740483,0.0052443487,0.0008485638,0.00497332,0.00924884,0.0022801547],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0139806345,0.0055437596,0.0031139571,0.006228956,0.0038848326,0.006932349,0.0067208433,0.0050516343,0.072825946],"category_scores_gemma":[0.03442703,0.0025033,0.0018119877,0.003961997,0.0017659183,0.009005952,0.008727986,0.010023743,0.098256044],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023935462,0.00014173369,0.00022960299,0.0006189861,0.000037869337,0.00014323054,0.00016080131,0.0008300544,0.0049036,0.0037261792,0.94642633,0.042542364],"study_design_scores_gemma":[0.0008661307,0.00024534637,0.0022721135,0.00053776224,0.000101963546,0.00060538197,0.0005227256,0.036900323,0.03150887,0.01904401,0.90708166,0.00031373274],"about_ca_topic_score_codex":0.018141158,"about_ca_topic_score_gemma":0.02474468,"teacher_disagreement_score":0.072825946,"about_ca_system_score_codex":0.005860169,"about_ca_system_score_gemma":0.0111247515,"threshold_uncertainty_score":0.24362701},"labels":[],"label_agreement":null},{"id":"W2251004534","doi":"","title":"Fourteen Light Tasks for comparing Analogical and Phrase-based Machine Translation","year":2014,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Transliteration; Machine translation; Phrase; Computer science; Scripting language; Natural language processing; Artificial intelligence; Rule-based machine translation; Translation (biology); Example-based machine translation; Machine translation software usability; Programming language","score_opus":0.058601602908245884,"score_gpt":0.33216916151273396,"score_spread":0.27356755860448806,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2251004534","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.84517705,0.015775105,0.07087881,0.0012048543,0.0026644939,0.0033554784,0.023417685,0.008212156,0.029314423],"genre_scores_gemma":[0.7610103,0.002359075,0.1320142,0.0012777715,0.0006137888,0.00367432,0.08832413,0.0013461462,0.009380348],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9933809,0.002359647,0.001229694,0.0013067678,0.0013801843,0.00034278154],"domain_scores_gemma":[0.98539484,0.009156359,0.00071274955,0.0023788724,0.0016592576,0.0006979499],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00574626,0.0021374046,0.0015439704,0.0040783104,0.0017554958,0.0020217367,0.0020260205,0.0025671537,0.0067741615],"category_scores_gemma":[0.020799503,0.0004760612,0.001618525,0.0030396278,0.0010875226,0.0029669208,0.0032732594,0.0020442468,0.003872226],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.014547103,0.005805582,0.031568866,0.008883675,0.002668289,0.0007248577,0.00091486744,0.032905865,0.041824576,0.005537467,0.074670896,0.779948],"study_design_scores_gemma":[0.008642872,0.021390041,0.25891095,0.0017492734,0.0041044247,0.004068796,0.0038429862,0.29032326,0.17043675,0.03642344,0.19893827,0.0011690015],"about_ca_topic_score_codex":0.0030277888,"about_ca_topic_score_gemma":0.005815649,"teacher_disagreement_score":0.0067741615,"about_ca_system_score_codex":0.0012638405,"about_ca_system_score_gemma":0.001475768,"threshold_uncertainty_score":0.030389488},"labels":[],"label_agreement":null},{"id":"W2251008434","doi":"10.3115/v1/p14-2078","title":"Mutual Disambiguation for Entity Linking","year":2014,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University; Polytechnique Montréal","funders":"","keywords":"Computer science; Entity linking; Natural language processing; Artificial intelligence","score_opus":0.013484143093841636,"score_gpt":0.27985967114611926,"score_spread":0.2663755280522776,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2251008434","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00514979,0.0010127467,0.98126316,0.00024143475,0.00013329517,0.00009538449,0.00045049525,0.0065900716,0.0050636395],"genre_scores_gemma":[0.15148279,0.0009047348,0.8362266,0.0003128234,0.0002524882,0.00030351707,0.0030439626,0.0012306247,0.0062423665],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9937883,0.0026993477,0.00040634832,0.0013738094,0.0014599515,0.00027221028],"domain_scores_gemma":[0.9952265,0.0025336011,0.00031979143,0.0012530518,0.0005599763,0.00010708055],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0046926406,0.0012768977,0.0011677926,0.005356481,0.0021945084,0.0023169145,0.0023254803,0.0016809779,0.009263885],"category_scores_gemma":[0.013553719,0.0008659327,0.0013118102,0.0045946552,0.0016529238,0.0074735424,0.0052532447,0.0016665602,0.0058838823],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00040509264,0.00012174855,0.0013402953,0.0008057244,0.00021630607,0.00033303397,0.0006665213,0.048353527,0.011153066,0.16387245,0.03926031,0.7334719],"study_design_scores_gemma":[0.00008957861,0.00011694042,0.001366624,0.00013621837,0.00014957694,0.0009452246,0.00025010825,0.4538918,0.039159898,0.39322045,0.110516265,0.00015719136],"about_ca_topic_score_codex":0.001270677,"about_ca_topic_score_gemma":0.0024196026,"teacher_disagreement_score":0.009263885,"about_ca_system_score_codex":0.0009851411,"about_ca_system_score_gemma":0.001443248,"threshold_uncertainty_score":0.03099072},"labels":[],"label_agreement":null},{"id":"W2251013081","doi":"10.63317/4i8tgm5asdq4","title":"A disambiguation resource extracted from Wikipedia for semantic annotation","year":2012,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Entity linking; Information retrieval; Task (project management); Annotation; Natural language processing; Ontology; Word (group theory); Resource (disambiguation); Point (geometry); Artificial intelligence; Relation (database); Semantic annotation; Knowledge base; Linguistics; Data mining","score_opus":0.019164845378133826,"score_gpt":0.285825939011608,"score_spread":0.2666610936334742,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2251013081","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09645796,0.011839649,0.212058,0.0024618476,0.0024103364,0.0021347196,0.51227105,0.026966494,0.1333999],"genre_scores_gemma":[0.15792422,0.0035353582,0.35888273,0.0007065735,0.00038917287,0.0013093088,0.46070895,0.0024607312,0.014082943],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9987717,0.00022728997,0.00026914958,0.00032771323,0.00030369745,0.000100281155],"domain_scores_gemma":[0.9968395,0.0010413346,0.00037087654,0.0004531118,0.0010569532,0.00023819772],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008102951,0.0013695922,0.00088816194,0.020405818,0.0022845003,0.00157937,0.0011530075,0.0011376706,0.01094689],"category_scores_gemma":[0.0059432546,0.00042328835,0.0007033716,0.010954044,0.0006057994,0.0028139935,0.0024032646,0.000985998,0.0075442847],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006309249,0.0005146243,0.010210438,0.009282473,0.00030973193,0.0084612835,0.0036216355,0.0030451864,0.04631681,0.041613866,0.38032448,0.49566853],"study_design_scores_gemma":[0.00005063803,0.00008044467,0.010555729,0.0010871057,0.00023600322,0.0032654328,0.0013988964,0.0043008327,0.01710904,0.011676808,0.9500611,0.00017793641],"about_ca_topic_score_codex":0.00802828,"about_ca_topic_score_gemma":0.017211284,"teacher_disagreement_score":0.020405818,"about_ca_system_score_codex":0.0008303373,"about_ca_system_score_gemma":0.0032932393,"threshold_uncertainty_score":0.036621034},"labels":[],"label_agreement":null},{"id":"W2251026193","doi":"","title":"Unsupervised Multiword Segmentation of Large Corpora using Prediction-Driven Decomposition of n-grams","year":2014,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Artificial intelligence; Natural language processing; Segmentation; Lexicon; Word (group theory); Predictability; Decomposition; Text segmentation; Linguistics; Mathematics","score_opus":0.014945440768241556,"score_gpt":0.2973863426147029,"score_spread":0.28244090184646137,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2251026193","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.046851244,0.0004365319,0.94343036,0.00017014603,0.00008880762,0.00021877309,0.000983974,0.005492185,0.0023280014],"genre_scores_gemma":[0.16919173,0.00024655688,0.8199614,0.00011338917,0.000081260165,0.00049321225,0.005586128,0.0011311182,0.0031952003],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9986426,0.0004518612,0.00010335371,0.00049623504,0.00021873767,0.000087234155],"domain_scores_gemma":[0.9964167,0.0015072447,0.0003813838,0.0007607611,0.0008173097,0.00011659814],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009716307,0.0012201065,0.0008664337,0.0023917013,0.0009325133,0.0014991623,0.0011588742,0.0007758511,0.002268033],"category_scores_gemma":[0.004948481,0.000605359,0.00095308665,0.0027502717,0.0007939818,0.0022996243,0.0013066243,0.001426635,0.0025512727],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004377236,0.00028454085,0.006675478,0.00055650924,0.00021147983,0.00043454216,0.0012642998,0.04792095,0.14673994,0.009689462,0.014248788,0.77153623],"study_design_scores_gemma":[0.00003801941,0.00014977073,0.0076342938,0.000056603712,0.00006505195,0.00026089282,0.0003415259,0.91501355,0.041874554,0.020840403,0.013647501,0.000077870194],"about_ca_topic_score_codex":0.0040260465,"about_ca_topic_score_gemma":0.01138222,"teacher_disagreement_score":0.0040260465,"about_ca_system_score_codex":0.0007426097,"about_ca_system_score_gemma":0.0013560278,"threshold_uncertainty_score":0.008005261},"labels":[],"label_agreement":null},{"id":"W2251026829","doi":"10.3115/v1/p15-2046","title":"A Lexicalized Tree Kernel for Open Information Extraction","year":2015,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; National Institute of Informatics; Alberta Innovates; Alberta Innovates - Technology Futures","keywords":"Computer science; Kernel (algebra); Information extraction; Natural language processing; Artificial intelligence; Tree (set theory); Joint (building); Computational linguistics; Mathematics; Engineering; Discrete mathematics; Combinatorics","score_opus":0.05184732558618997,"score_gpt":0.35818173216752913,"score_spread":0.30633440658133915,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2251026829","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007722787,0.0005532119,0.97660714,0.00013477598,0.00011084838,0.00012526794,0.0010377655,0.011974561,0.00173369],"genre_scores_gemma":[0.13894121,0.00049783033,0.8461261,0.00014110217,0.00013070965,0.000248493,0.0064854347,0.001494093,0.0059351106],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99840873,0.00027379036,0.00025032766,0.00038802542,0.0004979262,0.00018120561],"domain_scores_gemma":[0.99700314,0.00085959525,0.0001825965,0.0008596494,0.00096167106,0.00013337024],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014494881,0.0006133535,0.0013625628,0.0042734314,0.00094266084,0.0026541208,0.0015981772,0.0010536159,0.00826508],"category_scores_gemma":[0.006594593,0.0005569256,0.0013437184,0.0046358746,0.000564732,0.0054679024,0.0033914188,0.001483005,0.0073933043],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00042533717,0.00030397216,0.0012336653,0.00030517468,0.00010743711,0.00023752662,0.00023707695,0.0044037625,0.01552444,0.0382416,0.023376284,0.9156037],"study_design_scores_gemma":[0.00013131589,0.00017676452,0.0018093877,0.0001356718,0.00015015408,0.00061404496,0.0003953435,0.7096797,0.033390675,0.19863452,0.05479296,0.000089445195],"about_ca_topic_score_codex":0.0022073844,"about_ca_topic_score_gemma":0.003401439,"teacher_disagreement_score":0.00826508,"about_ca_system_score_codex":0.0006572216,"about_ca_system_score_gemma":0.0015458799,"threshold_uncertainty_score":0.027649462},"labels":[],"label_agreement":null},{"id":"W2251071163","doi":"10.3115/v1/w14-5316","title":"The NRC System for Discriminating Similar Languages","year":2014,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":57,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Computer science; Discriminative model; Classifier (UML); Artificial intelligence; Natural language processing; Generative grammar; Voting; Group (periodic table); Task (project management); Feature (linguistics); F1 score; Linguistics","score_opus":0.009560383488284923,"score_gpt":0.27528216314463666,"score_spread":0.26572177965635174,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2251071163","genre_codex":"software","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.043539044,0.0015728913,0.3621772,0.0010812039,0.00082240254,0.001874131,0.058134407,0.448552,0.08224675],"genre_scores_gemma":[0.1872673,0.0006093086,0.5792931,0.0011759537,0.00023178835,0.001152959,0.16244832,0.013068708,0.05475259],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.996329,0.00045553417,0.00026599193,0.001073004,0.0015074649,0.00036893287],"domain_scores_gemma":[0.9949809,0.00063556543,0.00022567813,0.0016146504,0.0021505412,0.00039262095],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030340496,0.0017261526,0.0014972007,0.004436281,0.002321131,0.0024297729,0.0037194993,0.0015556559,0.031838432],"category_scores_gemma":[0.0079035815,0.00076246023,0.0011349304,0.0023165618,0.0008780546,0.0039664237,0.003797364,0.0020511826,0.042171646],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004654991,0.0002677594,0.0072514163,0.00066549564,0.000124027,0.0005083805,0.0007197017,0.0026273362,0.034351252,0.007873061,0.40383884,0.54130715],"study_design_scores_gemma":[0.00022378436,0.00041497592,0.016477097,0.00028299197,0.00024227132,0.0034252969,0.0010669865,0.19882666,0.07175939,0.014058584,0.6926385,0.00058346195],"about_ca_topic_score_codex":0.07971243,"about_ca_topic_score_gemma":0.099435,"teacher_disagreement_score":0.07971243,"about_ca_system_score_codex":0.0021865105,"about_ca_system_score_gemma":0.0044869296,"threshold_uncertainty_score":0.15849686},"labels":[],"label_agreement":null},{"id":"W2251142302","doi":"","title":"Natural Language Generation and Summarization at RALI","year":2013,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Automatic summarization; Computer science; Natural language generation; Natural language processing; Artificial intelligence; Natural (archaeology); Natural language; Linguistics; History","score_opus":0.007345351739512658,"score_gpt":0.23565433136445332,"score_spread":0.22830897962494068,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2251142302","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018418996,0.0058969613,0.88838077,0.008201065,0.0012479557,0.00037516,0.0016592173,0.027055077,0.048764877],"genre_scores_gemma":[0.18540207,0.0041840314,0.6665424,0.0013138409,0.0016810885,0.0005339139,0.008067843,0.0063799103,0.12589496],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99713814,0.0011124446,0.00013174026,0.00086416694,0.0005565574,0.00019705841],"domain_scores_gemma":[0.9982881,0.0007914324,0.00008243997,0.00031453604,0.00037037308,0.00015310124],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030026538,0.00092095917,0.0009623954,0.0016654142,0.0013388799,0.0034642154,0.0012898822,0.001460783,0.016886242],"category_scores_gemma":[0.0054743527,0.0006066325,0.0009507635,0.0013250678,0.000967596,0.002989682,0.0018480045,0.0024615428,0.015759682],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00043485392,0.00031871276,0.00069289876,0.00046942264,0.000098673496,0.0005452025,0.0030885895,0.010471463,0.044168886,0.08014414,0.13019979,0.7293673],"study_design_scores_gemma":[0.00022067297,0.00045599198,0.0016189786,0.00022604529,0.00008783194,0.0009055822,0.0006391763,0.1742224,0.08725378,0.10934526,0.62480175,0.00022252635],"about_ca_topic_score_codex":0.0022725305,"about_ca_topic_score_gemma":0.0022894843,"teacher_disagreement_score":0.016886242,"about_ca_system_score_codex":0.001539147,"about_ca_system_score_gemma":0.0010698308,"threshold_uncertainty_score":0.056490064},"labels":[],"label_agreement":null},{"id":"W2251155774","doi":"10.3115/v1/w14-5708","title":"Multiword noun compound bracketing using Wikipedia","year":2014,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Computer Research Institute of Montréal","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Bracketing (phenomenology); Computer science; Natural language processing; Noun; Artificial intelligence; Association (psychology); Meaning (existential); Relation (database); Word (group theory); Linguistics; Psychology; Data mining","score_opus":0.018686203349936707,"score_gpt":0.28183455948191277,"score_spread":0.26314835613197607,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2251155774","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17776011,0.0016831012,0.7648177,0.00045059263,0.0005337549,0.0008717128,0.0051857354,0.038040187,0.010657032],"genre_scores_gemma":[0.29302016,0.00046191106,0.67996424,0.00012336242,0.00009456587,0.00022886631,0.015495036,0.0013926454,0.009219162],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9968575,0.00060200837,0.00028928355,0.0012153581,0.00088149257,0.0001543393],"domain_scores_gemma":[0.99577904,0.001583895,0.0004737678,0.00088546606,0.0010637764,0.00021406812],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015278101,0.0012384716,0.001246248,0.0062388345,0.001356615,0.00203073,0.0014690127,0.0011548855,0.007290996],"category_scores_gemma":[0.007167073,0.0005374354,0.0010756996,0.0039786366,0.0004857037,0.005972495,0.0030937733,0.000945661,0.005536293],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00059620757,0.00036618506,0.009039404,0.0011333806,0.00032640208,0.0014077232,0.0013837548,0.011486992,0.052302197,0.010232281,0.020269526,0.891456],"study_design_scores_gemma":[0.000106214946,0.0005978758,0.017172758,0.0002443376,0.0002642391,0.0048152865,0.0030374033,0.6500003,0.15697742,0.04937554,0.11710405,0.00030458954],"about_ca_topic_score_codex":0.0065716733,"about_ca_topic_score_gemma":0.008495467,"teacher_disagreement_score":0.007290996,"about_ca_system_score_codex":0.00059500115,"about_ca_system_score_gemma":0.0016771667,"threshold_uncertainty_score":0.024390817},"labels":[],"label_agreement":null},{"id":"W2251190814","doi":"10.18653/v1/w15-2709","title":"Idiom Paraphrases: Seventh Heaven vs Cloud Nine","year":2015,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"University of Sussex; York University","keywords":"Paraphrase; Identification (biology); Heaven; Computer science; Meaning (existential); Natural language processing; Artificial intelligence; Word (group theory); Domain (mathematical analysis); Space (punctuation); Linguistics; Psychology; Literature; Philosophy; Mathematics; Art","score_opus":0.023233788348324704,"score_gpt":0.2835585979350292,"score_spread":0.2603248095867045,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2251190814","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.75344855,0.0075472165,0.16521448,0.0037213485,0.00075683714,0.00050712156,0.022338657,0.0037761754,0.04268974],"genre_scores_gemma":[0.8513884,0.0010865324,0.1212459,0.00064414804,0.00012358138,0.00013547839,0.017986374,0.000478781,0.0069107693],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99862754,0.0003950714,0.00019828585,0.0003145404,0.00034048854,0.00012403609],"domain_scores_gemma":[0.99601585,0.0018266088,0.00040984934,0.0008676453,0.0007345879,0.00014543792],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008782838,0.00044054224,0.0005548452,0.0025317483,0.0010390537,0.0015699884,0.00080678676,0.00081486657,0.0054787053],"category_scores_gemma":[0.009352541,0.00017374184,0.00040956514,0.0023797792,0.0008043897,0.0028601587,0.0022226262,0.0011237689,0.0023360292],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002642257,0.0003094991,0.021220904,0.0021021739,0.00018727242,0.0031191155,0.0050175427,0.0042236815,0.056732178,0.05254991,0.07342323,0.77847236],"study_design_scores_gemma":[0.0003396685,0.0010376573,0.06490997,0.001149178,0.0004311477,0.018275533,0.010643435,0.17713377,0.14556588,0.13993543,0.4402961,0.00028220212],"about_ca_topic_score_codex":0.0013244281,"about_ca_topic_score_gemma":0.003494426,"teacher_disagreement_score":0.0054787053,"about_ca_system_score_codex":0.0005567882,"about_ca_system_score_gemma":0.0005814404,"threshold_uncertainty_score":0.01832813},"labels":[],"label_agreement":null},{"id":"W2251195625","doi":"10.18653/v1/w15-4204","title":"Sar-graphs: A Linked Linguistic Knowledge Resource Connecting Facts with Language","year":2015,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Banting and Best Diabetes Centre, University of Toronto; Bundesministerium für Bildung und Forschung","keywords":"Computer science; Resource (disambiguation); Linguistics; Natural language processing; Artificial intelligence","score_opus":0.024561392148326776,"score_gpt":0.29337081986991265,"score_spread":0.2688094277215859,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2251195625","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019675659,0.0014332759,0.52911246,0.0016602643,0.00024858204,0.00067172927,0.342106,0.05032466,0.05476731],"genre_scores_gemma":[0.10228579,0.0017999208,0.5148023,0.0006584222,0.000114712275,0.0010021364,0.3642325,0.004056811,0.01104736],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.998973,0.00021305382,0.00012452304,0.00034812785,0.00028046317,0.000060837425],"domain_scores_gemma":[0.9954984,0.0023192395,0.0004146114,0.0010541067,0.00054781657,0.00016583607],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00095492456,0.0011054351,0.00064538687,0.012355043,0.0012794915,0.002846543,0.001781489,0.001170204,0.026067318],"category_scores_gemma":[0.009798143,0.00087885594,0.0012063305,0.008676252,0.00077557313,0.008340391,0.003424069,0.0016187762,0.008626959],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00033136326,0.0002987349,0.010748029,0.003553511,0.00036529062,0.0015633041,0.003229816,0.014108879,0.0075095766,0.18667592,0.34319833,0.4284173],"study_design_scores_gemma":[0.000043828382,0.000043353062,0.00433947,0.0005362338,0.00014892707,0.0006439957,0.00095881714,0.028051734,0.006957585,0.18052812,0.77765036,0.000097580494],"about_ca_topic_score_codex":0.011511556,"about_ca_topic_score_gemma":0.028223084,"teacher_disagreement_score":0.026067318,"about_ca_system_score_codex":0.0012699667,"about_ca_system_score_gemma":0.00232429,"threshold_uncertainty_score":0.08720386},"labels":[],"label_agreement":null},{"id":"W2251202635","doi":"10.18653/v1/w15-3044","title":"Multi-level Evaluation for Machine Translation","year":2015,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Benchmarking; Computer science; Metric (unit); Machine translation; Variance (accounting); Translation (biology); Task (project management); Artificial intelligence; Variation (astronomy); Machine learning; Quality (philosophy); Correlation; Natural language processing; Data mining; Mathematics","score_opus":0.22863827713585122,"score_gpt":0.39581432284221973,"score_spread":0.16717604570636851,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2251202635","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06929811,0.004390586,0.9074314,0.00038935582,0.00021789849,0.00043185023,0.001408439,0.009677725,0.006754669],"genre_scores_gemma":[0.5737456,0.00038209904,0.41686785,0.000208933,0.00016004834,0.0005005555,0.004521359,0.0010331089,0.0025803817],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9814306,0.009330575,0.0013730265,0.00204988,0.005132713,0.00068316143],"domain_scores_gemma":[0.9781582,0.010517969,0.0017684178,0.0037196437,0.005079626,0.0007562228],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012174683,0.0015003642,0.0016627302,0.0048719463,0.0012190384,0.0035069128,0.0016166175,0.0018525954,0.0047734827],"category_scores_gemma":[0.02880629,0.00041079742,0.0013792916,0.0034348837,0.0010784284,0.0036338295,0.003368581,0.0015261816,0.002191642],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019652522,0.0005165861,0.017674701,0.0017729237,0.0011337713,0.0002993123,0.0006607344,0.097138256,0.04819579,0.019864237,0.020607412,0.7901711],"study_design_scores_gemma":[0.00010299811,0.0010758283,0.015285253,0.0001714231,0.00023516071,0.00036119085,0.0001905796,0.9178394,0.031443268,0.020318082,0.012840577,0.0001362975],"about_ca_topic_score_codex":0.0020235,"about_ca_topic_score_gemma":0.0035343075,"teacher_disagreement_score":0.012174683,"about_ca_system_score_codex":0.0018801078,"about_ca_system_score_gemma":0.0012933188,"threshold_uncertainty_score":0.064386606},"labels":[],"label_agreement":null},{"id":"W2251205163","doi":"10.18653/v1/w15-3404","title":"Projective methods for mining missing translations in DBpedia","year":2015,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"Fonds de recherche du Québec – Nature et technologies","keywords":"Computer science; String (physics); Task (project management); Baseline (sea); Annotation; Set (abstract data type); Information retrieval; RDF Schema; Property (philosophy); Semantic Web; Natural language processing; Artificial intelligence; RDF; SPARQL; Mathematics; Programming language","score_opus":0.09451784315911392,"score_gpt":0.43909625356383236,"score_spread":0.3445784104047184,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2251205163","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024813255,0.0014979964,0.96073824,0.0005194481,0.00013866447,0.00026644368,0.0041599222,0.0058563724,0.0020096696],"genre_scores_gemma":[0.19236563,0.0007268132,0.7855853,0.0004412469,0.0002046231,0.0005589964,0.017828166,0.000791085,0.0014981778],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9906178,0.0037481496,0.0009723041,0.0026150935,0.0016655964,0.00038106105],"domain_scores_gemma":[0.97650766,0.015182496,0.001581402,0.0040419893,0.0023711738,0.0003153272],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009209779,0.0020530098,0.0014999063,0.0054181353,0.001849144,0.0034029083,0.003761963,0.0019857234,0.0035952353],"category_scores_gemma":[0.0369605,0.0012348307,0.0021419034,0.0069080833,0.0019166443,0.006118694,0.00510348,0.0029646351,0.0025008223],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010039123,0.00061078096,0.015002449,0.002913516,0.0009597051,0.0019545075,0.0024737974,0.057594456,0.014909357,0.07223952,0.05046316,0.7798748],"study_design_scores_gemma":[0.00029946686,0.00041356112,0.006035932,0.00037258377,0.00038068384,0.0015351116,0.002171413,0.7524323,0.019366642,0.16785885,0.04895146,0.00018196716],"about_ca_topic_score_codex":0.0033834646,"about_ca_topic_score_gemma":0.006623112,"teacher_disagreement_score":0.009209779,"about_ca_system_score_codex":0.00084997265,"about_ca_system_score_gemma":0.002775581,"threshold_uncertainty_score":0.04870659},"labels":[],"label_agreement":null},{"id":"W2251211118","doi":"","title":"Combining Intra- and Multi-sentential Rhetorical Parsing for Document-level Discourse Analysis","year":2013,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":138,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Parsing; Computer science; Natural language processing; Artificial intelligence; Bottom-up parsing; Top-down parsing; Parser combinator; Rhetorical question; Set (abstract data type); S-attributed grammar; Margin (machine learning); Tree (set theory); Programming language; Linguistics; Machine learning","score_opus":0.03722023247038017,"score_gpt":0.33072461625782107,"score_spread":0.2935043837874409,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2251211118","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0033987106,0.00026328213,0.9905912,0.0003658351,0.000051420207,0.00009603317,0.00024823193,0.0040396824,0.00094560726],"genre_scores_gemma":[0.05329413,0.0001975184,0.9434916,0.00014771339,0.00010678196,0.00013167654,0.00085516664,0.0006733431,0.0011020921],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99338055,0.0035403876,0.00046257555,0.0012519396,0.0011035001,0.00026106002],"domain_scores_gemma":[0.9796419,0.013730582,0.00090530363,0.0028750729,0.0025223738,0.00032478533],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009100784,0.0021006714,0.0016576919,0.0050488813,0.001319063,0.005586059,0.002762497,0.0027430505,0.0040202397],"category_scores_gemma":[0.02076046,0.0012031443,0.001756972,0.0029801838,0.0019506856,0.0089914175,0.0040434008,0.0038439494,0.004796571],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023257216,0.00039544067,0.0035671794,0.000887069,0.0002662636,0.00040905908,0.0021712724,0.032148328,0.038681902,0.06505278,0.014578943,0.84160924],"study_design_scores_gemma":[0.000079385434,0.00014371377,0.001563285,0.00016739531,0.00021945313,0.00038837778,0.00042790303,0.73102623,0.062650226,0.16794783,0.035160262,0.00022587604],"about_ca_topic_score_codex":0.0025361327,"about_ca_topic_score_gemma":0.004515996,"teacher_disagreement_score":0.009100784,"about_ca_system_score_codex":0.0012627754,"about_ca_system_score_gemma":0.003684237,"threshold_uncertainty_score":0.048130095},"labels":[],"label_agreement":null},{"id":"W2251238100","doi":"","title":"Indexing Spoken Documents with Hierarchical Semantic Structures: Semantic Tree-to-string Alignment Models","year":2011,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Search engine indexing; String (physics); Tree (set theory); Information retrieval; Natural language processing; Index (typography); Artificial intelligence; Semantic computing; Hierarchical database model; Semantic Web; Data mining; World Wide Web","score_opus":0.020569973639011887,"score_gpt":0.2508908526425403,"score_spread":0.2303208790035284,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2251238100","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014738434,0.00051035965,0.9801524,0.0005145418,0.00008986108,0.00007137815,0.00043585632,0.0015447093,0.0019424494],"genre_scores_gemma":[0.35626638,0.0011534869,0.6337322,0.00036736758,0.0003378994,0.00038187712,0.0027757431,0.00055433065,0.00443068],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99784243,0.000904488,0.0001792649,0.00044741106,0.0004955491,0.00013084094],"domain_scores_gemma":[0.9931044,0.0042117583,0.00070140563,0.0010011714,0.00080821116,0.00017301636],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025231242,0.0005720549,0.0015031169,0.001634458,0.00096642453,0.0027010683,0.002250182,0.0017295971,0.0037365756],"category_scores_gemma":[0.014851687,0.00060117006,0.0010768669,0.005346143,0.0013409609,0.01238466,0.0016905803,0.002080619,0.0019680967],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007036465,0.00036508808,0.0032630938,0.00069449004,0.00016434495,0.00032750872,0.0014179167,0.22779752,0.009229918,0.3130099,0.01943435,0.42359227],"study_design_scores_gemma":[0.00002862151,0.0000741272,0.00039393394,0.000022147766,0.000025570811,0.000117484924,0.00014168333,0.8441151,0.0026771799,0.14862837,0.0037394767,0.000036334233],"about_ca_topic_score_codex":0.0048939576,"about_ca_topic_score_gemma":0.0046173413,"teacher_disagreement_score":0.0048939576,"about_ca_system_score_codex":0.0012673651,"about_ca_system_score_gemma":0.0017805727,"threshold_uncertainty_score":0.013343692},"labels":[],"label_agreement":null},{"id":"W2251262458","doi":"10.18653/v1/d13-1030","title":"Interpreting Anaphoric Shell Nouns using Antecedents of Cataphoric Shell Nouns as Training Data","year":2013,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Advanced Research Projects Agency; University of Toronto","keywords":"Noun; Computer science; Natural language processing; Shell (structure); Artificial intelligence; Baseline (sea); Nominalization; Training set; Obstacle; Natural (archaeology); Natural language; Range (aeronautics); Proper noun; History; Engineering","score_opus":0.05275107700764495,"score_gpt":0.31728390928216704,"score_spread":0.26453283227452207,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2251262458","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.25385502,0.003297522,0.6512447,0.0020068977,0.0007938468,0.0010461684,0.009406001,0.039388187,0.03896163],"genre_scores_gemma":[0.41562644,0.0011307774,0.55250543,0.0005149464,0.00018227538,0.00024168358,0.02053027,0.0013770381,0.007891103],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9931116,0.0024825383,0.00082092657,0.0023023458,0.0010409332,0.00024162003],"domain_scores_gemma":[0.9532418,0.02919514,0.002931499,0.008132261,0.006024273,0.00047505024],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009346027,0.0022939339,0.0012445363,0.0045765284,0.0019082369,0.0056862794,0.0020418318,0.002767068,0.007365198],"category_scores_gemma":[0.04133713,0.0012097078,0.0014482208,0.002334295,0.0011111044,0.007625485,0.0033530341,0.0030159685,0.0077928],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010730682,0.0006800688,0.060701586,0.0023347456,0.0005458595,0.00080981635,0.0039123073,0.011457391,0.059207056,0.014863706,0.031752832,0.8126616],"study_design_scores_gemma":[0.00031988398,0.0004994409,0.05843631,0.0015234826,0.00068051554,0.0033438974,0.006056514,0.439528,0.22192933,0.09871801,0.16859955,0.0003650696],"about_ca_topic_score_codex":0.0040724035,"about_ca_topic_score_gemma":0.010856239,"teacher_disagreement_score":0.009346027,"about_ca_system_score_codex":0.00086428784,"about_ca_system_score_gemma":0.0021834758,"threshold_uncertainty_score":0.049427092},"labels":[],"label_agreement":null},{"id":"W2251281667","doi":"10.3115/v1/p14-2138","title":"Does the Phonology of L1 Show Up in L2 Texts?","year":2014,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Innovates","keywords":"Bigram; Phonology; Computer science; Character (mathematics); Discriminative model; Natural language processing; Artificial intelligence; Task (project management); Word (group theory); Linguistics; Identification (biology); Language model; Mathematics; Trigram; Philosophy","score_opus":0.007043832333426662,"score_gpt":0.25139587543638997,"score_spread":0.2443520431029633,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2251281667","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9889388,0.00040089543,0.0049820873,0.00043894182,0.000031090836,0.000030525884,0.00032184419,0.00017232877,0.004683435],"genre_scores_gemma":[0.99676347,0.00014308371,0.0023308299,0.00008830993,0.00002894076,0.000014115042,0.00018763158,0.000056566612,0.0003870051],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99892265,0.00035706698,0.000090287904,0.00038441637,0.0001284015,0.00011725481],"domain_scores_gemma":[0.9744279,0.01867467,0.0027960986,0.0019138313,0.0013053026,0.00088221097],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016347917,0.00042836132,0.0005479115,0.0012563684,0.0005714851,0.0023337498,0.00045992676,0.0012317067,0.0035609216],"category_scores_gemma":[0.019210624,0.00036216757,0.00027939148,0.00097282697,0.0011331894,0.0036728664,0.0007328399,0.0010714394,0.002113728],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002722424,0.00052738586,0.3451949,0.0013776942,0.00022091068,0.0013748578,0.006108379,0.0010068427,0.3400803,0.0014005713,0.0012680651,0.2987177],"study_design_scores_gemma":[0.00009263758,0.0009973792,0.88670915,0.00014687344,0.00017926506,0.0022737568,0.005131878,0.011406316,0.086202584,0.0034710648,0.0032639466,0.00012516102],"about_ca_topic_score_codex":0.0011017302,"about_ca_topic_score_gemma":0.0015295438,"teacher_disagreement_score":0.0035609216,"about_ca_system_score_codex":0.0003633801,"about_ca_system_score_gemma":0.00030663982,"threshold_uncertainty_score":0.011912525},"labels":[],"label_agreement":null},{"id":"W2251302843","doi":"","title":"Graph Propagation for Paraphrasing Out-of-Vocabulary Words in Statistical Machine Translation","year":2013,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":36,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Natural language processing; Phrase; Artificial intelligence; Vocabulary; Machine translation; Graph; Translation (biology); Theoretical computer science; Linguistics","score_opus":0.02043648907779574,"score_gpt":0.29375780919302025,"score_spread":0.27332132011522453,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2251302843","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01172065,0.0007264048,0.9813586,0.00022979871,0.000076468,0.0001571999,0.00027173755,0.004290743,0.0011684713],"genre_scores_gemma":[0.18846413,0.0010985289,0.80186325,0.00032573566,0.00018116143,0.00037872905,0.0027229826,0.0010727408,0.0038927926],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.997924,0.0010959647,0.00010928757,0.00032684545,0.00045902128,0.000084760235],"domain_scores_gemma":[0.9912951,0.0060021784,0.0004600483,0.0011211444,0.0010157343,0.000105822095],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024029422,0.001512409,0.0009868128,0.0035300127,0.0009758566,0.00095539365,0.0018956889,0.0016224433,0.0028943936],"category_scores_gemma":[0.009762242,0.000702402,0.0010513434,0.0038284913,0.0013172519,0.003539058,0.0015956783,0.0016759878,0.0019592815],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036039768,0.000303338,0.001169843,0.00056573475,0.0001725122,0.00040342845,0.0003563236,0.13664126,0.023643265,0.017788958,0.016006405,0.8025885],"study_design_scores_gemma":[0.00006103963,0.00012840131,0.0004597607,0.000029415114,0.00007712999,0.00022211435,0.00006245653,0.937514,0.016082171,0.03940868,0.0059149377,0.00003984787],"about_ca_topic_score_codex":0.0064683817,"about_ca_topic_score_gemma":0.01293412,"teacher_disagreement_score":0.0064683817,"about_ca_system_score_codex":0.000822683,"about_ca_system_score_gemma":0.0012247744,"threshold_uncertainty_score":0.01286149},"labels":[],"label_agreement":null},{"id":"W2251306619","doi":"","title":"Using Other Learner Corpora in the 2013 NLI Shared Task","year":2013,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Task (project management); Domain adaptation; Natural language processing; Artificial intelligence; Training set; Adaptation (eye); Simple (philosophy); Domain (mathematical analysis); Task analysis; Speech recognition; Psychology","score_opus":0.040346983024968454,"score_gpt":0.2907303994468658,"score_spread":0.2503834164218973,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2251306619","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.22888356,0.0031539996,0.41861147,0.010413748,0.0060079284,0.0068700304,0.08098919,0.069734775,0.17533527],"genre_scores_gemma":[0.31571236,0.00059065304,0.3987622,0.003906649,0.0007255916,0.0068219407,0.22267714,0.011412084,0.03939141],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.97488344,0.013815747,0.0016228071,0.0045576184,0.0042431206,0.0008771811],"domain_scores_gemma":[0.9539856,0.018807279,0.0010974394,0.014527742,0.009518716,0.0020632157],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.024372548,0.002176433,0.0018761238,0.0028807577,0.004912216,0.004101674,0.0039709816,0.004329826,0.025145775],"category_scores_gemma":[0.050609056,0.0012716497,0.0016013004,0.003952057,0.0020927857,0.009337034,0.010951388,0.00911049,0.01732181],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002544106,0.0041447957,0.012869472,0.0023185166,0.00051648234,0.002343663,0.0061995205,0.035773087,0.027949918,0.019902127,0.45308226,0.43235603],"study_design_scores_gemma":[0.0014178674,0.0012352043,0.015692925,0.0006806029,0.0003007472,0.0026026445,0.0044377097,0.13099433,0.08889573,0.025115743,0.7278528,0.0007736754],"about_ca_topic_score_codex":0.008236472,"about_ca_topic_score_gemma":0.015327692,"teacher_disagreement_score":0.025145775,"about_ca_system_score_codex":0.0025874157,"about_ca_system_score_gemma":0.005121712,"threshold_uncertainty_score":0.12889588},"labels":[],"label_agreement":null},{"id":"W2251314350","doi":"10.63317/4sbnbg6dq83u","title":"Hashtag Occurrences, Layout and Translation: A Corpus-driven Analysis of Tweets Published by the Canadian Government","year":2014,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Machine translation; Government (linguistics); Pipeline (software); Social media; Translation (biology); Lemmatisation; Precision and recall; Quality (philosophy); Information retrieval; Linguistics; World Wide Web","score_opus":0.011587525342687907,"score_gpt":0.23795188638153478,"score_spread":0.22636436103884688,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2251314350","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.899799,0.000949496,0.0046145595,0.00088206405,0.00011159009,0.00031515848,0.08462223,0.0007763322,0.007929672],"genre_scores_gemma":[0.8740082,0.0008000444,0.013321815,0.00016673538,0.000071513394,0.00028143157,0.10176553,0.0003555102,0.009229328],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9987941,0.00014607649,0.00005687787,0.00017313659,0.0005696353,0.00026012058],"domain_scores_gemma":[0.99288684,0.002158157,0.0004259976,0.00034726577,0.0038241553,0.0003576484],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007895401,0.00047598363,0.00042403652,0.00548451,0.0033401088,0.0015832207,0.001025762,0.0005964297,0.0022790898],"category_scores_gemma":[0.0063307397,0.0003620286,0.000504167,0.010339027,0.0011448534,0.0007612025,0.000916646,0.00089811435,0.0013403262],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015457721,0.0006293111,0.5192971,0.0019011723,0.000437391,0.0025600267,0.034675956,0.009173509,0.059575953,0.0083938,0.11837333,0.24343663],"study_design_scores_gemma":[0.00004429164,0.000102848055,0.8350875,0.00014805204,0.00022350397,0.00067419175,0.02044537,0.02992112,0.012342962,0.0012126561,0.0996164,0.00018116689],"about_ca_topic_score_codex":0.9146298,"about_ca_topic_score_gemma":0.9588214,"teacher_disagreement_score":0.08537018,"about_ca_system_score_codex":0.009224563,"about_ca_system_score_gemma":0.019205414,"threshold_uncertainty_score":0.17174584},"labels":[],"label_agreement":null},{"id":"W2251385272","doi":"","title":"Cross-lingual Discourse Relation Analysis: A corpus study and a semi-supervised classification system","year":2014,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Relation (database); Natural language processing; Artificial intelligence; Divergence (linguistics); Task (project management); Linguistics; Discourse analysis; Annotation; Corpus linguistics","score_opus":0.016873456085195405,"score_gpt":0.3060699483211083,"score_spread":0.28919649223591287,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2251385272","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.77576435,0.0014706677,0.20344199,0.0007065613,0.0002016438,0.0012218447,0.003173354,0.0031457504,0.010874004],"genre_scores_gemma":[0.7132918,0.00039080487,0.27340022,0.00011504541,0.000115780436,0.0011677295,0.0068009854,0.00030162936,0.00441604],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99533975,0.0023487897,0.00042793897,0.0011359896,0.00063713605,0.00011036291],"domain_scores_gemma":[0.9808725,0.011186796,0.001000903,0.0025561098,0.0039592995,0.00042434403],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0040923855,0.0005533635,0.00056728325,0.0035952555,0.0022853406,0.0018298546,0.0011868983,0.0008407784,0.0024819968],"category_scores_gemma":[0.01234172,0.00039727782,0.000458155,0.0032217873,0.0013285961,0.0031336187,0.0021890493,0.0011756815,0.00096047815],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010415736,0.0022172017,0.048585366,0.0014639157,0.00030132494,0.0021218152,0.01605533,0.007916707,0.13901097,0.0084037995,0.01649349,0.7563886],"study_design_scores_gemma":[0.0004133364,0.0013393561,0.14428405,0.0004176694,0.0006015988,0.004085861,0.019312548,0.5145898,0.21656114,0.010075557,0.08790667,0.00041245433],"about_ca_topic_score_codex":0.006798897,"about_ca_topic_score_gemma":0.010922997,"teacher_disagreement_score":0.006798897,"about_ca_system_score_codex":0.0010695858,"about_ca_system_score_gemma":0.0013806676,"threshold_uncertainty_score":0.021642864},"labels":[],"label_agreement":null},{"id":"W2251404283","doi":"10.63317/3o7gcvqkiaid","title":"Morphological parsing of Swahili using crowdsourced lexical resources","year":2014,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Swahili; Computer science; Parsing; Natural language processing; Artificial intelligence; Resource (disambiguation); Declaration; Lexical analysis; World Wide Web; Linguistics; Programming language","score_opus":0.02418380819301267,"score_gpt":0.2815513555588706,"score_spread":0.25736754736585793,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2251404283","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.35233533,0.00092710054,0.4024629,0.001658047,0.000609237,0.0012729556,0.08174259,0.057941012,0.101050876],"genre_scores_gemma":[0.4290669,0.00040966662,0.42397037,0.00035351628,0.00009550899,0.00053250627,0.11854431,0.006310333,0.020716967],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99915874,0.00014717734,0.00011050294,0.00029060442,0.00020468503,0.00008830402],"domain_scores_gemma":[0.99704856,0.001004591,0.00015564922,0.0006943708,0.0010030497,0.0000937213],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008456712,0.0008787036,0.0007797197,0.004925024,0.0018126286,0.0024015047,0.0013592268,0.0010704289,0.016356338],"category_scores_gemma":[0.003915164,0.00069946813,0.00079255155,0.0033893501,0.0006155848,0.0029068582,0.0027540016,0.0012113365,0.009517275],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00090686156,0.00066152797,0.015957871,0.0025173353,0.00020954451,0.0044442974,0.012061404,0.0076868646,0.09083089,0.038680397,0.17147973,0.65456325],"study_design_scores_gemma":[0.00022421777,0.00027314015,0.035460584,0.0010096154,0.00036911623,0.0024890576,0.016607862,0.21091993,0.15454836,0.067351244,0.5103724,0.00037448795],"about_ca_topic_score_codex":0.012345573,"about_ca_topic_score_gemma":0.031291336,"teacher_disagreement_score":0.016356338,"about_ca_system_score_codex":0.0008990587,"about_ca_system_score_gemma":0.002470365,"threshold_uncertainty_score":0.05471742},"labels":[],"label_agreement":null},{"id":"W2251445713","doi":"","title":"Capturing syntactico-semantic regularities among terms: An application of the FrameNet methodology to terminology","year":2012,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"FrameNet; Computer science; Terminology; Annotation; Focus (optics); Natural language processing; Artificial intelligence; Point (geometry); Resource (disambiguation); Lexicon; Computational linguistics; Semantics (computer science); Linguistics; Parsing; Programming language","score_opus":0.03784322549439183,"score_gpt":0.3385109969376508,"score_spread":0.30066777144325896,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2251445713","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03556982,0.0005616429,0.9552153,0.00063761347,0.00006405993,0.0003107327,0.0013795337,0.0008436173,0.0054177474],"genre_scores_gemma":[0.34298027,0.0006445834,0.6509994,0.00012266416,0.00009275464,0.0003823841,0.0028016346,0.0004037556,0.0015724885],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99304855,0.0030874289,0.0007766063,0.0010150187,0.0017360752,0.00033637224],"domain_scores_gemma":[0.9868073,0.007427548,0.001155709,0.0019125951,0.0023856855,0.00031126765],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008608997,0.00093339523,0.0016129145,0.009330446,0.0019354278,0.0047161463,0.0021974163,0.0012935319,0.0030477028],"category_scores_gemma":[0.02592189,0.00068961387,0.001542614,0.007895039,0.0025541587,0.013722197,0.003695484,0.0016695061,0.0005761269],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037805812,0.00021276136,0.009640882,0.0007437462,0.00020574783,0.00044882175,0.003538319,0.02434735,0.011048693,0.6117911,0.0048611583,0.3327834],"study_design_scores_gemma":[0.000039172377,0.00010628677,0.0035336635,0.00027346742,0.00028787734,0.00040402415,0.0016770195,0.2925686,0.011033356,0.6658524,0.024118556,0.00010560426],"about_ca_topic_score_codex":0.013845355,"about_ca_topic_score_gemma":0.016418558,"teacher_disagreement_score":0.013845355,"about_ca_system_score_codex":0.0027597593,"about_ca_system_score_gemma":0.0033064743,"threshold_uncertainty_score":0.045529246},"labels":[],"label_agreement":null},{"id":"W2251497744","doi":"10.3115/v1/w15-1518","title":"Morpho-syntactic Regularities in Continuous Word Representations: A multilingual study.","year":2015,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Word2vec; German; Natural language processing; Artificial intelligence; Word (group theory); Set (abstract data type); Linguistics; Morpho; Programming language","score_opus":0.03663931931208418,"score_gpt":0.3308042806045714,"score_spread":0.2941649612924872,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2251497744","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98832405,0.0005651645,0.0060454016,0.00012511469,0.00002949047,0.000024311466,0.0013903582,0.00014194101,0.003354176],"genre_scores_gemma":[0.9947344,0.00012246495,0.0027980905,0.000043452947,0.000015716765,0.00002398796,0.0015015478,0.00010253439,0.0006577768],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9987349,0.0005011956,0.00012999473,0.00040954954,0.00015504625,0.00006929232],"domain_scores_gemma":[0.98804945,0.0075429864,0.001234577,0.0019068855,0.0009979829,0.00026806784],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014602622,0.00048464438,0.00038856716,0.0011713855,0.0005187953,0.0019444858,0.00053990143,0.00058981084,0.0044812867],"category_scores_gemma":[0.018505966,0.00033291517,0.00045151633,0.0016794333,0.0009828227,0.0028767951,0.0016367774,0.0013881094,0.0013437516],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0040692524,0.0024271104,0.31829172,0.0023227953,0.0016563518,0.0027801783,0.031963322,0.008373311,0.13933122,0.010881118,0.015177953,0.4627257],"study_design_scores_gemma":[0.00075794425,0.0029386715,0.74622774,0.00044577583,0.0013280902,0.011041115,0.022620594,0.07914817,0.06122497,0.03318926,0.04048723,0.0005904751],"about_ca_topic_score_codex":0.0057525723,"about_ca_topic_score_gemma":0.005765528,"teacher_disagreement_score":0.0057525723,"about_ca_system_score_codex":0.00044311496,"about_ca_system_score_gemma":0.00030216685,"threshold_uncertainty_score":0.014991403},"labels":[],"label_agreement":null},{"id":"W2251547673","doi":"10.63317/54n4hrwwu9u9","title":"Measuring Interlanguage: Native Language Identification with L1-influence Metrics","year":2012,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Bigram; Natural language processing; Artificial intelligence; Interlanguage; Identification (biology); Language identification; Word (group theory); Machine translation; Task (project management); Natural language; Linguistics","score_opus":0.01889991639026808,"score_gpt":0.2762415163566559,"score_spread":0.25734159996638784,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2251547673","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.73760283,0.0015593495,0.23377804,0.000261107,0.00010447955,0.000318557,0.0019421062,0.0034253527,0.021008184],"genre_scores_gemma":[0.9425898,0.00020796435,0.052440006,0.000070658265,0.000081170874,0.00026924832,0.0020197534,0.0006404348,0.001681082],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9925483,0.002730041,0.00070065295,0.0014719577,0.0021930418,0.000356011],"domain_scores_gemma":[0.96343344,0.020798594,0.0036640363,0.003293955,0.007278503,0.0015314106],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0061008227,0.0010220128,0.0008026674,0.005994332,0.0014092696,0.0025214625,0.0007896383,0.0010740003,0.0022165026],"category_scores_gemma":[0.0432542,0.00037643363,0.00048316605,0.0032479868,0.0014521796,0.004769775,0.002939338,0.0011647319,0.0017594914],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009342331,0.0007908223,0.28563094,0.0010832415,0.00047597286,0.0006293607,0.008173423,0.011336864,0.047654815,0.00445943,0.007868769,0.6309621],"study_design_scores_gemma":[0.00010759862,0.0015422392,0.49946764,0.00021500308,0.00039208872,0.0032958903,0.004500233,0.28588852,0.16417871,0.014604745,0.025347501,0.00045975012],"about_ca_topic_score_codex":0.0028193614,"about_ca_topic_score_gemma":0.004426499,"teacher_disagreement_score":0.0061008227,"about_ca_system_score_codex":0.0007921767,"about_ca_system_score_gemma":0.00079491903,"threshold_uncertainty_score":0.03226459},"labels":[],"label_agreement":null},{"id":"W2251568075","doi":"10.3115/v1/s14-2030","title":"CNRC-TMT: Second Language Writing Assistant System Description","year":2014,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Computer science; Machine translation; Sentence; Task (project management); Natural language processing; Phrase; Writing system; Context (archaeology); Artificial intelligence; Word (group theory); SemEval; Linguistics","score_opus":0.015593061491684792,"score_gpt":0.23864574130455846,"score_spread":0.22305267981287366,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2251568075","genre_codex":"software","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016384957,0.0017536934,0.27935356,0.0015614552,0.00093120604,0.0034425117,0.061995756,0.5773797,0.05719709],"genre_scores_gemma":[0.103773326,0.0009041822,0.5585031,0.0024146307,0.00040719044,0.0052717784,0.21753155,0.025594376,0.0855998],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9984415,0.00017071879,0.00018755802,0.00048492607,0.000566736,0.000148481],"domain_scores_gemma":[0.99851793,0.00017379202,0.0000867864,0.0003760006,0.0006503104,0.00019521263],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014255397,0.0013118215,0.0012497444,0.0014857408,0.00083640986,0.0029391842,0.0030425205,0.0017085017,0.040351983],"category_scores_gemma":[0.0041222493,0.0008383644,0.0005952782,0.0013676175,0.0005576392,0.0023240733,0.0018534096,0.002003791,0.05913977],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007540118,0.00035341913,0.0024213747,0.0011857018,0.00006250402,0.00079303695,0.00042840294,0.0024834538,0.0505882,0.0058302567,0.6200615,0.31503806],"study_design_scores_gemma":[0.00032190155,0.0004292487,0.003835773,0.00014114121,0.00005870408,0.0038500933,0.00014189917,0.05484505,0.07131911,0.0030063628,0.86178905,0.0002617805],"about_ca_topic_score_codex":0.00809633,"about_ca_topic_score_gemma":0.0073056724,"teacher_disagreement_score":0.040351983,"about_ca_system_score_codex":0.0016046995,"about_ca_system_score_gemma":0.0032286241,"threshold_uncertainty_score":0.13499087},"labels":[],"label_agreement":null},{"id":"W2251734840","doi":"10.18653/v1/w15-3911","title":"Multiple System Combination for Transliteration","year":2015,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Innovates","keywords":"Transliteration; Computer science; Task (project management); Focus (optics); Natural language processing; Context (archaeology); Artificial intelligence; Base (topology); Information retrieval; Speech recognition","score_opus":0.02684884921905284,"score_gpt":0.2751991885393128,"score_spread":0.24835033932025996,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2251734840","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2628267,0.010669907,0.56872034,0.0020517614,0.0031115694,0.0022631418,0.008267735,0.1180069,0.024081867],"genre_scores_gemma":[0.43742874,0.0010879347,0.51905465,0.0010127035,0.0005013138,0.0012331901,0.021184698,0.005510365,0.012986342],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.98845553,0.004329016,0.0009604538,0.0034662955,0.0021194639,0.00066920323],"domain_scores_gemma":[0.9850006,0.0059050783,0.00043632725,0.0053677387,0.0027802852,0.0005099269],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007580494,0.0048708254,0.0036666947,0.0026170935,0.0018761131,0.0026144288,0.002692677,0.0029512867,0.012482084],"category_scores_gemma":[0.017326087,0.0013761072,0.0029552586,0.003236463,0.0008384618,0.004877449,0.005080208,0.0048673917,0.01167501],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.003082284,0.00096898456,0.007912364,0.0033960007,0.0027202305,0.00139493,0.00096343976,0.040110245,0.07842906,0.0017472414,0.05896752,0.80030763],"study_design_scores_gemma":[0.0012068722,0.004068077,0.011436529,0.00039040897,0.004274902,0.0045803664,0.0011764285,0.48752406,0.336383,0.0108031,0.13744707,0.0007093036],"about_ca_topic_score_codex":0.0036613073,"about_ca_topic_score_gemma":0.008417112,"teacher_disagreement_score":0.012482084,"about_ca_system_score_codex":0.0008922856,"about_ca_system_score_gemma":0.0023073426,"threshold_uncertainty_score":0.04175669},"labels":[],"label_agreement":null},{"id":"W2251747132","doi":"","title":"Automatically Assessing Free Texts","year":2012,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Lethbridge","funders":"","keywords":"Grading (engineering); Computer science; Task (project management); Natural language processing; Automation; Latent semantic analysis; Artificial intelligence; Set (abstract data type); Training set; Process (computing); Machine learning; Information retrieval; Engineering; Programming language","score_opus":0.014939720462263632,"score_gpt":0.29894106486639643,"score_spread":0.2840013444041328,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2251747132","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.46706918,0.0025002328,0.46824348,0.00094366487,0.00068602874,0.0016814263,0.016767094,0.022724567,0.019384347],"genre_scores_gemma":[0.6646468,0.00058034714,0.2804246,0.00021267982,0.0004228389,0.00084166817,0.03830641,0.0011576923,0.013406963],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9869034,0.0050217975,0.0012207536,0.002823866,0.00368324,0.00034694196],"domain_scores_gemma":[0.9283223,0.041811876,0.005112959,0.0061637545,0.01709582,0.0014933103],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005363866,0.0014555161,0.0012391206,0.007592632,0.0009850811,0.004158668,0.0017245477,0.0019319674,0.0077309464],"category_scores_gemma":[0.053949624,0.00038452534,0.0006440572,0.0032279675,0.000656768,0.005209467,0.0024048816,0.0012895291,0.0065058046],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008543468,0.0005861608,0.018849175,0.0015189328,0.00017269078,0.0003720784,0.001467924,0.006866951,0.064482145,0.0038289109,0.026326055,0.8746747],"study_design_scores_gemma":[0.00041966102,0.0017003064,0.14534494,0.00062441087,0.0003271835,0.0018955766,0.0039993026,0.31273577,0.29512924,0.04777125,0.1895561,0.0004962597],"about_ca_topic_score_codex":0.00091078214,"about_ca_topic_score_gemma":0.0016560174,"teacher_disagreement_score":0.0077309464,"about_ca_system_score_codex":0.0007562754,"about_ca_system_score_gemma":0.0010770649,"threshold_uncertainty_score":0.028367162},"labels":[],"label_agreement":null},{"id":"W2251786640","doi":"10.3115/v1/p15-2108","title":"Lexical Comparison Between Wikipedia and Twitter Corpora by Using Word Embeddings","year":2015,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":35,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Natural language processing; Computational linguistics; Zhàng; Artificial intelligence; Volume (thermodynamics); Word (group theory); Joint (building); Association (psychology); Linguistics; History; Engineering; China; Psychology; Philosophy","score_opus":0.07377720145457405,"score_gpt":0.3422680295978668,"score_spread":0.2684908281432927,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2251786640","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.57636446,0.0071999044,0.115481086,0.0021590807,0.004516806,0.00082662015,0.23860723,0.011338715,0.043506153],"genre_scores_gemma":[0.46313712,0.0017358309,0.11450944,0.00031625427,0.00045434787,0.0011921399,0.40810964,0.001987094,0.008558184],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99760145,0.0006393876,0.0005340501,0.00049858686,0.0005715789,0.00015491563],"domain_scores_gemma":[0.9932243,0.0021876183,0.00055731164,0.0010729962,0.0026520186,0.0003056954],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001679275,0.0006594348,0.00064154604,0.012078779,0.0012583331,0.0023417897,0.0008195454,0.000631495,0.0055645267],"category_scores_gemma":[0.013270203,0.00038503172,0.00069651334,0.010272702,0.00042931407,0.005464577,0.0029548388,0.000735158,0.0054166657],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021313066,0.0008161738,0.05461894,0.005539371,0.0012690218,0.0019586247,0.0041753342,0.004094017,0.05703493,0.02075142,0.29115003,0.55646074],"study_design_scores_gemma":[0.0005470785,0.00078994886,0.15446696,0.001555965,0.0011882191,0.0028849735,0.022725405,0.09881511,0.055271223,0.041787308,0.6194624,0.00050540187],"about_ca_topic_score_codex":0.003448709,"about_ca_topic_score_gemma":0.011032696,"teacher_disagreement_score":0.012078779,"about_ca_system_score_codex":0.000548081,"about_ca_system_score_gemma":0.0010650742,"threshold_uncertainty_score":0.018615246},"labels":[],"label_agreement":null},{"id":"W2251810465","doi":"","title":"Improving AMBER, an MT Evaluation Metric","year":2012,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Metric (unit); Computer science; Task (project management); Machine translation; Translation (biology); Simplex; Artificial intelligence; Algorithm; Pattern recognition (psychology); Mathematics; Chemistry; Combinatorics; Engineering","score_opus":0.026918997588655222,"score_gpt":0.3156656344312011,"score_spread":0.28874663684254587,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2251810465","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09915528,0.019170474,0.8069521,0.0019925253,0.0020978595,0.00089502777,0.0052593164,0.02403344,0.040443968],"genre_scores_gemma":[0.31866693,0.0018764841,0.65259314,0.00083524536,0.0007461153,0.00067662064,0.008290177,0.0029270826,0.013388334],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9810184,0.008748253,0.0019005514,0.001991057,0.00595033,0.0003913885],"domain_scores_gemma":[0.9773731,0.009764036,0.0013174436,0.0034925314,0.007545879,0.00050693285],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009864446,0.0020661259,0.0022393356,0.006411909,0.0015509288,0.0028602758,0.0016421472,0.0020147574,0.0039378977],"category_scores_gemma":[0.041777696,0.0004566869,0.0008384898,0.003759219,0.0009199787,0.005800717,0.0028345494,0.002214738,0.0029297462],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000563381,0.0002565363,0.004127717,0.0010103441,0.00041663588,0.00015573385,0.0003712438,0.03391292,0.01736244,0.012511074,0.043363906,0.88594806],"study_design_scores_gemma":[0.00034758268,0.0026054312,0.016874494,0.00032936284,0.0004763561,0.0023417873,0.00033262421,0.7248832,0.07495703,0.044408944,0.13192338,0.00051980076],"about_ca_topic_score_codex":0.0039155227,"about_ca_topic_score_gemma":0.007519195,"teacher_disagreement_score":0.009864446,"about_ca_system_score_codex":0.0017092696,"about_ca_system_score_gemma":0.0016102876,"threshold_uncertainty_score":0.052168846},"labels":[],"label_agreement":null},{"id":"W2251816551","doi":"","title":"The mathematics of language learning","year":2013,"lang":"en","type":"article","venue":"Työväentutkimus Vuosikirja","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Artificial intelligence; Parsing; Natural language processing; Syntax; Variety (cybernetics); Deep learning; Machine learning","score_opus":0.006887951944767457,"score_gpt":0.24850236511829077,"score_spread":0.2416144131735233,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2251816551","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007375707,0.10175015,0.4756768,0.09697962,0.005896768,0.00011931808,0.0016988029,0.001166624,0.30933627],"genre_scores_gemma":[0.5131716,0.08725207,0.22177829,0.023665791,0.0268871,0.0010372298,0.0030039812,0.001181482,0.12202245],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9966066,0.0013608199,0.00024454412,0.0007965737,0.000820881,0.00017072256],"domain_scores_gemma":[0.99530834,0.0032903014,0.00019918819,0.00066964887,0.00040905311,0.0001233847],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031557637,0.0012201611,0.0012658843,0.001773107,0.001545091,0.006248474,0.001640832,0.0022071945,0.02001391],"category_scores_gemma":[0.011524586,0.00051853695,0.0012137752,0.0014805383,0.009428635,0.011115585,0.004566586,0.0062886737,0.007408629],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00000842041,0.000008588134,0.00014355993,0.00014867829,0.000024784275,0.000034081626,0.0001798695,0.0011513025,0.00010273227,0.95338416,0.015870824,0.028942855],"study_design_scores_gemma":[0.0000047445487,0.000006494195,0.00007028998,0.00005855823,0.0000031302152,0.000036747377,0.00002519776,0.0016097717,0.000052649033,0.9588016,0.0393228,0.000007921556],"about_ca_topic_score_codex":0.0017194485,"about_ca_topic_score_gemma":0.00089168,"teacher_disagreement_score":0.02001391,"about_ca_system_score_codex":0.002783129,"about_ca_system_score_gemma":0.0016362899,"threshold_uncertainty_score":0.06695318},"labels":[],"label_agreement":null},{"id":"W2251832051","doi":"","title":"Kriya - The SFU System for Translation Task at WMT-12","year":2012,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Czech; Computer science; Phrase; Machine translation; Decoding methods; Natural language processing; Task (project management); Pipeline (software); Baseline (sea); Artificial intelligence; BLEU; Speech translation; Machine translation system; Speech recognition; Translation (biology); Linguistics; Algorithm; Programming language; Engineering","score_opus":0.02463293999319214,"score_gpt":0.27177678896220786,"score_spread":0.24714384896901573,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2251832051","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12970345,0.0039362935,0.2962871,0.0030821834,0.0096726185,0.0066754417,0.18900211,0.23968023,0.1219606],"genre_scores_gemma":[0.203248,0.0007532648,0.27742615,0.00075148855,0.0009359484,0.0031813886,0.38817415,0.034967285,0.09056228],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99326634,0.001986483,0.00067663135,0.0014349205,0.0016645353,0.00097109994],"domain_scores_gemma":[0.99139506,0.00078129204,0.00018455503,0.0022296072,0.004456831,0.000952663],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006733649,0.0031426207,0.0025293555,0.002622224,0.0022856144,0.0033812819,0.0022183983,0.0021406373,0.045330774],"category_scores_gemma":[0.0125714755,0.0011064918,0.0012314821,0.0023061442,0.0006463373,0.0030277285,0.0036278837,0.0030541515,0.07370263],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0026397535,0.0006070726,0.0025482194,0.001817941,0.00031074468,0.0008220934,0.0012695243,0.00480874,0.07789873,0.0035125744,0.5697732,0.33399144],"study_design_scores_gemma":[0.0013107354,0.0020540154,0.011507348,0.0002485866,0.00029594664,0.0019237075,0.0010342514,0.04269086,0.213897,0.005440946,0.71905756,0.000539109],"about_ca_topic_score_codex":0.0058254413,"about_ca_topic_score_gemma":0.008260004,"teacher_disagreement_score":0.045330774,"about_ca_system_score_codex":0.0013629424,"about_ca_system_score_gemma":0.0038846522,"threshold_uncertainty_score":0.1516465},"labels":[],"label_agreement":null},{"id":"W2251870912","doi":"10.18653/v1/d15-1014","title":"Indicative Tweet Generation: An Extractive Summarization Problem?","year":2015,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Automatic summarization; Computer science; Natural language processing; Information retrieval","score_opus":0.04805614236140524,"score_gpt":0.3124861268716186,"score_spread":0.2644299845102133,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2251870912","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.026115118,0.0015300066,0.9541404,0.007530417,0.00074559305,0.00066775316,0.0019670709,0.0025729402,0.004730676],"genre_scores_gemma":[0.23827638,0.0019055755,0.7399809,0.0017501685,0.0017715057,0.0012057525,0.004847304,0.0011786068,0.0090839],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9924695,0.003419979,0.0008782336,0.0012940951,0.0017085068,0.00022967496],"domain_scores_gemma":[0.9278754,0.051594958,0.0053071915,0.008210514,0.0064865923,0.0005253263],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008535461,0.0014025879,0.0014489271,0.003076899,0.0017221629,0.003773891,0.0021091308,0.0028652875,0.0031350579],"category_scores_gemma":[0.07863252,0.0008050553,0.00093023496,0.003977204,0.001516254,0.005795159,0.0019485571,0.0025334938,0.002472742],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005715172,0.00033181772,0.007424409,0.002111884,0.0003034179,0.0021386845,0.007447708,0.024495983,0.025016153,0.08998825,0.05260794,0.7875623],"study_design_scores_gemma":[0.00025345947,0.00039504396,0.004975282,0.00085270114,0.00038685784,0.0027719678,0.005105029,0.40147045,0.070866965,0.33036327,0.1822691,0.00028991836],"about_ca_topic_score_codex":0.0012378637,"about_ca_topic_score_gemma":0.0015643322,"teacher_disagreement_score":0.008535461,"about_ca_system_score_codex":0.0008910829,"about_ca_system_score_gemma":0.0012490959,"threshold_uncertainty_score":0.045140386},"labels":[],"label_agreement":null},{"id":"W2251940628","doi":"10.3115/v1/w14-5706","title":"A Comparative Study of Different Classification Methods for the Identification of Brazilian Portuguese Multiword Expressions","year":2014,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Brazilian Portuguese; Computer science; Identification (biology); Portuguese; Artificial intelligence; Natural language processing; Linguistics","score_opus":0.06705337433864526,"score_gpt":0.42410453352915384,"score_spread":0.35705115919050856,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2251940628","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.43598005,0.011141831,0.52400124,0.0011554656,0.0005840414,0.00089950557,0.003166462,0.0077618514,0.015309571],"genre_scores_gemma":[0.47659203,0.004036389,0.50644785,0.0001740353,0.00018279815,0.0007256131,0.006147883,0.00102408,0.0046693855],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9914106,0.004258231,0.0009683068,0.0010567252,0.0020022418,0.00030385135],"domain_scores_gemma":[0.96871823,0.024269188,0.0011019792,0.0012512505,0.004305319,0.000353958],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008605555,0.0014446767,0.0009896377,0.0102187,0.00094841345,0.002325698,0.0010773102,0.0008758215,0.0022982748],"category_scores_gemma":[0.027739774,0.0003539889,0.0009110534,0.0063163126,0.0006127669,0.0033328482,0.0012464469,0.00094254414,0.0020754654],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009602896,0.00027904357,0.017655388,0.002222701,0.00029081432,0.00024145501,0.0024607207,0.0025511526,0.0208605,0.0018403007,0.0036431265,0.9469946],"study_design_scores_gemma":[0.0005425768,0.0022165554,0.18159312,0.0021402119,0.0016658423,0.005433056,0.016378522,0.5010725,0.18494985,0.009263527,0.09405794,0.0006863164],"about_ca_topic_score_codex":0.0028740647,"about_ca_topic_score_gemma":0.0041983547,"teacher_disagreement_score":0.0102187,"about_ca_system_score_codex":0.00065337954,"about_ca_system_score_gemma":0.001193501,"threshold_uncertainty_score":0.045511067},"labels":[],"label_agreement":null},{"id":"W2251951653","doi":"","title":"The Impact of Deep Hierarchical Discourse Structures in the Evaluation of Text Coherence","year":2014,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":66,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Treebank; Computer science; Coherence (philosophical gambling strategy); Rhetorical question; Style (visual arts); Natural language processing; Discourse analysis; Linguistics; Artificial intelligence; Set (abstract data type); Parsing; Philosophy; Mathematics; Literature","score_opus":0.02087804179287861,"score_gpt":0.3689828923904542,"score_spread":0.34810485059757557,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2251951653","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.690662,0.00313277,0.28838828,0.0016266763,0.0002007731,0.00056767836,0.0015605773,0.0050603617,0.008800855],"genre_scores_gemma":[0.9302685,0.00018659643,0.067209065,0.000115280505,0.000034444984,0.00011359124,0.0010506271,0.00019942559,0.0008224412],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9887398,0.007671378,0.000547953,0.001359243,0.0013873065,0.0002942825],"domain_scores_gemma":[0.9246325,0.06350557,0.0027795492,0.0040340223,0.0038006431,0.0012476824],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009946513,0.0014304679,0.00079590175,0.002510895,0.00092741684,0.0029171463,0.0010467352,0.0017720264,0.0022869546],"category_scores_gemma":[0.059683144,0.0004910715,0.00046900916,0.001363476,0.0013138928,0.0078617,0.0030560235,0.0022991016,0.00049909693],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0067780777,0.0014680516,0.05085728,0.002345594,0.0008255286,0.000393243,0.0061222133,0.13011612,0.09462155,0.015555933,0.00813956,0.68277687],"study_design_scores_gemma":[0.0003155793,0.0022878232,0.018938549,0.0002792256,0.000304199,0.00019655927,0.0015997242,0.907706,0.0418807,0.022113374,0.0042134216,0.00016484245],"about_ca_topic_score_codex":0.00509445,"about_ca_topic_score_gemma":0.009079055,"teacher_disagreement_score":0.009946513,"about_ca_system_score_codex":0.001424779,"about_ca_system_score_gemma":0.0012381432,"threshold_uncertainty_score":0.052602828},"labels":[],"label_agreement":null},{"id":"W2251970502","doi":"10.63317/37r3xzsc3g5u","title":"Linked Open Data and Web Corpus Data for noun compound bracketing","year":2014,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure","funders":"","keywords":"Bracketing (phenomenology); Computer science; Linked data; Task (project management); Noun; World Wide Web; Information retrieval; Open data; Resource (disambiguation); Natural language processing; Semantic Web","score_opus":0.08505609058429106,"score_gpt":0.3571034207995201,"score_spread":0.2720473302152291,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2251970502","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10901196,0.001918035,0.56159604,0.003516161,0.0009918052,0.0010490157,0.25496635,0.036825977,0.030124601],"genre_scores_gemma":[0.24188225,0.00096912414,0.41129667,0.00034913814,0.000256838,0.001287924,0.3295449,0.003316951,0.011096174],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99577737,0.0010887962,0.00053984096,0.00082613906,0.0015748078,0.00019306682],"domain_scores_gemma":[0.9832644,0.007155842,0.0009450834,0.004655032,0.0034153576,0.0005642635],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003667364,0.0006068167,0.0007518246,0.00881945,0.0020393499,0.0036678882,0.0013436357,0.0013362182,0.011457337],"category_scores_gemma":[0.024547778,0.00067520904,0.0010160906,0.008706836,0.001192066,0.007112958,0.0035287098,0.0024266036,0.00701322],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016102033,0.0009307498,0.024582626,0.002272279,0.0002729857,0.0017969662,0.004328197,0.014864987,0.027004506,0.23520327,0.19759779,0.48953545],"study_design_scores_gemma":[0.00019812453,0.00021002674,0.023588894,0.00062698877,0.00022773481,0.0012664434,0.0036087118,0.23505072,0.0464273,0.18011163,0.5083754,0.00030801533],"about_ca_topic_score_codex":0.01474185,"about_ca_topic_score_gemma":0.023788193,"teacher_disagreement_score":0.01474185,"about_ca_system_score_codex":0.0015062268,"about_ca_system_score_gemma":0.0040090387,"threshold_uncertainty_score":0.038328588},"labels":[{"model":"gemma","categories":[],"domain":null,"study_design":"bench_or_experimental","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"medium"},{"model":"gpt","categories":[],"domain":null,"study_design":"observational","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"low"}],"label_agreement":"split"},{"id":"W2251989420","doi":"","title":"Towards a Hybrid Rule-based and Statistical Arabic-French Machine Translation System","year":2013,"lang":"en","type":"article","venue":"Recent Advances in Natural Language Processing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Machine translation; Computer science; Natural language processing; Artificial intelligence; Phrase; Arabic; Machine translation software usability; Example-based machine translation; Rule-based machine translation; Evaluation of machine translation; Transfer-based machine translation; Computer-assisted translation; Translation (biology); Synchronous context-free grammar; Scheme (mathematics); BLEU; Modern Standard Arabic; Quality (philosophy); Linguistics","score_opus":0.0067527020823977485,"score_gpt":0.2728687706527462,"score_spread":0.2661160685703484,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2251989420","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.030804338,0.00057517656,0.94333273,0.00029422817,0.00015757987,0.00022220523,0.0004929796,0.020503594,0.0036170997],"genre_scores_gemma":[0.15866129,0.00026640255,0.8334328,0.00029828073,0.00011748061,0.00023872702,0.0016088007,0.00033146143,0.005044699],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993771,0.00014400527,0.00009140412,0.00019086497,0.00016124395,0.00003550105],"domain_scores_gemma":[0.999201,0.00017127345,0.000049977265,0.00010094896,0.0004360989,0.000040642975],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010120074,0.0007286949,0.0008844508,0.0010317911,0.00062961754,0.0014411316,0.0008685353,0.0009949697,0.0024578646],"category_scores_gemma":[0.0013021365,0.0003812601,0.0006612215,0.00075597485,0.00036247825,0.0009271137,0.00068029884,0.00065768906,0.0036968146],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037588403,0.0003149104,0.0029539242,0.00033163233,0.00028211108,0.0011027467,0.0002447088,0.044646848,0.2002144,0.009609201,0.011496195,0.72842747],"study_design_scores_gemma":[0.00007764564,0.00028947133,0.002161464,0.00004480887,0.00022093717,0.00096409686,0.00009473598,0.8816633,0.083215095,0.0052043535,0.025986187,0.00007785668],"about_ca_topic_score_codex":0.003873786,"about_ca_topic_score_gemma":0.0035653168,"teacher_disagreement_score":0.003873786,"about_ca_system_score_codex":0.00040287754,"about_ca_system_score_gemma":0.0009951224,"threshold_uncertainty_score":0.008222342},"labels":[],"label_agreement":null},{"id":"W2251994480","doi":"10.18653/v1/k15-1012","title":"Cross-lingual Transfer for Unsupervised Dependency Parsing Without Parallel Data","year":2015,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":46,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of New Brunswick","funders":"University of Melbourne; National ICT Australia","keywords":"Treebank; Computer science; Dependency grammar; Natural language processing; Artificial intelligence; Parsing; Dependency (UML); Word (group theory); Transfer of learning; Linguistics","score_opus":0.11854537142404836,"score_gpt":0.37999803309476393,"score_spread":0.26145266167071557,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2251994480","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029622173,0.0003610993,0.928773,0.00032661032,0.00017085644,0.00027014667,0.0018107609,0.03227475,0.006390623],"genre_scores_gemma":[0.28183645,0.00034953453,0.68560654,0.00055155077,0.00013240526,0.00077948614,0.014000436,0.0051206117,0.01162288],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9978672,0.00078205223,0.00013542948,0.00078554294,0.0002719783,0.00015780021],"domain_scores_gemma":[0.99541986,0.0018495885,0.00016920487,0.0018602645,0.0006230184,0.000078153294],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034043787,0.0015820777,0.0009895891,0.001941425,0.0012346198,0.0011960018,0.00204286,0.0015356746,0.0091757765],"category_scores_gemma":[0.0073246956,0.00097595516,0.0015212672,0.0022760287,0.0011574759,0.0053753234,0.0043067676,0.0034152693,0.0069715553],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035112508,0.0005916256,0.003216812,0.0005006974,0.00039572763,0.00054237625,0.0007057848,0.043865573,0.034649774,0.01975367,0.038386565,0.8570402],"study_design_scores_gemma":[0.00013556804,0.00021732245,0.0041301465,0.00009153191,0.00017970779,0.00063664996,0.00041518177,0.7725747,0.07604198,0.10231972,0.04311115,0.00014627315],"about_ca_topic_score_codex":0.003482299,"about_ca_topic_score_gemma":0.007969746,"teacher_disagreement_score":0.0091757765,"about_ca_system_score_codex":0.00096991926,"about_ca_system_score_gemma":0.0018676397,"threshold_uncertainty_score":0.030696034},"labels":[],"label_agreement":null},{"id":"W2252045241","doi":"","title":"Flexible Structural Analysis of Near-Meet-Semilattices for Typed Unification-Based Grammar Design","year":2012,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Unification; Parsing; Computer science; Programming language; Head-driven phrase structure grammar; Grammar; Rule-based machine translation; Type (biology); Theoretical computer science; Algorithm; Natural language processing; Artificial intelligence; Generative grammar; Linguistics","score_opus":0.07928709696875265,"score_gpt":0.3722236565658698,"score_spread":0.2929365595971172,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2252045241","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005357228,0.00003409848,0.9925493,0.00008194221,0.000021967522,0.00004446135,0.00004784645,0.0005551083,0.0013081343],"genre_scores_gemma":[0.12889004,0.000074489675,0.8677093,0.0000639966,0.000032382606,0.00018862315,0.00026116104,0.00048847473,0.0022915725],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99708086,0.0008593098,0.0003103642,0.0003754041,0.0011751412,0.00019887532],"domain_scores_gemma":[0.9971706,0.0013128625,0.00019675161,0.0006031904,0.00055155705,0.0001650349],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035646579,0.0006699596,0.00063666404,0.0011983196,0.0013144398,0.002395982,0.0018613152,0.0008448368,0.0050290795],"category_scores_gemma":[0.0064921863,0.001015898,0.0018995444,0.0009352793,0.0024871188,0.0040143942,0.0025501992,0.002240656,0.0013641807],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009931222,0.00008744186,0.000889023,0.00020172048,0.000037770988,0.00040245088,0.001029485,0.06096716,0.017751954,0.8418993,0.0018212374,0.074813165],"study_design_scores_gemma":[0.000023872544,0.000040048857,0.00010241621,0.000046580328,0.000023167562,0.000119842356,0.00015932546,0.33300376,0.011209428,0.64597666,0.009261044,0.000033888366],"about_ca_topic_score_codex":0.0020105345,"about_ca_topic_score_gemma":0.003660196,"teacher_disagreement_score":0.0050290795,"about_ca_system_score_codex":0.0015543543,"about_ca_system_score_gemma":0.0023884007,"threshold_uncertainty_score":0.018851936},"labels":[],"label_agreement":null},{"id":"W2252096820","doi":"10.3115/v1/w14-3420","title":"Using statistical parsing to detect agrammatic aphasia","year":2014,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Baycrest Hospital; University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada; National Institutes of Health","keywords":"Aphasia; Computer science; Parsing; Natural language processing; Artificial intelligence; Classifier (UML); Context (archaeology); Grammar; Top-down parsing; Agrammatism; Speech recognition; Linguistics; Psychology; Cognitive psychology; Sentence","score_opus":0.025397210026510797,"score_gpt":0.3184345147845668,"score_spread":0.293037304758056,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2252096820","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8520819,0.0005660436,0.13660832,0.00023796364,0.000058231522,0.0001399654,0.0014595088,0.004458583,0.0043896497],"genre_scores_gemma":[0.91654164,0.0001900475,0.08058849,0.00009021468,0.000034444,0.00007020976,0.001697142,0.00019453838,0.00059325626],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9993697,0.00019500534,0.000074157,0.00014509821,0.00016453798,0.000051636696],"domain_scores_gemma":[0.9950858,0.0033716469,0.0005457231,0.00025439012,0.00063941226,0.00010308957],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00073669245,0.0004652053,0.00037332,0.0027872862,0.0002815749,0.00085237937,0.00034017695,0.0004887911,0.00080315437],"category_scores_gemma":[0.005836294,0.0001665376,0.00027456382,0.0012344762,0.00031171256,0.0005811664,0.0004178968,0.0003769247,0.00051398686],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00049568387,0.0003049262,0.21611407,0.00026041907,0.0002964632,0.0019660536,0.00083626516,0.012093438,0.14816332,0.0017198004,0.005465134,0.6122844],"study_design_scores_gemma":[0.00008340003,0.0008080892,0.46433154,0.00010150097,0.00033528,0.008493569,0.0008413214,0.4008711,0.099304006,0.015924145,0.008668806,0.00023727005],"about_ca_topic_score_codex":0.0012639114,"about_ca_topic_score_gemma":0.001969356,"teacher_disagreement_score":0.0027872862,"about_ca_system_score_codex":0.00015446673,"about_ca_system_score_gemma":0.00055389066,"threshold_uncertainty_score":0.0038960576},"labels":[],"label_agreement":null},{"id":"W2252107976","doi":"10.63317/4ts6k3m8qm7t","title":"An Iterative Approach for Mining Parallel Sentences in a Comparable Corpus","year":2014,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Machine translation; Sentence; Bootstrapping (finance); Classifier (UML); Natural language processing; Artificial intelligence; Focus (optics); Parallel corpora","score_opus":0.02557996846621315,"score_gpt":0.2952854227501761,"score_spread":0.269705454283963,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2252107976","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.035743333,0.00068372063,0.94971365,0.00042626043,0.00015368428,0.0014562092,0.0031208897,0.004932338,0.0037698625],"genre_scores_gemma":[0.07518191,0.0002335713,0.91281545,0.00015267401,0.00011834186,0.0008358928,0.0076292655,0.00045433678,0.0025784033],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99340606,0.0018055381,0.000917867,0.0019975386,0.00163971,0.00023320885],"domain_scores_gemma":[0.97995526,0.010946723,0.00090868113,0.0024772256,0.005341171,0.00037110926],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0051206383,0.0014344853,0.0014272294,0.007601019,0.0023708267,0.0030694224,0.0035286695,0.0018302301,0.007208447],"category_scores_gemma":[0.025050115,0.0012462521,0.001990117,0.006874331,0.0011966315,0.004007243,0.0037483769,0.0025052784,0.0038785841],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00088311743,0.00087384204,0.008102516,0.001989779,0.0006910899,0.0017593238,0.0035958842,0.008737008,0.08459437,0.02110161,0.019126687,0.8485447],"study_design_scores_gemma":[0.0006099489,0.0018983369,0.021428838,0.000611765,0.0024072041,0.0066161044,0.0054845456,0.6107505,0.10023019,0.12567793,0.123835914,0.00044876392],"about_ca_topic_score_codex":0.0045638233,"about_ca_topic_score_gemma":0.011151988,"teacher_disagreement_score":0.007601019,"about_ca_system_score_codex":0.00086890615,"about_ca_system_score_gemma":0.005202767,"threshold_uncertainty_score":0.027080834},"labels":[],"label_agreement":null},{"id":"W2252149878","doi":"10.3115/v1/w14-2503","title":"Sociolinguistics for Computational Social Science","year":2014,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Sociolinguistics; Computer science; Linguistics","score_opus":0.015382959414539469,"score_gpt":0.31927956967011417,"score_spread":0.3038966102555747,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2252149878","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0068304287,0.048945796,0.8318307,0.024597706,0.0027883141,0.00042983497,0.0032939909,0.0024815653,0.07880167],"genre_scores_gemma":[0.21060477,0.029050902,0.71861625,0.0045188414,0.006087763,0.001864088,0.0057729743,0.0008537693,0.02263065],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.997329,0.0013956917,0.00019803972,0.0005513777,0.00044681213,0.00007902096],"domain_scores_gemma":[0.99181104,0.005733011,0.00047843342,0.0012565788,0.00046842857,0.0002524546],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0040509063,0.0013972771,0.0018899974,0.0060542356,0.002002691,0.0064809485,0.0028101397,0.0021109672,0.022969728],"category_scores_gemma":[0.012355077,0.00058624917,0.002237475,0.0055947322,0.0073933075,0.007845652,0.0045805234,0.0041570277,0.006879118],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000011543908,0.000031371277,0.00055651594,0.0003969737,0.000055937675,0.00015772461,0.00068153837,0.00234728,0.00016014579,0.9436708,0.009593589,0.042336483],"study_design_scores_gemma":[0.000004732292,0.0000066582684,0.00023120058,0.00009532023,0.0000065455147,0.00005305036,0.00017404565,0.005413183,0.000030645213,0.9465444,0.04743013,0.000010165803],"about_ca_topic_score_codex":0.0041581546,"about_ca_topic_score_gemma":0.0029813913,"teacher_disagreement_score":0.022969728,"about_ca_system_score_codex":0.0028416621,"about_ca_system_score_gemma":0.0023720542,"threshold_uncertainty_score":0.076841354},"labels":[],"label_agreement":null},{"id":"W2252217503","doi":"","title":"Simple or Complex? Classifying Questions by Answering Complexity","year":2012,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Lethbridge","funders":"","keywords":"Simple (philosophy); Computer science; Question answering; Artificial intelligence; Natural language processing; Machine learning; Information retrieval","score_opus":0.08240711857575736,"score_gpt":0.34465438156235134,"score_spread":0.26224726298659395,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2252217503","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.45013696,0.0017270298,0.51515704,0.004522748,0.00029569422,0.001215064,0.0036808955,0.002406384,0.02085816],"genre_scores_gemma":[0.82963467,0.00052934536,0.16219011,0.0005075371,0.0005073781,0.0004714642,0.0030190155,0.0002117548,0.002928763],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9950216,0.001383653,0.0006312319,0.0011134517,0.0014787628,0.0003713058],"domain_scores_gemma":[0.969887,0.02079125,0.0033559366,0.0016249511,0.003273757,0.0010670399],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034592885,0.00082576927,0.00086615136,0.0050042835,0.0006655621,0.0041037253,0.0010269353,0.0017556052,0.0054711164],"category_scores_gemma":[0.036310524,0.00027900125,0.001129918,0.0024987867,0.0016010262,0.006644908,0.0018443896,0.001696261,0.0014229299],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011604198,0.00071973645,0.1800417,0.0013491849,0.00047324743,0.0005370979,0.004951833,0.0123472875,0.049102653,0.068412915,0.01375277,0.66715115],"study_design_scores_gemma":[0.00020293106,0.00069094374,0.24292505,0.0004442354,0.0005485596,0.0020187425,0.005685481,0.37770674,0.034965422,0.2910207,0.043396425,0.00039482827],"about_ca_topic_score_codex":0.0012156513,"about_ca_topic_score_gemma":0.0010259848,"teacher_disagreement_score":0.0054711164,"about_ca_system_score_codex":0.00092276995,"about_ca_system_score_gemma":0.0007300079,"threshold_uncertainty_score":0.018302679},"labels":[],"label_agreement":null},{"id":"W2252267789","doi":"10.3115/v1/p14-1048","title":"A Linear-Time Bottom-Up Discourse Parser with Constraints and Post-Editing","year":2014,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":152,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada; University of Toronto; University of Pennsylvania","keywords":"Computer science; Top-down and bottom-up design; Parsing; Programming language","score_opus":0.006619083121696493,"score_gpt":0.2536527261051099,"score_spread":0.24703364298341343,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2252267789","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0096444525,0.00036384698,0.9609017,0.0005694328,0.000114696406,0.00011385193,0.0007723237,0.025151826,0.0023679296],"genre_scores_gemma":[0.17036207,0.00039332476,0.81623286,0.00034253107,0.000119090626,0.00015054882,0.0026916794,0.0022459654,0.007461896],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99894434,0.00028499856,0.00007805104,0.00038369515,0.00021403175,0.00009503925],"domain_scores_gemma":[0.9953458,0.002728277,0.000180197,0.001068267,0.00057540735,0.00010206029],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001622251,0.0013486319,0.0016867346,0.00069065334,0.001012291,0.002200913,0.0029751742,0.0018731118,0.008187864],"category_scores_gemma":[0.0057345405,0.0011639993,0.0012550899,0.0011190819,0.0007363296,0.004812293,0.0012808442,0.0032277696,0.008325867],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003572341,0.0002704397,0.0017572062,0.000945,0.00016015534,0.0006374493,0.0006110719,0.09852849,0.06103119,0.028170472,0.04508465,0.76244664],"study_design_scores_gemma":[0.000041264422,0.00007147738,0.0005323785,0.00005034129,0.000100282356,0.00038197855,0.0001295712,0.9112642,0.03709122,0.02811832,0.022156402,0.00006258124],"about_ca_topic_score_codex":0.005570561,"about_ca_topic_score_gemma":0.011868669,"teacher_disagreement_score":0.008187864,"about_ca_system_score_codex":0.00090624404,"about_ca_system_score_gemma":0.0037418245,"threshold_uncertainty_score":0.027391136},"labels":[],"label_agreement":null},{"id":"W2252272516","doi":"10.18653/v1/w15-3014","title":"Montreal Neural Machine Translation Systems for WMT’15","year":2015,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":144,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Canadian Institute for Advanced Research; Université de Montréal","funders":"Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs; Samsung; Compute Canada; Türkiye Bilimsel ve Teknolojik Araştırma Kurumu; Canadian Institute for Advanced Research","keywords":"Machine translation; Leverage (statistics); Computer science; German; Natural language processing; Artificial intelligence; Variety (cybernetics); Machine translation system; Language model; Linguistics","score_opus":0.03653425763530201,"score_gpt":0.2899250259595661,"score_spread":0.2533907683242641,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2252272516","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03170266,0.0132067595,0.3671429,0.0176073,0.017884638,0.0034997584,0.17387739,0.19256923,0.18250932],"genre_scores_gemma":[0.10844444,0.0022132508,0.37942088,0.001533993,0.001334056,0.0018967206,0.33084467,0.010525059,0.1637869],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9965012,0.0009238405,0.0002322179,0.00083132676,0.0011307639,0.00038064],"domain_scores_gemma":[0.99638796,0.00032134907,0.00013087531,0.0011093476,0.0016617221,0.0003887071],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0051310966,0.0028678281,0.0015462347,0.0025884903,0.0022601204,0.0032070621,0.003204464,0.0021546932,0.05912855],"category_scores_gemma":[0.008223002,0.00091679965,0.001387419,0.0029405986,0.00078088336,0.00337715,0.0036317345,0.0026345344,0.035108007],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025698447,0.00009767219,0.00044170427,0.00034147466,0.00014133773,0.0002643974,0.00014027074,0.0062616616,0.0048190122,0.014036074,0.8187854,0.15441406],"study_design_scores_gemma":[0.0003379863,0.00028272957,0.0019022549,0.00013283343,0.00008884415,0.00036604327,0.00010287073,0.10286987,0.018020066,0.023058008,0.85268366,0.0001548006],"about_ca_topic_score_codex":0.07787136,"about_ca_topic_score_gemma":0.12689881,"teacher_disagreement_score":0.07787136,"about_ca_system_score_codex":0.004895196,"about_ca_system_score_gemma":0.008212027,"threshold_uncertainty_score":0.19780469},"labels":[],"label_agreement":null},{"id":"W2252548980","doi":"","title":"Les atouts multiples de la lemmatisation : l'exemple du latin","year":2002,"lang":"fr","type":"preprint","venue":"Open Repository and Bibliography (University of Liège)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Linguistic Association","funders":"","keywords":"Lemmatisation; Parsing; Computer science; Natural language processing; Artificial intelligence; Linguistics; Humanities; Philosophy","score_opus":0.03031034496217131,"score_gpt":0.2557886265537648,"score_spread":0.22547828159159347,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2252548980","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07353265,0.005970225,0.3055807,0.019047974,0.0054975697,0.00062674505,0.002332333,0.008721789,0.57869005],"genre_scores_gemma":[0.47900867,0.0040510413,0.15713137,0.0019865897,0.0024966265,0.00069989776,0.002173145,0.007089328,0.34536326],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9848763,0.006232081,0.0010767002,0.0018196688,0.0053390423,0.0006561274],"domain_scores_gemma":[0.96710503,0.018039735,0.0017229915,0.004781785,0.007502267,0.00084829034],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0054640975,0.0008259963,0.00092773716,0.0028315252,0.004342782,0.0066179596,0.0013088426,0.0023752917,0.03897906],"category_scores_gemma":[0.032628946,0.00071233226,0.0008278062,0.003942303,0.0031374088,0.0061580585,0.002341755,0.0027779487,0.014029808],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008112686,0.00022541622,0.0041755354,0.00089843536,0.00010004917,0.0065690856,0.012649189,0.00228843,0.0268322,0.52912414,0.08708059,0.3292456],"study_design_scores_gemma":[0.00009939522,0.00011124594,0.0021290625,0.00020386408,0.000045234206,0.0024603442,0.00199989,0.0025856665,0.010588723,0.031789787,0.94789714,0.00008964021],"about_ca_topic_score_codex":0.00953915,"about_ca_topic_score_gemma":0.010616684,"teacher_disagreement_score":0.03897906,"about_ca_system_score_codex":0.0028560683,"about_ca_system_score_gemma":0.0048068273,"threshold_uncertainty_score":0.13039792},"labels":[],"label_agreement":null},{"id":"W2256934714","doi":"10.18329/09757597/2015/8101","title":"Biolinguistics, Natural Language Processing, and Digital Libraries","year":2015,"lang":"en","type":"article","venue":"World Digital Libraries - An international journal","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Natural (archaeology); Computer science; Natural language processing; Linguistics; History; Philosophy; Archaeology","score_opus":0.016768895136141708,"score_gpt":0.2752886984733457,"score_spread":0.258519803337204,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2256934714","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.053094804,0.12815635,0.2782549,0.070304975,0.0011809182,0.00014055347,0.00030692676,0.0018569416,0.4667036],"genre_scores_gemma":[0.6542985,0.051988624,0.22585982,0.0074267653,0.0014947053,0.0002370678,0.00035607105,0.000416624,0.05792177],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99800414,0.0011828188,0.00013241441,0.0002054892,0.00037566968,0.000099372635],"domain_scores_gemma":[0.99443686,0.0041063237,0.00055419805,0.0003144692,0.00036560104,0.00022256881],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021423001,0.00043007656,0.00044724403,0.0038945524,0.0023693836,0.008937512,0.0009869307,0.0010883796,0.007074523],"category_scores_gemma":[0.00589099,0.0002557527,0.00027553513,0.0052117435,0.011350739,0.009643293,0.0022325453,0.001447335,0.0014288975],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000039385774,0.000053487263,0.0012082797,0.0004627628,0.000014135821,0.00026604178,0.00298331,0.0008326196,0.0007267351,0.8679507,0.006426851,0.11903571],"study_design_scores_gemma":[0.000006255218,0.000018324228,0.0014054518,0.00027421897,0.000011559448,0.0005802434,0.0037589858,0.0019631656,0.00090254523,0.8650593,0.12598783,0.000032007618],"about_ca_topic_score_codex":0.0050739674,"about_ca_topic_score_gemma":0.00628,"teacher_disagreement_score":0.008937512,"about_ca_system_score_codex":0.003589014,"about_ca_system_score_gemma":0.0029026403,"threshold_uncertainty_score":0.026040256},"labels":[],"label_agreement":null},{"id":"W2257177366","doi":"","title":"Re-Discussion on Defining Standards of Chinese Noun-Quantity Compound Word","year":2016,"lang":"en","type":"article","venue":"Cross-cultural communication","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Compound; Noun; Word (group theory); Linguistics; Proper noun; Computer science; Natural language processing; Style (visual arts); Word formation; History; Philosophy","score_opus":0.02044474904959632,"score_gpt":0.3646146679132648,"score_spread":0.3441699188636685,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2257177366","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13669752,0.029381454,0.436289,0.06862807,0.012970867,0.0008196606,0.0016172817,0.0012724127,0.3123237],"genre_scores_gemma":[0.76751393,0.019799313,0.15430342,0.008478113,0.002626016,0.0008274367,0.0024059445,0.00076949154,0.043276362],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9940753,0.0013877514,0.0012261845,0.000861446,0.002143083,0.00030622442],"domain_scores_gemma":[0.99275655,0.0013506475,0.00032501068,0.0006865185,0.0046067885,0.00027444548],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005133539,0.00079526065,0.00047171186,0.003350761,0.0029370065,0.0035611263,0.0015661467,0.0012575998,0.004017808],"category_scores_gemma":[0.01085498,0.00030579642,0.00066046446,0.0035404556,0.0048582978,0.01152162,0.0021371709,0.0034683445,0.0010512633],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007131364,0.000033603592,0.0062139123,0.0009137352,0.000021861144,0.00044213168,0.01320185,0.00082821306,0.0048228386,0.7770092,0.029152168,0.16728917],"study_design_scores_gemma":[0.00004119593,0.00011832907,0.016332688,0.0010565729,0.00008034878,0.001841005,0.019176392,0.007239122,0.010322012,0.23871626,0.70486313,0.00021298196],"about_ca_topic_score_codex":0.017264817,"about_ca_topic_score_gemma":0.009117165,"teacher_disagreement_score":0.017264817,"about_ca_system_score_codex":0.004082302,"about_ca_system_score_gemma":0.00806391,"threshold_uncertainty_score":0.03432864},"labels":[],"label_agreement":null},{"id":"W2258219731","doi":"10.5539/elt.v9n3p13","title":"A Comparative Study of Google Translate Translations: An Error Analysis of English-to-Persian and Persian-to-English Translations","year":2016,"lang":"en","type":"article","venue":"English Language Teaching","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":51,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Persian; Natural language processing; Interpreter; Computer science; Linguistics; Machine translation; Artificial intelligence; Error analysis; Test (biology); Mathematics","score_opus":0.01995651727704178,"score_gpt":0.3242397793253058,"score_spread":0.304283262048264,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2258219731","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99114394,0.0007179103,0.0041683195,0.000099996956,0.000062342406,0.00007361482,0.00092253875,0.00017664969,0.0026346233],"genre_scores_gemma":[0.9868506,0.0004272769,0.0076526576,0.000050698134,0.00004265963,0.00008234929,0.002910239,0.00020433044,0.0017790806],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9905172,0.0035611275,0.0016592302,0.00093204994,0.0030299616,0.00030043817],"domain_scores_gemma":[0.9305812,0.04537516,0.0050690444,0.0028022034,0.015733529,0.0004389114],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0043051094,0.0006756117,0.00065923034,0.0054159206,0.0008239253,0.0011115893,0.0005609412,0.0005949192,0.0012743851],"category_scores_gemma":[0.038905866,0.0002119488,0.0006866784,0.006751158,0.0010827375,0.0018164841,0.0010925725,0.00045925795,0.0006769345],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0071603428,0.00096031267,0.38674235,0.0045419363,0.0008352903,0.008949894,0.10053329,0.006907404,0.047611408,0.0025826865,0.007727273,0.42544776],"study_design_scores_gemma":[0.00016589588,0.0040300745,0.7985297,0.00044258774,0.00079820707,0.014318518,0.056048147,0.028960409,0.07172713,0.002105438,0.022556199,0.0003177546],"about_ca_topic_score_codex":0.0037501082,"about_ca_topic_score_gemma":0.004907089,"teacher_disagreement_score":0.0054159206,"about_ca_system_score_codex":0.0006016576,"about_ca_system_score_gemma":0.00069367816,"threshold_uncertainty_score":0.022767842},"labels":[],"label_agreement":null},{"id":"W2258967370","doi":"10.1515/cog-2015-0101","title":"Machine Meets Man: Evaluating the Psychological Reality of Corpus-based Probabilistic Models","year":2016,"lang":"en","type":"article","venue":"Cognitive Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":54,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Probabilistic logic; Natural language processing; Verb; Context (archaeology); Artificial intelligence; Linguistics; Selection (genetic algorithm); Statistical model; Empirical research; Psychology; Statistics; Mathematics","score_opus":0.10545473174629237,"score_gpt":0.39953899459145875,"score_spread":0.29408426284516637,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2258967370","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.87149346,0.0009385842,0.11529594,0.00367118,0.0001296592,0.00017974946,0.000343909,0.00026183162,0.0076857177],"genre_scores_gemma":[0.97726256,0.00013266657,0.02195768,0.0001123211,0.000051212282,0.00010073238,0.00018409322,0.000035477125,0.00016326344],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9803017,0.0169853,0.000395801,0.0011668717,0.0009871862,0.00016307249],"domain_scores_gemma":[0.6557133,0.32588613,0.005526656,0.008119784,0.0036055164,0.0011485763],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.045330353,0.00072584464,0.00079258514,0.0020721317,0.00127887,0.005348981,0.001736501,0.0016875699,0.0032318425],"category_scores_gemma":[0.19615617,0.0006979546,0.00075836823,0.0016721572,0.0044721044,0.0066264197,0.003148952,0.0026256507,0.00021128498],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0024727415,0.00081239524,0.10866515,0.0006149836,0.0013277909,0.0003228736,0.0058645788,0.65967965,0.0017812188,0.10865963,0.0035938777,0.10620509],"study_design_scores_gemma":[0.000060639668,0.0002311844,0.007984671,0.000052515858,0.000055156997,0.00007692595,0.0006469058,0.93552005,0.0004831786,0.05409605,0.00074143504,0.000051350675],"about_ca_topic_score_codex":0.004516687,"about_ca_topic_score_gemma":0.0046662856,"teacher_disagreement_score":0.045330353,"about_ca_system_score_codex":0.0027785841,"about_ca_system_score_gemma":0.00085584924,"threshold_uncertainty_score":0.23973274},"labels":[],"label_agreement":null},{"id":"W2260214588","doi":"","title":"Reordering: a stepping-stone to perfect Thai Sign generation","year":2007,"lang":"en","type":"article","venue":"Computational intelligence","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Computer science; Sign language; Sentence; Grammar; Sign (mathematics); Natural language processing; Phrase; Matching (statistics); Vocabulary; Artificial intelligence; Code (set theory); Programming language; Speech recognition; Linguistics; Mathematics","score_opus":0.04082978311332475,"score_gpt":0.33334147363280214,"score_spread":0.2925116905194774,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2260214588","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01030231,0.00013585697,0.9804684,0.00014458245,0.0001371401,0.0002190381,0.000080834376,0.0041073505,0.0044044093],"genre_scores_gemma":[0.11180659,0.00021183786,0.8782057,0.00017069979,0.00005146074,0.000121534606,0.0005192867,0.0010018514,0.007911014],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99877137,0.00030884784,0.00012056882,0.00021925334,0.0004749405,0.00010495109],"domain_scores_gemma":[0.9981395,0.000336462,0.00011640598,0.0005971459,0.00071754673,0.000093012146],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011959843,0.00083011546,0.0006708564,0.00091863767,0.00065876055,0.001376878,0.001161082,0.00066900987,0.0054859333],"category_scores_gemma":[0.0031206652,0.00044010632,0.0006935037,0.0006603682,0.0010575554,0.001747014,0.0014559701,0.0011739028,0.0024871577],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002589413,0.00009084313,0.0014610329,0.00034225883,0.00004315606,0.00076583104,0.001440572,0.014652266,0.089346156,0.041268438,0.0075004757,0.8428301],"study_design_scores_gemma":[0.00013979131,0.0008905015,0.0020955973,0.00020755915,0.00013623957,0.0028031832,0.00093396136,0.46233898,0.2841959,0.04182678,0.20415153,0.00028012268],"about_ca_topic_score_codex":0.0020513195,"about_ca_topic_score_gemma":0.0028749986,"teacher_disagreement_score":0.0054859333,"about_ca_system_score_codex":0.00034540464,"about_ca_system_score_gemma":0.0011806621,"threshold_uncertainty_score":0.01835221},"labels":[],"label_agreement":null},{"id":"W2266623466","doi":"10.4000/lapurdum.2393","title":"Hizkeren arteko aldakortasun sintaktikoa aztertzeko metodologiaren nondik norakoak : BASYQUE aplikazioa","year":2012,"lang":"eu","type":"article","venue":"Lapurdum","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Nautical Research Society","funders":"","keywords":"Sociology","score_opus":0.02803016633507035,"score_gpt":0.28657350055247527,"score_spread":0.25854333421740494,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2266623466","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0803768,0.012068049,0.32687044,0.017304232,0.0014677705,0.00020496057,0.00085404026,0.0013010883,0.55955255],"genre_scores_gemma":[0.62794614,0.007810072,0.18075211,0.0020257928,0.00051776186,0.0002232355,0.001054423,0.0010654349,0.17860505],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.997505,0.00089859165,0.00016882998,0.0005292883,0.0006800689,0.00021818969],"domain_scores_gemma":[0.99774027,0.0010190617,0.00014124194,0.00048382403,0.00050361775,0.0001119024],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021291624,0.00066181575,0.0005552203,0.0013740296,0.002021753,0.010002964,0.001245549,0.0013541913,0.021311233],"category_scores_gemma":[0.0050018877,0.0005484484,0.00060801336,0.0014391484,0.0044025704,0.00837663,0.0037276098,0.0027186133,0.0047058836],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018884051,0.00008742308,0.0025925082,0.0007313879,0.000045776873,0.0006356302,0.011179211,0.0009383086,0.0077780876,0.792373,0.014030981,0.16941886],"study_design_scores_gemma":[0.00002729619,0.000046772642,0.0034594387,0.00041460697,0.000051462717,0.0010854176,0.010780343,0.0037984755,0.0083015235,0.49366915,0.47828254,0.000082895465],"about_ca_topic_score_codex":0.003841956,"about_ca_topic_score_gemma":0.008339779,"teacher_disagreement_score":0.021311233,"about_ca_system_score_codex":0.0023870934,"about_ca_system_score_gemma":0.002248029,"threshold_uncertainty_score":0.071293116},"labels":[],"label_agreement":null},{"id":"W2268028085","doi":"10.48550/arxiv.1308.2149","title":"Frameworks for Reasoning about Syntax that Utilize Quotation and Evaluation","year":2013,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"","keywords":"Syntax; Computer science; Abstract syntax; Abstract syntax tree; Syntax error; Expression (computer science); Semantics (computer science); Natural language processing; Value (mathematics); Programming language; Artificial intelligence; Linguistics","score_opus":0.07323299892100535,"score_gpt":0.2442084540608029,"score_spread":0.17097545513979756,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2268028085","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0009726288,0.00038553053,0.9902132,0.0010291015,0.00015431379,0.00015699101,0.00012624463,0.0011793015,0.0057826657],"genre_scores_gemma":[0.074273236,0.0009521014,0.9163177,0.00069975655,0.00040141906,0.00056651776,0.00052006677,0.0010784083,0.0051908297],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.97379225,0.012378353,0.0028516732,0.004018521,0.005544247,0.0014150166],"domain_scores_gemma":[0.97918564,0.009824631,0.0019226514,0.005491565,0.0029576048,0.0006178325],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03405557,0.0027687792,0.0029511843,0.009068598,0.0056724222,0.014442728,0.006846728,0.0050781113,0.009623544],"category_scores_gemma":[0.044048287,0.0030075398,0.0075732046,0.0068093664,0.016092505,0.044917647,0.013263219,0.0068149315,0.003169953],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000017188942,0.000010029693,0.000117366566,0.000091915674,0.000024409801,0.000088001805,0.00069165597,0.0015611268,0.00022922416,0.98683864,0.0011727074,0.009157707],"study_design_scores_gemma":[0.000020410936,0.000013478807,0.00005486914,0.000102755796,0.000041448417,0.00010785621,0.00025181955,0.012713636,0.0008901988,0.95496756,0.03077317,0.0000628989],"about_ca_topic_score_codex":0.009601454,"about_ca_topic_score_gemma":0.007955945,"teacher_disagreement_score":0.03405557,"about_ca_system_score_codex":0.0071462328,"about_ca_system_score_gemma":0.005723158,"threshold_uncertainty_score":0.18010527},"labels":[],"label_agreement":null},{"id":"W2268635234","doi":"10.1017/cbo9780511801686.001","title":"Preface","year":2008,"lang":"en","type":"book-chapter","venue":"Cambridge University Press eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Variety (cybernetics); Computer science; Construct (python library); Variation (astronomy); Visualization; Range (aeronautics); Data science; Artificial intelligence; Natural language processing; Linguistics; Engineering; Programming language","score_opus":0.01917908829125878,"score_gpt":0.2045821718623231,"score_spread":0.18540308357106433,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2268635234","genre_codex":"other","genre_gemma":"editorial","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"editorial","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00092375564,0.007619717,0.006129879,0.0100552365,0.030652542,0.00024907303,0.005545892,0.0014436747,0.9373803],"genre_scores_gemma":[0.0030941109,0.0031957005,0.0026067577,0.0017406482,0.0037115684,0.000109227236,0.0042039272,0.00073218776,0.9806058],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.999366,0.000071110044,0.000037990943,0.0001294055,0.00034697787,0.00004841529],"domain_scores_gemma":[0.9984327,0.0002699238,0.00005649774,0.00016927184,0.00084784813,0.00022371986],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006716573,0.00093590596,0.0007881082,0.0019971118,0.0019970967,0.0040396024,0.0013824222,0.0013145311,0.49705568],"category_scores_gemma":[0.004336988,0.0003745799,0.0005906229,0.0017665355,0.0006759598,0.003399569,0.0019266466,0.0029559713,0.36166686],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00001756296,0.00002098792,0.00006612937,0.00011923385,0.0000017013718,0.000048762882,0.00012887975,0.0000801739,0.00014387521,0.012299787,0.9300275,0.057045594],"study_design_scores_gemma":[0.0000014354542,0.000007143776,0.00010803021,0.00007846543,7.261767e-7,0.000055557237,0.000045142955,0.000023329396,0.000043278873,0.0021546166,0.99747974,0.0000024256453],"about_ca_topic_score_codex":0.002193335,"about_ca_topic_score_gemma":0.0032394594,"teacher_disagreement_score":0.49705568,"about_ca_system_score_codex":0.0017792789,"about_ca_system_score_gemma":0.0014420364,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W2268831157","doi":"","title":"Corpus-based Machine Translation: Its Current Development and Perspectives","year":2015,"lang":"en","type":"article","venue":"International Forum of Teaching and Studies","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Machine translation; Computer science; Artificial intelligence; Natural language processing; Example-based machine translation; Corpus linguistics; Parallel corpora; Computer-assisted translation; Machine translation software usability; Rule-based machine translation; Text corpus; Computational linguistics","score_opus":0.09023991017756336,"score_gpt":0.36226063237468953,"score_spread":0.27202072219712614,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2268831157","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0008774162,0.9548415,0.025054295,0.006405118,0.0013240007,0.000053851687,0.00013516677,0.00025083768,0.011057783],"genre_scores_gemma":[0.012315797,0.95423853,0.023912814,0.0021485635,0.0038942413,0.00013168853,0.0004667119,0.00020007008,0.0026914938],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9954276,0.0019176847,0.00042849788,0.0008314166,0.001245472,0.00014926812],"domain_scores_gemma":[0.9859439,0.009580696,0.0004580307,0.00072119053,0.003003153,0.0002930491],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0084545445,0.0012287298,0.0021281273,0.0052423887,0.0012582792,0.0059637735,0.0030837522,0.0033183328,0.0073927226],"category_scores_gemma":[0.011790636,0.0009449953,0.0010955397,0.011220173,0.0040399632,0.010618087,0.0026199094,0.003194303,0.004344925],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000099152014,0.000081859296,0.00072162604,0.008529321,0.00010671497,0.00021336002,0.0008091232,0.0019843762,0.0009902723,0.07892379,0.035258,0.8722825],"study_design_scores_gemma":[0.000023957404,0.00010524854,0.0012448874,0.004374736,0.0001076905,0.00092156907,0.00081890146,0.005933706,0.0010249608,0.056759916,0.92858505,0.00009932019],"about_ca_topic_score_codex":0.003872627,"about_ca_topic_score_gemma":0.0020353913,"teacher_disagreement_score":0.0084545445,"about_ca_system_score_codex":0.0027478181,"about_ca_system_score_gemma":0.004643307,"threshold_uncertainty_score":0.044712424},"labels":[],"label_agreement":null},{"id":"W2268951307","doi":"","title":"SAT-MICRO: petit mais costaud !","year":2008,"lang":"pt","type":"article","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Prevention of Organ Failure","funders":"","keywords":"Humanities; Philosophy; DPLL algorithm; Art; Mathematics; Computer science","score_opus":0.013996474544270327,"score_gpt":0.23465106760121102,"score_spread":0.22065459305694068,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2268951307","genre_codex":"software","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0077887005,0.0027068574,0.21965368,0.009925423,0.007057402,0.0004503689,0.021858975,0.5967942,0.13376437],"genre_scores_gemma":[0.09599063,0.001647976,0.17300093,0.0063881692,0.003564248,0.00117327,0.050671287,0.22476125,0.44280234],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9972458,0.0006988741,0.00014185753,0.0006153586,0.0009118601,0.00038640315],"domain_scores_gemma":[0.99331325,0.0015509872,0.00013606054,0.002552846,0.0015859971,0.00086086284],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0027117094,0.0028588318,0.0025234027,0.0016292545,0.0018280534,0.0046061715,0.0047113206,0.002211094,0.39126852],"category_scores_gemma":[0.008274419,0.0027980984,0.0020807963,0.0024655892,0.0011378619,0.007739379,0.0041841804,0.004633539,0.2507838],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015238901,0.00012137589,0.0003875644,0.00028266635,0.000070520364,0.000083363164,0.00009726178,0.0013370417,0.0037065065,0.01117082,0.8759128,0.10530615],"study_design_scores_gemma":[0.00093360455,0.00040754667,0.0014534768,0.00019781993,0.00010512341,0.00020652771,0.00011938826,0.044589926,0.012480184,0.03288445,0.90645313,0.00016883216],"about_ca_topic_score_codex":0.0049335426,"about_ca_topic_score_gemma":0.0089241825,"teacher_disagreement_score":0.39126852,"about_ca_system_score_codex":0.0016005025,"about_ca_system_score_gemma":0.0018097932,"threshold_uncertainty_score":0.8682816},"labels":[],"label_agreement":null},{"id":"W2268973493","doi":"","title":"Principles and Practicalities of Corpus Design in Language Retrieval: Issues in the Digitization of the Beynon Corpus of Early Twentieth-Century Sm’algyax Materials","year":2010,"lang":"en","type":"article","venue":"ScholarSpace (University of Hawaii at Manoa)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"International Council for Canadian Studies","keywords":"Digitization; Computer science; Linguistics; Function (biology); Natural language processing; Corpus linguistics; Artificial intelligence; Ethnography; Sociology; Anthropology; Philosophy","score_opus":0.014620966694064227,"score_gpt":0.24266165534174663,"score_spread":0.2280406886476824,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2268973493","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015203261,0.0015123588,0.95861274,0.009624737,0.00045987213,0.002208022,0.00033911364,0.00080359454,0.01123626],"genre_scores_gemma":[0.04640578,0.0005394591,0.94315,0.0010386384,0.0002585664,0.0046766577,0.00038142418,0.00067142345,0.002878077],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.8779132,0.09782631,0.0080851065,0.0048423726,0.010474198,0.0008588038],"domain_scores_gemma":[0.7205215,0.19144225,0.008603383,0.05334356,0.024397204,0.0016921834],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1424769,0.00077498954,0.0015276254,0.006226625,0.007245899,0.01883832,0.0056074513,0.003746157,0.0048331562],"category_scores_gemma":[0.26578063,0.0027738714,0.0008714504,0.008291971,0.019686943,0.022326916,0.010016261,0.0061660483,0.0022563182],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00040832857,0.00024537923,0.0057316655,0.0023507462,0.0000800476,0.00063321023,0.06400127,0.0057224985,0.015871968,0.50859576,0.015320163,0.38103896],"study_design_scores_gemma":[0.000596511,0.0007827025,0.0071436292,0.0019331553,0.00017475992,0.0028773483,0.025341079,0.027411446,0.032066166,0.39622614,0.5050213,0.00042579987],"about_ca_topic_score_codex":0.004078977,"about_ca_topic_score_gemma":0.0067920773,"teacher_disagreement_score":0.1424769,"about_ca_system_score_codex":0.003694492,"about_ca_system_score_gemma":0.008129702,"threshold_uncertainty_score":0.7534989},"labels":[],"label_agreement":null},{"id":"W2270918357","doi":"","title":"A Poor Man's Translation Memory Using Machine Translation Evaluation Metrics","year":2012,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Machine translation; Computer science; Translation (biology); Implementation; Similarity (geometry); Artificial intelligence; Machine learning; Natural language processing; Programming language","score_opus":0.0713268138229567,"score_gpt":0.33494203921185667,"score_spread":0.2636152253889,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2270918357","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13166535,0.0013675137,0.8374774,0.00090625405,0.00016016315,0.00025699317,0.0006029698,0.015678743,0.01188461],"genre_scores_gemma":[0.72323614,0.0003143653,0.27090323,0.00024319747,0.00007079377,0.00022654675,0.00071962114,0.0012382022,0.0030479603],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99261296,0.00370907,0.00079550646,0.0009401952,0.0016626046,0.0002795456],"domain_scores_gemma":[0.9824644,0.008142236,0.0012455541,0.0045440877,0.0033052403,0.00029855574],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007844432,0.0013878999,0.0012760126,0.002753,0.0009906986,0.004287017,0.001670107,0.001305749,0.004057302],"category_scores_gemma":[0.037019204,0.0003835228,0.00060046796,0.0030624075,0.0009631045,0.0066955113,0.0020054022,0.0010526022,0.002303138],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008930566,0.0002682028,0.014630485,0.0006951552,0.000317149,0.0003560647,0.0011831722,0.018597929,0.0456967,0.034493834,0.011241298,0.871627],"study_design_scores_gemma":[0.00015063689,0.001642756,0.01297399,0.00034106217,0.0005593008,0.0021277706,0.0010922757,0.6456436,0.19660555,0.1009634,0.037591014,0.0003086612],"about_ca_topic_score_codex":0.0017752673,"about_ca_topic_score_gemma":0.0021573736,"teacher_disagreement_score":0.007844432,"about_ca_system_score_codex":0.00079720234,"about_ca_system_score_gemma":0.0013288605,"threshold_uncertainty_score":0.041485846},"labels":[],"label_agreement":null},{"id":"W2273793835","doi":"","title":"Borrowing of discourse functions of English \"suggest\" by Spanish \"sugerir\" in biomedical research articles: a contrastive study","year":2010,"lang":"en","type":"book-chapter","venue":"Buleria (Universidad de León)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Linguistics; Verb; Popularity; Context (archaeology); Perspective (graphical); Style (visual arts); Psychology; Subcategorization; Computer science; Artificial intelligence; Philosophy; History; Literature; Art","score_opus":0.020256454118477117,"score_gpt":0.3085228033202442,"score_spread":0.2882663492017671,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2273793835","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9690534,0.00160012,0.0019338892,0.00085855933,0.00007180681,0.00007554108,0.00014793314,0.000012088382,0.026246587],"genre_scores_gemma":[0.99584997,0.00059685943,0.0010568927,0.0002982475,0.00011875238,0.000113924354,0.00016546331,0.000051809307,0.0017480309],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9949058,0.0030649053,0.00041709704,0.0005950061,0.0007729089,0.00024420363],"domain_scores_gemma":[0.95875484,0.029986054,0.0054583047,0.0019182868,0.0033347367,0.0005478294],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006708879,0.00054596836,0.00052160845,0.004073725,0.002639469,0.004644898,0.0007223826,0.0013093376,0.0049998453],"category_scores_gemma":[0.032492813,0.00036611874,0.00029263395,0.0026293772,0.004591115,0.0055670417,0.0042524706,0.0017035949,0.0006607896],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011098365,0.00044199495,0.06986472,0.0012095674,0.000090089045,0.0016013212,0.83625245,0.00005489242,0.017388377,0.035427537,0.0014094852,0.035149716],"study_design_scores_gemma":[0.00023315562,0.0007584626,0.27719456,0.0016416128,0.00036249575,0.006337564,0.58481014,0.0010184455,0.00970448,0.010572055,0.10722249,0.00014448976],"about_ca_topic_score_codex":0.0022363714,"about_ca_topic_score_gemma":0.0026001574,"teacher_disagreement_score":0.006708879,"about_ca_system_score_codex":0.0024514708,"about_ca_system_score_gemma":0.0009655328,"threshold_uncertainty_score":0.03548032},"labels":[],"label_agreement":null},{"id":"W22745232","doi":"10.1101/gr.136572.111","title":"TRADUCTION - Translation Quality Assessment: Linguistic Description vs Social Evaluation: Linguistic Description vs Social Evaluation","year":2001,"lang":"en","type":"article","venue":"Meta: Journal des traducteurs = translators' journal","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Canadian Institutes of Health Research","keywords":"Linguistics; Linguistic analysis; Quality assessment; Deep linguistic processing; Quality (philosophy); Computer science; Linguistic description; Natural language processing; Linguistic change; Artificial intelligence; Evaluation methods; Philosophy; Epistemology","score_opus":0.15806980254785147,"score_gpt":0.38789356767728245,"score_spread":0.22982376512943098,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W22745232","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.25903222,0.007997298,0.65970075,0.006808567,0.0011579316,0.0021264844,0.0073582297,0.00974471,0.046073806],"genre_scores_gemma":[0.813436,0.0006688509,0.1740851,0.0005169573,0.0003391618,0.0008178201,0.0044613197,0.0009331662,0.0047415495],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9458917,0.034280773,0.004280068,0.0034494405,0.01122762,0.0008703874],"domain_scores_gemma":[0.881052,0.08217094,0.009425006,0.010528763,0.015183369,0.0016398842],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.041943457,0.001957722,0.0018361869,0.00779029,0.001318165,0.0068094665,0.0016250745,0.0022569785,0.008377585],"category_scores_gemma":[0.14070356,0.00035499907,0.001817957,0.005360077,0.0021021692,0.0054757628,0.0045764856,0.0019484663,0.0020326036],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0039654463,0.0006762562,0.042098135,0.0027604993,0.001660489,0.00038203393,0.0022235645,0.050968923,0.007038235,0.018565143,0.022241106,0.8474203],"study_design_scores_gemma":[0.000895823,0.0023579148,0.042535506,0.0011042681,0.0011837106,0.0008014793,0.0029771314,0.7916518,0.021373032,0.09992911,0.03458417,0.0006060442],"about_ca_topic_score_codex":0.0037609856,"about_ca_topic_score_gemma":0.0033342787,"teacher_disagreement_score":0.041943457,"about_ca_system_score_codex":0.0030793678,"about_ca_system_score_gemma":0.0022372291,"threshold_uncertainty_score":0.22182089},"labels":[],"label_agreement":null},{"id":"W2277746464","doi":"","title":"Non-standard transcription of Innu: An essential ingredient of its documentation","year":2015,"lang":"en","type":"article","venue":"Americanae (AECID Library)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Documentation; Transcription (linguistics); Technical documentation; Linguistics; History; Computer science","score_opus":0.0109619178205593,"score_gpt":0.26609286025027135,"score_spread":0.25513094242971207,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2277746464","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12818192,0.006241212,0.54871523,0.031294737,0.009044212,0.0018314897,0.05679364,0.011065292,0.20683226],"genre_scores_gemma":[0.59112,0.0037793051,0.30985212,0.0027508328,0.0013297821,0.0011923842,0.031942554,0.0053250357,0.05270807],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.98555714,0.0045175715,0.0016264668,0.0010078071,0.006583583,0.000707426],"domain_scores_gemma":[0.91912925,0.01691533,0.0023367677,0.010629281,0.0502077,0.00078160624],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010314354,0.0006095341,0.00081825204,0.0033810765,0.00363316,0.0056813797,0.0015760822,0.0010319464,0.0075369547],"category_scores_gemma":[0.05497289,0.0004141416,0.00021609277,0.0054548704,0.0031721296,0.002416727,0.002337718,0.0029337576,0.005216153],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005252861,0.00009294408,0.0090437895,0.0020189239,0.000031978336,0.0011641895,0.045145277,0.0024125099,0.027768813,0.09187199,0.16458814,0.65533626],"study_design_scores_gemma":[0.00002644624,0.00007097834,0.014281277,0.0014017553,0.000042516534,0.0008261527,0.021935267,0.008637401,0.025146363,0.019410014,0.9079915,0.00023024042],"about_ca_topic_score_codex":0.23447768,"about_ca_topic_score_gemma":0.22593625,"teacher_disagreement_score":0.23447768,"about_ca_system_score_codex":0.0075209844,"about_ca_system_score_gemma":0.023068758,"threshold_uncertainty_score":0.4662258},"labels":[],"label_agreement":null},{"id":"W2278303037","doi":"","title":"Fast FPT Algorithms for Computing Rooted Agreement Forests: Theory and Experiments (Extended Abstract)","year":2010,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Intuition; Phylogenetic tree; Algorithm; Branching (polymer chemistry); Combinatorics; Tree (set theory); Mathematics; Binary logarithm; Computer science; Discrete mathematics; Chemistry","score_opus":0.015272993452372666,"score_gpt":0.31481743849213245,"score_spread":0.2995444450397598,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2278303037","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3269174,0.0053402395,0.59531605,0.0032682791,0.0012368321,0.00072909973,0.003931535,0.032786373,0.030474167],"genre_scores_gemma":[0.42871928,0.0005885516,0.5621371,0.00048784172,0.00017254049,0.00043125683,0.0034236582,0.0011391366,0.002900571],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99432784,0.0016029808,0.0004145324,0.0009979748,0.0018375883,0.0008190505],"domain_scores_gemma":[0.96248186,0.026910355,0.0010264963,0.004827644,0.00394434,0.0008092308],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0077060424,0.0016744585,0.0015094063,0.001895937,0.0020406991,0.002549708,0.003001224,0.0022927958,0.009366313],"category_scores_gemma":[0.02811011,0.00058333325,0.0015031885,0.0038757187,0.0015569237,0.008183357,0.0020570809,0.0029640135,0.0026876992],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0038439452,0.002073284,0.008149367,0.0011928452,0.0003236775,0.00025158102,0.0004936034,0.27787834,0.011665669,0.03165241,0.06735823,0.59511703],"study_design_scores_gemma":[0.000480686,0.00030678877,0.0009465033,0.000046330722,0.00006701474,0.00014728437,0.00016174436,0.9528687,0.008681351,0.032192767,0.0040631457,0.00003769446],"about_ca_topic_score_codex":0.012449536,"about_ca_topic_score_gemma":0.009170607,"teacher_disagreement_score":0.012449536,"about_ca_system_score_codex":0.0041163126,"about_ca_system_score_gemma":0.0038479008,"threshold_uncertainty_score":0.04075396},"labels":[],"label_agreement":null},{"id":"W2280098327","doi":"10.3138/cmlr.63.2.293","title":"<b>Nesselhauf, Nadja.</b> (2005). <i>Collocations in a Learner Corpus.</i> Amsterdam: John Benjamins. Pp. xii, 332. US$126.00 / €105.00 (cloth).","year":2006,"lang":"en","type":"article","venue":"Canadian Modern Language Review/ La Revue canadienne des langues vivantes","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Linguistics; Art; Humanities; Philosophy","score_opus":0.008810229368587873,"score_gpt":0.22623975032563443,"score_spread":0.21742952095704654,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2280098327","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0020293484,0.8163049,0.018563664,0.047791924,0.014578709,0.00009921291,0.017820437,0.0014699282,0.08134185],"genre_scores_gemma":[0.023350539,0.7335892,0.029234376,0.0131006185,0.006310817,0.00027995734,0.027802838,0.0016794688,0.16465224],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9992955,0.00014129665,0.00008670791,0.00010187393,0.00029595834,0.000078726735],"domain_scores_gemma":[0.99574935,0.0018357586,0.00023053729,0.0001493779,0.0018879176,0.0001470156],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017486353,0.0010926641,0.00071453175,0.006343449,0.0023082308,0.0033962992,0.0013885379,0.0012906675,0.08466123],"category_scores_gemma":[0.0063571753,0.0007172959,0.00055532745,0.009469267,0.0010311435,0.00892008,0.0016041892,0.0023141638,0.058118515],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000030919993,0.000009455504,0.00043573225,0.000924354,0.000015540521,0.00009937035,0.00039910694,0.000047247908,0.00034376324,0.002175492,0.8588962,0.13662279],"study_design_scores_gemma":[0.00000889287,0.0000073309147,0.0034221716,0.001014742,0.0000311748,0.00039431403,0.00046309753,0.00007469992,0.0004949215,0.0029130308,0.9911567,0.000018975057],"about_ca_topic_score_codex":0.06997973,"about_ca_topic_score_gemma":0.17663348,"teacher_disagreement_score":0.08466123,"about_ca_system_score_codex":0.0017174548,"about_ca_system_score_gemma":0.0037303183,"threshold_uncertainty_score":0.28322005},"labels":[],"label_agreement":null},{"id":"W2286447845","doi":"","title":"Struggling Languages in a Wired World: How Best to Use the Internet in Language Revitalization","year":2008,"lang":"en","type":"dissertation","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"BATES; Indigenous; Media studies; Indigenous language; History; Linguistics; Sociology; Engineering; Philosophy","score_opus":0.016379836884384574,"score_gpt":0.30214785770441244,"score_spread":0.28576802082002783,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2286447845","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.046479188,0.045012746,0.017301384,0.6629666,0.0021474308,0.00008298636,0.00011231116,0.00037706416,0.22552021],"genre_scores_gemma":[0.6989447,0.10249945,0.049063914,0.022505699,0.0016936332,0.00021817816,0.00034200368,0.00064625376,0.124086164],"study_design_codex":"design_other","study_design_gemma":"qualitative","domain_scores_codex":[0.9987037,0.0007969078,0.00003755342,0.00010574871,0.00016958443,0.0001865966],"domain_scores_gemma":[0.9970187,0.0012917732,0.00024094932,0.00024146873,0.0005097872,0.0006974045],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030307022,0.0003048931,0.00025044617,0.0009839043,0.0029515447,0.008948431,0.000775227,0.0018309184,0.012452262],"category_scores_gemma":[0.005741304,0.00021191326,0.0002995532,0.0016257315,0.004686791,0.023370722,0.002442341,0.004344531,0.0034461776],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007012466,0.00022193926,0.0024016874,0.0005502702,0.000027701391,0.00027600065,0.039737087,0.0005745231,0.001411007,0.24085349,0.21847932,0.49539682],"study_design_scores_gemma":[0.000025521165,0.000058101057,0.0030892263,0.0011931128,0.000042659645,0.00032257484,0.10960378,0.0013273268,0.0013570966,0.22889943,0.6540171,0.000064051994],"about_ca_topic_score_codex":0.0053838757,"about_ca_topic_score_gemma":0.010259127,"teacher_disagreement_score":0.012452262,"about_ca_system_score_codex":0.0033683248,"about_ca_system_score_gemma":0.004389939,"threshold_uncertainty_score":0.04165703},"labels":[],"label_agreement":null},{"id":"W2289290552","doi":"10.3765/bls.v27i1.1105","title":"Object Marking and Agentivity in Navajo Causatives","year":2001,"lang":"en","type":"article","venue":"Proceedings of the Annual Meeting of the Berkeley Linguistics Society","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Navajo; Object (grammar); Computer science; Artificial intelligence; Linguistics; Philosophy","score_opus":0.011415687401143388,"score_gpt":0.25694079901736466,"score_spread":0.2455251116162213,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2289290552","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.31021184,0.0031817586,0.037786305,0.0023788866,0.00049765664,0.00008119753,0.0003473804,0.00059553864,0.64491946],"genre_scores_gemma":[0.98176277,0.0007972358,0.0036569887,0.00008457432,0.00008534616,0.000024947629,0.00015560197,0.00014033973,0.013292321],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9992873,0.00027101598,0.000058879334,0.00012587161,0.00015787958,0.00009911097],"domain_scores_gemma":[0.9985501,0.00083365996,0.00018302863,0.00019657206,0.00017133419,0.00006529948],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013732832,0.00041753045,0.0006222486,0.0011255436,0.002183691,0.0037203473,0.0006977727,0.0007913332,0.011003738],"category_scores_gemma":[0.004372046,0.0004743788,0.00036598355,0.00087799097,0.0032858443,0.004866034,0.002866214,0.0011247114,0.00086163764],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021942996,0.000050733528,0.0028442598,0.00027767738,0.000025155789,0.0006144576,0.010137247,0.00055106316,0.0050152866,0.9441951,0.003368161,0.032701377],"study_design_scores_gemma":[0.00006336809,0.00007025562,0.008531192,0.00045844173,0.000115627096,0.0009125688,0.0059613865,0.0036556576,0.0049801553,0.87977564,0.09541209,0.00006360682],"about_ca_topic_score_codex":0.0038071554,"about_ca_topic_score_gemma":0.0056566545,"teacher_disagreement_score":0.011003738,"about_ca_system_score_codex":0.0011700551,"about_ca_system_score_gemma":0.0006648248,"threshold_uncertainty_score":0.036811173},"labels":[],"label_agreement":null},{"id":"W2292982127","doi":"10.3765/bls.v35i1.3630","title":"Focused N-Words and Double Negation Readings in Negative Concord Languages","year":2009,"lang":"en","type":"article","venue":"Proceedings of the Annual Meeting of the Berkeley Linguistics Society","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Negation; Linguistics; Computer science; Natural language processing; Mathematics; Programming language; Philosophy","score_opus":0.009347096773520141,"score_gpt":0.26695511273083433,"score_spread":0.2576080159573142,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2292982127","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.53549826,0.00070776895,0.20732826,0.0039102165,0.000620451,0.00012055812,0.00053789996,0.0011258546,0.25015074],"genre_scores_gemma":[0.9784021,0.000084479725,0.010495004,0.000311782,0.00012187469,0.000036721714,0.00019877772,0.00021476789,0.01013447],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.998613,0.00047145248,0.0001312539,0.00031809814,0.00026035233,0.00020579407],"domain_scores_gemma":[0.99648726,0.001726687,0.0002809551,0.00045647245,0.00086445245,0.00018424133],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001726575,0.0006323414,0.00055521156,0.0012863213,0.0035791579,0.0036010765,0.0012315706,0.0014945598,0.006965237],"category_scores_gemma":[0.0048231166,0.0007892061,0.0005553691,0.0009496735,0.0042565367,0.009399953,0.0030482884,0.0023966283,0.0008626977],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001531535,0.00004439297,0.0011359708,0.00007707837,0.000011024211,0.000777657,0.0036657122,0.00024660965,0.003869632,0.9771911,0.002916303,0.0099113975],"study_design_scores_gemma":[0.000054580454,0.000036971087,0.0013197969,0.000071741095,0.000037129415,0.001206281,0.003204873,0.005941301,0.0064087836,0.96063954,0.021020507,0.00005838136],"about_ca_topic_score_codex":0.0034233613,"about_ca_topic_score_gemma":0.004173882,"teacher_disagreement_score":0.006965237,"about_ca_system_score_codex":0.0018666622,"about_ca_system_score_gemma":0.0008730482,"threshold_uncertainty_score":0.023301065},"labels":[],"label_agreement":null},{"id":"W2293438896","doi":"10.3115/v1/n15-1095","title":"Joint Generation of Transliterations from Multiple Representations","year":2015,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Transliteration; Computer science; Spelling; Natural language processing; Artificial intelligence; Pronunciation; Joint (building); Lexicon; Speech recognition; Machine translation; Linguistics","score_opus":0.10044819521633011,"score_gpt":0.3160813897396517,"score_spread":0.2156331945233216,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2293438896","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.048765685,0.0015488124,0.9159196,0.0009783,0.000764512,0.00022656363,0.0040053776,0.019913923,0.00787721],"genre_scores_gemma":[0.5479386,0.0010381077,0.41963878,0.0004081328,0.00026261396,0.00038683123,0.015035908,0.0024823062,0.012808673],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9980229,0.0006356951,0.00014179974,0.000663199,0.0003702482,0.00016614266],"domain_scores_gemma":[0.9966176,0.0013345045,0.00024316706,0.0010205654,0.0007120524,0.00007219283],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013098434,0.0018931824,0.0013080543,0.0017549585,0.0005655055,0.0014440179,0.0013393206,0.001282645,0.004997286],"category_scores_gemma":[0.009162979,0.00055628637,0.0016881919,0.0020399136,0.0005714428,0.002240271,0.0019441962,0.0017032105,0.006102902],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005602233,0.00023922756,0.0038992877,0.000825848,0.00023993562,0.0010079677,0.00047386574,0.08326108,0.048064042,0.016372921,0.037383348,0.8076724],"study_design_scores_gemma":[0.000119834534,0.00021049738,0.0018373044,0.00010907116,0.00024312256,0.00096518046,0.00027201718,0.7752782,0.1293221,0.04572711,0.045779593,0.00013592416],"about_ca_topic_score_codex":0.0021645527,"about_ca_topic_score_gemma":0.003182792,"teacher_disagreement_score":0.004997286,"about_ca_system_score_codex":0.0008247945,"about_ca_system_score_gemma":0.0014200966,"threshold_uncertainty_score":0.016717553},"labels":[],"label_agreement":null},{"id":"W2293453615","doi":"10.3115/v1/n15-1169","title":"Reserating the awesometastic: An automatic extension of the WordNet taxonomy for novel terms","year":2015,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":30,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"WordNet; Computer science; Taxonomy (biology); Extension (predicate logic); Natural language processing; Lexicon; Slang; Artificial intelligence; Information retrieval; Linguistics; Programming language","score_opus":0.0970038646474274,"score_gpt":0.3117357808644425,"score_spread":0.2147319162170151,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2293453615","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.043238297,0.0014976485,0.8401469,0.001192781,0.0008841496,0.0013010802,0.018069033,0.072265,0.021405086],"genre_scores_gemma":[0.05494029,0.0004718852,0.9119465,0.00029003381,0.00009759536,0.00040244556,0.022688998,0.0031934376,0.005968878],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99847764,0.00029222324,0.00018293544,0.00036836753,0.000597073,0.00008172778],"domain_scores_gemma":[0.99494666,0.0019648548,0.0003249982,0.001072378,0.0014106701,0.00028043918],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020721692,0.0010974584,0.00075176003,0.006420807,0.0014576564,0.001722408,0.0014379217,0.0007768358,0.011592465],"category_scores_gemma":[0.009500819,0.000831546,0.0010282428,0.0029558518,0.0008937773,0.0068389275,0.0038589272,0.0015610037,0.006850068],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004376184,0.00024234955,0.005382704,0.0021050293,0.00009934583,0.00072331936,0.0023012222,0.0032022607,0.027220242,0.05667415,0.111145265,0.7904665],"study_design_scores_gemma":[0.00009411573,0.0002457258,0.004407226,0.0006225281,0.00011359855,0.0017725965,0.001448756,0.0812168,0.022292696,0.063423306,0.8242111,0.00015149993],"about_ca_topic_score_codex":0.0031535667,"about_ca_topic_score_gemma":0.011984998,"teacher_disagreement_score":0.011592465,"about_ca_system_score_codex":0.00080009655,"about_ca_system_score_gemma":0.002575978,"threshold_uncertainty_score":0.03878069},"labels":[],"label_agreement":null},{"id":"W2293856418","doi":"","title":"Research on Sentence Construction Rules in Standard Chinese","year":2016,"lang":"en","type":"article","venue":"Canadian social science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Sentence; Notional amount; Linguistics; Computer science; Modal; Subject (documents); Natural language processing; Function (biology); Word (group theory); Order (exchange); Artificial intelligence; Philosophy; Business; World Wide Web","score_opus":0.021614519981272375,"score_gpt":0.34584001836612843,"score_spread":0.32422549838485604,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2293856418","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6886594,0.0041689747,0.21692061,0.0015684939,0.00018576643,0.0006421414,0.0015618275,0.0008102129,0.08548248],"genre_scores_gemma":[0.94209087,0.0016261343,0.047346886,0.00018163532,0.00006977861,0.0002813913,0.0018207005,0.00023117915,0.006351473],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9964289,0.0010674325,0.0005287067,0.0010591054,0.00070502394,0.0002108444],"domain_scores_gemma":[0.99138063,0.004679142,0.0009139001,0.00089499936,0.0019257755,0.00020550952],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031066742,0.00059146225,0.000487905,0.002920667,0.0017671536,0.002608672,0.0011928435,0.00042223049,0.0036943778],"category_scores_gemma":[0.0111602945,0.0006653468,0.000753111,0.0037230898,0.0033707863,0.0041972846,0.0008395723,0.0010315183,0.00058561837],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019528958,0.00014585047,0.0681383,0.0022740152,0.00020432161,0.0019420246,0.03387302,0.008829384,0.029569278,0.51967466,0.0070593758,0.32809457],"study_design_scores_gemma":[0.00014700195,0.00039230587,0.21087053,0.0009517349,0.00058983034,0.0044015,0.016195556,0.13751525,0.073360406,0.43632054,0.11878932,0.00046605605],"about_ca_topic_score_codex":0.017908344,"about_ca_topic_score_gemma":0.013593987,"teacher_disagreement_score":0.017908344,"about_ca_system_score_codex":0.0023271511,"about_ca_system_score_gemma":0.0036943483,"threshold_uncertainty_score":0.035608172},"labels":[],"label_agreement":null},{"id":"W2294164545","doi":"10.3115/v1/n15-1056","title":"English orthography is not \"close to optimal\"","year":2015,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Innovates","keywords":"Orthography; Spelling; Computer science; Transparency (behavior); Assertion; Consistency (knowledge bases); Word (group theory); Optimality theory; Linguistics; Natural language processing; Artificial intelligence; Phonology; Programming language; Reading (process)","score_opus":0.01922826403573538,"score_gpt":0.2783663388044835,"score_spread":0.2591380747687482,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2294164545","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8711639,0.00021751001,0.10872657,0.00042924145,0.000043605254,0.000016584216,0.00022634809,0.0010213732,0.018154938],"genre_scores_gemma":[0.97936606,0.00005718597,0.019454522,0.0000787625,0.000007565871,0.0000052902983,0.00017834583,0.000108216685,0.00074401253],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.999116,0.00018641,0.00010425666,0.00033207098,0.00018925527,0.00007199409],"domain_scores_gemma":[0.9980262,0.0008349439,0.00023832373,0.0005343832,0.0002940473,0.000072040486],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00043037673,0.00016514071,0.00033912278,0.00036699974,0.000582835,0.0015842363,0.00028393263,0.0003116163,0.0019867225],"category_scores_gemma":[0.0046089333,0.0002629486,0.00020479708,0.0004787344,0.0008363564,0.0010653117,0.0006045751,0.00038932185,0.0005752454],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010663052,0.00024512238,0.07354912,0.00076339627,0.00040008622,0.0010756152,0.0022795247,0.06836697,0.18858184,0.11731704,0.009255932,0.53709906],"study_design_scores_gemma":[0.00036089995,0.0011929962,0.12638405,0.00012639936,0.0004931575,0.0042608585,0.004335157,0.30030006,0.18012165,0.34374765,0.038473655,0.0002034716],"about_ca_topic_score_codex":0.0012580563,"about_ca_topic_score_gemma":0.0032548015,"teacher_disagreement_score":0.0019867225,"about_ca_system_score_codex":0.00032479918,"about_ca_system_score_gemma":0.000593546,"threshold_uncertainty_score":0.006646216},"labels":[],"label_agreement":null},{"id":"W2294508362","doi":"10.7202/1035932ar","title":"Proposition de protocole pour l’analyse des données textuelles : pour une démarche expérimentale en lexicométrie","year":2016,"lang":"fr","type":"article","venue":"Nouvelles perspectives en sciences sociales","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":28,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Art","score_opus":0.07846731784657346,"score_gpt":0.3714836107079668,"score_spread":0.29301629286139336,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2294508362","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019579621,0.00076914014,0.9619509,0.0017801683,0.00041313784,0.0035763583,0.002380004,0.006110815,0.0034398043],"genre_scores_gemma":[0.06990907,0.00049549673,0.915878,0.00061583106,0.00012672787,0.005765634,0.0026981048,0.001301335,0.003209778],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.88043875,0.07172291,0.011576768,0.009623176,0.025244037,0.0013944086],"domain_scores_gemma":[0.76429427,0.15194452,0.0047839005,0.04321173,0.034089524,0.0016760767],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.058607455,0.0025760613,0.0025083239,0.0033073402,0.0021701637,0.010816689,0.0036184448,0.004116027,0.016025899],"category_scores_gemma":[0.17263144,0.001953377,0.0024791392,0.0029260612,0.003203907,0.010566425,0.0046203313,0.005822504,0.0076994644],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0069274968,0.0024760056,0.007471653,0.0097684655,0.001693636,0.0005058882,0.0074331295,0.014223802,0.12662865,0.066855386,0.020198815,0.73581713],"study_design_scores_gemma":[0.0032045967,0.004615084,0.011052581,0.0028675827,0.0013955585,0.0012024937,0.008545445,0.18582335,0.413432,0.11023206,0.25642663,0.0012027408],"about_ca_topic_score_codex":0.0052120104,"about_ca_topic_score_gemma":0.0030434884,"teacher_disagreement_score":0.058607455,"about_ca_system_score_codex":0.0020679415,"about_ca_system_score_gemma":0.011574537,"threshold_uncertainty_score":0.30994958},"labels":[],"label_agreement":null},{"id":"W2294758342","doi":"","title":"Sorting suffixes of two-pattern strings","year":2004,"lang":"en","type":"article","venue":"eSpace (Curtin University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Curtin University of Technology","keywords":"Generalized suffix tree; Mathematics; Suffix tree; Morphism; Merge (version control); Trie; Suffix; Sorting; Time complexity; String (physics); sort; Generalization; Suffix array; Algorithm; Iterated function; Combinatorics; Discrete mathematics; Data structure; Computer science; Arithmetic","score_opus":0.008437641015166758,"score_gpt":0.2252269908351227,"score_spread":0.21678934981995596,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2294758342","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10029574,0.00034508682,0.8832038,0.00027354385,0.00019084131,0.00019919136,0.00032848044,0.0033428723,0.011820369],"genre_scores_gemma":[0.19303526,0.00025552063,0.79193914,0.00025100747,0.000064225096,0.0001834705,0.0013382885,0.0005456378,0.012387472],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99866855,0.00023790558,0.00019566911,0.00031685113,0.00042422887,0.0001567791],"domain_scores_gemma":[0.9966247,0.0014776768,0.0002535667,0.0007822543,0.00075705,0.00010484814],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00096733746,0.0005630967,0.00083289854,0.0012253141,0.0008164603,0.0017306973,0.0012211697,0.0010024756,0.0070045353],"category_scores_gemma":[0.0055118143,0.0004059693,0.0009084936,0.002294511,0.0012091785,0.0038169841,0.0017283487,0.0011079109,0.0031111373],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005821347,0.00024794412,0.0021106072,0.00072743755,0.000083885476,0.0006951377,0.0013475814,0.015523366,0.085951574,0.30369925,0.006098784,0.58293223],"study_design_scores_gemma":[0.00017955527,0.0007221008,0.0021155693,0.00016669871,0.00009419834,0.0015279762,0.00054739986,0.2698186,0.15693094,0.4722149,0.09552902,0.00015303292],"about_ca_topic_score_codex":0.00076896645,"about_ca_topic_score_gemma":0.0011547975,"teacher_disagreement_score":0.0070045353,"about_ca_system_score_codex":0.0008079577,"about_ca_system_score_gemma":0.000963261,"threshold_uncertainty_score":0.023432493},"labels":[],"label_agreement":null},{"id":"W2294823225","doi":"","title":"The DLSIUAES Team's Participation in the TAC 2008 Tracks","year":2008,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":31,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Textual entailment; Computer science; Logical consequence; Natural language processing; Artificial intelligence; Information retrieval; Data science","score_opus":0.01033335209673914,"score_gpt":0.2811794035435295,"score_spread":0.27084605144679036,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2294823225","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.62234795,0.0013113536,0.1947306,0.03587027,0.004563186,0.0046157097,0.011071857,0.008588146,0.11690091],"genre_scores_gemma":[0.63577163,0.0004287492,0.16379088,0.004276813,0.0012052995,0.0021573806,0.021709837,0.0028305666,0.16782886],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9926611,0.0033879045,0.00021461093,0.0009579455,0.0018855431,0.0008927697],"domain_scores_gemma":[0.9798126,0.003999099,0.00030643985,0.0016069812,0.009176492,0.0050983573],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013396095,0.0008387187,0.0008355424,0.0009573135,0.0021905017,0.0024946255,0.0015065022,0.0015834324,0.0121759605],"category_scores_gemma":[0.019183924,0.0003413897,0.0005733206,0.0005748393,0.0008292415,0.0018441885,0.0026883676,0.0023518407,0.005356178],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0042674774,0.0038417773,0.013366008,0.00027820468,0.00014486685,0.0014903485,0.015009968,0.0075273034,0.0449616,0.0075460947,0.4817679,0.41979843],"study_design_scores_gemma":[0.0010203563,0.0027834177,0.021112565,0.00010940162,0.000119318975,0.0006158359,0.015712751,0.03941272,0.026450632,0.0048247683,0.8876343,0.00020394578],"about_ca_topic_score_codex":0.015260748,"about_ca_topic_score_gemma":0.017200585,"teacher_disagreement_score":0.015260748,"about_ca_system_score_codex":0.0017924545,"about_ca_system_score_gemma":0.0037516996,"threshold_uncertainty_score":0.0708462},"labels":[],"label_agreement":null},{"id":"W2294935421","doi":"","title":"LIMSI $@$ WMT13","year":2013,"lang":"en","type":"article","venue":"Workshop on Statistical Machine Translation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Computer science","score_opus":0.019263307893557426,"score_gpt":0.2953767713408867,"score_spread":0.2761134634473293,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2294935421","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016021132,0.0039284737,0.3997038,0.0121828625,0.011749427,0.0007335696,0.024853721,0.06360228,0.46722475],"genre_scores_gemma":[0.05687919,0.002123389,0.25442532,0.003140573,0.0022911485,0.00079010625,0.09382872,0.023587396,0.56293416],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9978891,0.0007115785,0.0001283166,0.0005135122,0.0005030515,0.0002543612],"domain_scores_gemma":[0.996971,0.0006099902,0.000077939956,0.001068812,0.0008564026,0.00041573783],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003339704,0.0019169068,0.0019789208,0.0022296337,0.0020184412,0.005209002,0.0026623367,0.0019148025,0.26181614],"category_scores_gemma":[0.004817764,0.00076006044,0.0013681817,0.0026049006,0.00093997474,0.0031387394,0.0044917287,0.0030529436,0.20564632],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00038689587,0.00031066054,0.00026631073,0.0002907697,0.00007283513,0.00013684462,0.00010319136,0.0010249398,0.008744856,0.032119233,0.5800648,0.37647873],"study_design_scores_gemma":[0.0001209472,0.00020824649,0.0009630346,0.00014179606,0.0000555257,0.00027430296,0.000095057294,0.026383895,0.019377712,0.039008938,0.91330504,0.00006558506],"about_ca_topic_score_codex":0.004510803,"about_ca_topic_score_gemma":0.006043487,"teacher_disagreement_score":0.26181614,"about_ca_system_score_codex":0.0014607698,"about_ca_system_score_gemma":0.0025678582,"threshold_uncertainty_score":0.87586224},"labels":[],"label_agreement":null},{"id":"W2295586553","doi":"10.71781/9728","title":"Apprentissage des réseaux de neurones profonds et applications en traitement automatique de la langue naturelle","year":2014,"lang":"fr","type":"dissertation","venue":"Open MIND","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs; Agence Nationale de la Recherche; Compute Canada; Defense Advanced Research Projects Agency; Mitacs; Canadian Institute for Advanced Research","keywords":"Physics; Humanities; Philosophy","score_opus":0.017335707778531967,"score_gpt":0.35274136162228653,"score_spread":0.33540565384375454,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2295586553","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.069554135,0.0076706223,0.8957081,0.0011115498,0.0003638427,0.00009628174,0.00020929227,0.0032232313,0.022063006],"genre_scores_gemma":[0.6759073,0.007032279,0.27141324,0.00047937935,0.00020794602,0.00024273914,0.00057464,0.0005534787,0.04358899],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99894017,0.00019502752,0.00006305729,0.0002630283,0.0004452842,0.00009339274],"domain_scores_gemma":[0.9989543,0.0003458947,0.00010028543,0.00017541555,0.0003591654,0.00006490185],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010107129,0.0007665718,0.0007339525,0.0008357836,0.0005924812,0.002136574,0.0011104182,0.0013774026,0.0055270023],"category_scores_gemma":[0.0028389937,0.0004557545,0.0008906378,0.0009016723,0.0008407211,0.0019310004,0.0010727599,0.0010702952,0.0019500683],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00042525327,0.000140044,0.0023902904,0.0010035092,0.00014294987,0.0007396771,0.0007614441,0.19810267,0.1616535,0.04108306,0.0071541527,0.5864034],"study_design_scores_gemma":[0.000038605976,0.00046919798,0.0035442973,0.00019675748,0.000113103015,0.00080466253,0.00040984148,0.79840493,0.09224616,0.026773732,0.07687476,0.0001240243],"about_ca_topic_score_codex":0.0056112283,"about_ca_topic_score_gemma":0.004416512,"teacher_disagreement_score":0.0056112283,"about_ca_system_score_codex":0.0009299828,"about_ca_system_score_gemma":0.0011031501,"threshold_uncertainty_score":0.0184896},"labels":[],"label_agreement":null},{"id":"W2295761717","doi":"10.1007/978-3-319-27947-3_18","title":"Parsing with Partially Known Grammar","year":2015,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Natural language processing; Parsing; Artificial intelligence; Rule-based machine translation; Grammar; Context-free grammar; Constraint (computer-aided design); S-attributed grammar; Linguistics; Mathematics","score_opus":0.021210527231827855,"score_gpt":0.2648407256457872,"score_spread":0.24363019841395933,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2295761717","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004142082,0.0012233682,0.9340806,0.0009444046,0.0002772167,0.00008071744,0.001680629,0.016299658,0.041271262],"genre_scores_gemma":[0.15461423,0.0028417818,0.76978195,0.0006952705,0.00043801844,0.00023049503,0.014080758,0.010160403,0.04715719],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986651,0.00034266422,0.00012891347,0.00043245984,0.00034229842,0.00008860093],"domain_scores_gemma":[0.9984976,0.00072910066,0.00004187971,0.0005686753,0.00013852745,0.000024209678],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010679399,0.0012408597,0.0014354666,0.0015659492,0.0011033977,0.0033139957,0.002152743,0.0014375466,0.027464012],"category_scores_gemma":[0.0035388512,0.0017492899,0.002152784,0.0024744764,0.0018934133,0.0067723133,0.0028852292,0.0032878455,0.0136676505],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000790306,0.00006547257,0.00026227572,0.0006273603,0.000075177144,0.0003060904,0.0005968163,0.012419038,0.004500497,0.4776237,0.05382382,0.44962066],"study_design_scores_gemma":[0.000020394651,0.000017479442,0.000233065,0.00018389494,0.0000588843,0.0003291957,0.00007778111,0.042538702,0.008455275,0.8540272,0.09400121,0.00005692259],"about_ca_topic_score_codex":0.0014278364,"about_ca_topic_score_gemma":0.0019702888,"teacher_disagreement_score":0.027464012,"about_ca_system_score_codex":0.001117146,"about_ca_system_score_gemma":0.0013851284,"threshold_uncertainty_score":0.09187627},"labels":[],"label_agreement":null},{"id":"W2295971361","doi":"","title":"UofL at TAC 2011 Guided Summarization Task","year":2011,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Automatic summarization; Computer science; Task (project management); Latent semantic analysis; Component (thermodynamics); Artificial intelligence; Natural language processing; Engineering","score_opus":0.013652950461757712,"score_gpt":0.25156476095512453,"score_spread":0.23791181049336682,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2295971361","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.24446897,0.007974276,0.28885907,0.0046331105,0.005258415,0.00586457,0.13908085,0.24293183,0.060928825],"genre_scores_gemma":[0.29550788,0.00076705037,0.3138653,0.0015312175,0.00092054514,0.0035725627,0.33718324,0.006868183,0.039783996],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9948428,0.0021750808,0.0004562586,0.0012327745,0.000886823,0.00040623557],"domain_scores_gemma":[0.98972005,0.0038294552,0.00044381406,0.0016764381,0.00381388,0.0005163864],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0049946364,0.0027982686,0.0017983529,0.0025563817,0.002380709,0.0024880061,0.0021558641,0.0029345904,0.016072871],"category_scores_gemma":[0.01699117,0.00054048817,0.001028765,0.0020779,0.0005010224,0.0029918018,0.002022981,0.0020157355,0.010498256],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017525452,0.00068426563,0.0015095642,0.0018441513,0.00032214588,0.0007918898,0.001370342,0.008373142,0.032803185,0.0017783996,0.574631,0.3741393],"study_design_scores_gemma":[0.002761411,0.0031505371,0.016655259,0.00036771863,0.00063581107,0.0012695878,0.003429154,0.22270112,0.12775573,0.013375638,0.60724264,0.0006554817],"about_ca_topic_score_codex":0.012236173,"about_ca_topic_score_gemma":0.021050746,"teacher_disagreement_score":0.016072871,"about_ca_system_score_codex":0.0011501318,"about_ca_system_score_gemma":0.001993998,"threshold_uncertainty_score":0.05376911},"labels":[],"label_agreement":null},{"id":"W2296305237","doi":"","title":"CCNU at TAC 2008:Proceeding on Using Semantic Method for Automated Summarization Yield","year":2008,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Automatic summarization; Computer science; WordNet; Natural language processing; Information retrieval; Artificial intelligence; Semantic similarity; Multi-document summarization; Sentence","score_opus":0.022827369871995865,"score_gpt":0.3106221733520606,"score_spread":0.28779480348006475,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2296305237","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020102957,0.0026241287,0.91305006,0.0029267247,0.0013524452,0.0010382341,0.0039363075,0.030904986,0.024064146],"genre_scores_gemma":[0.07612118,0.0006663813,0.88636565,0.0006050973,0.00055811094,0.00054890726,0.010068441,0.0038933435,0.02117296],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99600714,0.0018077365,0.00032243587,0.0004897126,0.0012043453,0.00016864206],"domain_scores_gemma":[0.9922654,0.0013044255,0.00026670354,0.0013510496,0.0045503746,0.00026190418],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0060328813,0.0012024381,0.0010430834,0.0040127016,0.0013713444,0.0030433524,0.0015349326,0.0014013696,0.012004811],"category_scores_gemma":[0.010309444,0.0004013578,0.0008128401,0.0025947134,0.00076758635,0.0027773015,0.0011488643,0.0015395721,0.00599874],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036249057,0.00021459596,0.0009657072,0.00053471216,0.00008171192,0.00019484105,0.0012749038,0.0038154395,0.04193208,0.011391663,0.16743958,0.7717923],"study_design_scores_gemma":[0.00023474245,0.0010908811,0.005669829,0.0002758245,0.00017545285,0.00062209397,0.0009536863,0.12438336,0.12154543,0.019953512,0.7248525,0.00024265869],"about_ca_topic_score_codex":0.008787881,"about_ca_topic_score_gemma":0.010855472,"teacher_disagreement_score":0.012004811,"about_ca_system_score_codex":0.0016674339,"about_ca_system_score_gemma":0.002128269,"threshold_uncertainty_score":0.04016012},"labels":[],"label_agreement":null},{"id":"W2296617725","doi":"10.1109/icmla.2015.181","title":"Summary Sentence Classification Using Stylometry","year":2015,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Stylometry; Automatic summarization; Computer science; Artificial intelligence; Natural language processing; Benchmark (surveying); Sentence; Set (abstract data type); Naive Bayes classifier; Task (project management); Correctness; Document classification; Information retrieval; Support vector machine","score_opus":0.07811868697048134,"score_gpt":0.32632876662074345,"score_spread":0.2482100796502621,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2296617725","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1230931,0.0032044998,0.83686393,0.0006752871,0.0005669985,0.0008522052,0.008132498,0.018746769,0.007864716],"genre_scores_gemma":[0.40340778,0.0011830707,0.5640384,0.0001535817,0.0006306519,0.00050919113,0.022115037,0.00053657475,0.00742571],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988952,0.00018859444,0.00019992414,0.00030156944,0.00034914206,0.00006557671],"domain_scores_gemma":[0.9962419,0.0009335172,0.00066473935,0.00045122963,0.0015772615,0.00013133824],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011926856,0.0010592474,0.0008993179,0.004778323,0.000542522,0.0015715514,0.0006956507,0.0006586654,0.0035873945],"category_scores_gemma":[0.006260346,0.00018202333,0.0007419109,0.0020981259,0.000300062,0.002034301,0.0007541587,0.0006755376,0.0031730912],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00040074982,0.00008900971,0.0053844876,0.0004823641,0.00010463639,0.000212074,0.00034311847,0.0061106673,0.04211745,0.0028914243,0.016880125,0.9249838],"study_design_scores_gemma":[0.00015956654,0.0012505229,0.03969918,0.00023679448,0.00055303326,0.0014819202,0.0011896655,0.6866723,0.15478154,0.023535233,0.090208426,0.00023192828],"about_ca_topic_score_codex":0.00076222594,"about_ca_topic_score_gemma":0.0015480268,"teacher_disagreement_score":0.004778323,"about_ca_system_score_codex":0.00044254298,"about_ca_system_score_gemma":0.0006565283,"threshold_uncertainty_score":0.012001038},"labels":[],"label_agreement":null},{"id":"W2301563053","doi":"10.11649/cs.2015.013","title":"The System of Register Labels in plWordNet","year":2015,"lang":"en","type":"article","venue":"Cognitive Studies | Études cognitives","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Register (sociolinguistics); Computer science; WordNet; Consistency (knowledge bases); Homogeneous; Natural language processing; Lexicography; Word (group theory); Artificial intelligence; Sample (material); Range (aeronautics); Lexical database; Linguistics; Mathematics","score_opus":0.07437191228357502,"score_gpt":0.34703343766099587,"score_spread":0.27266152537742083,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2301563053","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.068202995,0.00050634664,0.86604893,0.0017732762,0.0003008465,0.0005716279,0.0046572266,0.006248186,0.051690556],"genre_scores_gemma":[0.4120566,0.0006005547,0.5512009,0.00047108412,0.0001094917,0.0012513249,0.00993175,0.0025714335,0.021806832],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9968348,0.0010957002,0.00038108783,0.0010384369,0.0004835099,0.00016645923],"domain_scores_gemma":[0.99639064,0.001168065,0.00032916857,0.0011511781,0.00084356323,0.00011741412],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035423401,0.0007362741,0.00063084165,0.0036063707,0.0023128937,0.004786103,0.0014432563,0.0011985896,0.010387938],"category_scores_gemma":[0.010982612,0.001024004,0.0007974557,0.0033511447,0.0031607915,0.011378164,0.003176496,0.0017290037,0.006042548],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00043769152,0.000103193874,0.012320528,0.0010705561,0.000077782584,0.0004788394,0.014494885,0.008238,0.02016028,0.5256002,0.031168373,0.38584965],"study_design_scores_gemma":[0.00007280943,0.00017683441,0.010904205,0.0007062773,0.00017839044,0.0010320323,0.006704789,0.07862953,0.027971318,0.346332,0.52709526,0.00019664482],"about_ca_topic_score_codex":0.0065385066,"about_ca_topic_score_gemma":0.0079892455,"teacher_disagreement_score":0.010387938,"about_ca_system_score_codex":0.0019722749,"about_ca_system_score_gemma":0.002459517,"threshold_uncertainty_score":0.034751058},"labels":[],"label_agreement":null},{"id":"W2303324179","doi":"","title":"Symbolic assessment of free text answers in a second-language tutoring system","year":2006,"lang":"en","type":"article","venue":"Loughborough University Institutional Repository (Loughborough University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Computer science; Reading (process); Natural language processing; Artificial intelligence; Text messaging; Linguistics; World Wide Web","score_opus":0.005025435272538701,"score_gpt":0.20272819241370557,"score_spread":0.19770275714116686,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2303324179","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.54874855,0.0001698967,0.43170688,0.00030296028,0.000044037744,0.00040458914,0.0004423834,0.009538011,0.008642746],"genre_scores_gemma":[0.7939519,0.000068592424,0.19872557,0.00006038229,0.000018386252,0.00020674751,0.0005107468,0.00018604963,0.0062715565],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99760145,0.0010835195,0.000101269485,0.00036441788,0.0007667385,0.00008265753],"domain_scores_gemma":[0.9905318,0.0064902846,0.0004727198,0.0007378116,0.0014136587,0.0003537025],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016262393,0.00048015668,0.0006657236,0.0011532145,0.0005331137,0.002152952,0.001101042,0.00083848345,0.006401597],"category_scores_gemma":[0.014484591,0.0002011006,0.00020071439,0.0006086694,0.0006321679,0.0020622185,0.001645646,0.000551503,0.0018023571],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015608749,0.0009647447,0.010987799,0.00045091147,0.00005047959,0.0006024726,0.01037503,0.01112511,0.097112015,0.006173624,0.0046048975,0.855992],"study_design_scores_gemma":[0.00043487202,0.0024816417,0.030266888,0.00016865831,0.0001250325,0.0012311189,0.0064506605,0.6609286,0.23812921,0.017166302,0.042316694,0.00030031204],"about_ca_topic_score_codex":0.0015891825,"about_ca_topic_score_gemma":0.0032628928,"teacher_disagreement_score":0.006401597,"about_ca_system_score_codex":0.0007016339,"about_ca_system_score_gemma":0.00088942633,"threshold_uncertainty_score":0.021415472},"labels":[],"label_agreement":null},{"id":"W2306626104","doi":"10.18438/b8tg69","title":"3rd International EBL Conference - Abstracts of Papers and Poster Sessions","year":2006,"lang":"en","type":"article","venue":"Evidence Based Library and Information Practice","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Library science; Information retrieval","score_opus":0.00958749136796548,"score_gpt":0.25537097276272475,"score_spread":0.24578348139475928,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2306626104","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012164717,0.07653685,0.033722576,0.09561922,0.11848948,0.0017336425,0.014643025,0.007046539,0.64004403],"genre_scores_gemma":[0.013870989,0.0096776895,0.012238781,0.004469715,0.008487223,0.000412481,0.006588857,0.0015257368,0.94272864],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9959369,0.0006841024,0.0003033517,0.0003721751,0.0022340221,0.0004695084],"domain_scores_gemma":[0.99075526,0.0012162619,0.00044510802,0.0006041628,0.0051235673,0.0018555066],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.009579936,0.0015752533,0.0021678375,0.005444635,0.0023733794,0.011719793,0.0024324444,0.0052133454,0.3524635],"category_scores_gemma":[0.0077901757,0.0006882813,0.002016292,0.002426533,0.0011516233,0.0054843593,0.0044193645,0.0034171655,0.1812053],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004336789,0.00028537292,0.00027180323,0.00043180652,0.000032946442,0.00016116284,0.000088718,0.00021923309,0.002247931,0.0015449465,0.8825095,0.1117728],"study_design_scores_gemma":[0.00007087285,0.000101968275,0.001363741,0.00035157974,0.00003973082,0.00015238197,0.00019091304,0.00042724886,0.0033037942,0.002789977,0.9911678,0.00003996659],"about_ca_topic_score_codex":0.00354331,"about_ca_topic_score_gemma":0.010483436,"teacher_disagreement_score":0.3524635,"about_ca_system_score_codex":0.0044929017,"about_ca_system_score_gemma":0.005061146,"threshold_uncertainty_score":0.9236322},"labels":[],"label_agreement":null},{"id":"W2311776679","doi":"","title":"Neutral Pronouns: A Modest Proposal Whose Time Has Come","year":2005,"lang":"en","type":"article","venue":"Canadian women's studies","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":43,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Personal pronoun; Linguistics; Philosophy; Psychology","score_opus":0.01642469294013147,"score_gpt":0.2545566684269231,"score_spread":0.23813197548679163,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2311776679","genre_codex":"commentary","genre_gemma":"commentary","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"commentary","genre_consensus":"commentary","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007797067,0.011396026,0.051261473,0.80427366,0.022270441,0.00015032844,0.00050187245,0.000476882,0.10187228],"genre_scores_gemma":[0.37276527,0.02180811,0.14819972,0.18036744,0.02543627,0.0008510093,0.00099103,0.0011886504,0.24839249],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9939329,0.0013225015,0.00023734037,0.0010686222,0.0026942282,0.0007444098],"domain_scores_gemma":[0.98625344,0.0031266264,0.0004039069,0.0016850034,0.007271822,0.0012592489],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01349394,0.0008894954,0.001045886,0.0028435343,0.010039252,0.01112087,0.006083663,0.008769392,0.010842315],"category_scores_gemma":[0.021501595,0.0005524247,0.0007555889,0.002787008,0.024997346,0.019948356,0.004215734,0.0100206155,0.0029380373],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00001819069,0.0000072894973,0.000109451954,0.000052883614,0.000004356757,0.000017157796,0.000819762,0.00003389771,0.00020782299,0.9436448,0.037536506,0.017547736],"study_design_scores_gemma":[0.000033700962,0.000014648703,0.00049561216,0.00019499518,0.000039975188,0.000056638,0.0025476774,0.0004935599,0.00059121347,0.45696056,0.5385133,0.000058130838],"about_ca_topic_score_codex":0.2332766,"about_ca_topic_score_gemma":0.27503926,"teacher_disagreement_score":0.2332766,"about_ca_system_score_codex":0.015888175,"about_ca_system_score_gemma":0.044190623,"threshold_uncertainty_score":0.46383756},"labels":[],"label_agreement":null},{"id":"W2311921240","doi":"10.18653/v1/p16-1160","title":"A Character-level Decoder without Explicit Segmentation for Neural Machine Translation","year":2016,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":182,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"Natural Sciences and Engineering Research Council of Canada; Compute Canada; Samsung; Canada Research Chairs; Canadian Institute for Advanced Research; Nvidia","keywords":"Machine translation; Computer science; Character (mathematics); Phrase; Segmentation; Encoder; Artificial intelligence; Speech recognition; Natural language processing; Translation (biology); Sequence (biology); Language model; Mathematics","score_opus":0.06178756432298701,"score_gpt":0.3361341854023093,"score_spread":0.27434662107932234,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2311921240","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17084432,0.0020400763,0.8067059,0.0008160814,0.00027522858,0.000116738,0.0004547777,0.009451731,0.009295129],"genre_scores_gemma":[0.72094953,0.00056780543,0.26716152,0.0005325259,0.00010753338,0.0001501317,0.0019297978,0.000556511,0.008044713],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99923396,0.0003344243,0.00005748918,0.00018009602,0.0001287373,0.00006522167],"domain_scores_gemma":[0.9984675,0.0008669868,0.000065319044,0.00021269842,0.00032718753,0.00006026834],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016271323,0.0007114167,0.00065715745,0.00046088896,0.0004182561,0.000797118,0.0009830536,0.0013145024,0.0026488034],"category_scores_gemma":[0.00549576,0.00043610652,0.00037281448,0.0006687451,0.0005401076,0.0023765066,0.0009377974,0.0018522076,0.0017914574],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009698093,0.00033059236,0.0018063127,0.00043572756,0.00019724565,0.00026881005,0.0002522262,0.3921398,0.04936047,0.014544216,0.0072241486,0.5324707],"study_design_scores_gemma":[0.000025811072,0.00008595333,0.00018709482,0.000010795258,0.00002716719,0.000043335986,0.000022208027,0.9784534,0.014622256,0.0050046756,0.0015061769,0.0000111784575],"about_ca_topic_score_codex":0.0061003314,"about_ca_topic_score_gemma":0.011161726,"teacher_disagreement_score":0.0061003314,"about_ca_system_score_codex":0.0006604311,"about_ca_system_score_gemma":0.0012811567,"threshold_uncertainty_score":0.012129664},"labels":[],"label_agreement":null},{"id":"W2313179063","doi":"","title":"Zodiac : Insertion automatique des signes diacritiques du français","year":2014,"lang":"fr","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Pipeline (software); Word (group theory); Programming language; Operating system; Linguistics; Philosophy","score_opus":0.012136025988754523,"score_gpt":0.25587842533733307,"score_spread":0.24374239934857855,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2313179063","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.052167386,0.0012403189,0.61586237,0.0013567014,0.0014136552,0.00048254302,0.010595223,0.2800547,0.036827173],"genre_scores_gemma":[0.21813868,0.0011140542,0.6593932,0.00081055117,0.00020713995,0.00045489767,0.0184942,0.020865561,0.08052178],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99931395,0.000107364605,0.00002896129,0.00023307266,0.00022081255,0.00009584916],"domain_scores_gemma":[0.99934095,0.00021086149,0.00003995642,0.00014402506,0.0002059482,0.000058288028],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007644606,0.0015330263,0.0007506576,0.0012014243,0.00074353634,0.0017233544,0.0009841941,0.0009620644,0.048732832],"category_scores_gemma":[0.0021868777,0.000653548,0.0009638154,0.0006214248,0.00063599175,0.0014865239,0.001272584,0.0013711628,0.017492102],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013770483,0.00017293672,0.008180038,0.0011208731,0.0001505765,0.0013395613,0.001882114,0.012559959,0.07235664,0.024285091,0.26923782,0.6073373],"study_design_scores_gemma":[0.00024602245,0.00016521812,0.007623888,0.0001666733,0.0000791592,0.0014521522,0.00058651384,0.10111538,0.09337559,0.0060823904,0.7888348,0.00027223164],"about_ca_topic_score_codex":0.023821834,"about_ca_topic_score_gemma":0.030478338,"teacher_disagreement_score":0.048732832,"about_ca_system_score_codex":0.0011728571,"about_ca_system_score_gemma":0.001215036,"threshold_uncertainty_score":0.16302758},"labels":[],"label_agreement":null},{"id":"W2316254469","doi":"10.5715/jnlp.17.2_51","title":"A Written Child Corpus with Editing History Tags","year":2010,"lang":"en","type":"article","venue":"Journal of Natural Language Processing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Intecsea (Canada)","funders":"","keywords":"Computer science; Natural language processing; Linguistics; World Wide Web; Information retrieval; History; Philosophy","score_opus":0.004839098735990272,"score_gpt":0.23134153373138322,"score_spread":0.22650243499539294,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2316254469","genre_codex":"other","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.28265312,0.002059133,0.08646283,0.003225854,0.0015264857,0.0011994814,0.2939183,0.012022782,0.31693205],"genre_scores_gemma":[0.6000476,0.00074769824,0.119245596,0.00042471918,0.00024464767,0.0014638873,0.20735264,0.0043797414,0.06609343],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99837047,0.0007405399,0.00014182072,0.00038571976,0.00029135047,0.0000700693],"domain_scores_gemma":[0.9889326,0.008053843,0.00029187364,0.0012009498,0.0012598899,0.0002607941],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012225541,0.00093647325,0.00048473562,0.0024269337,0.0016361612,0.0020261114,0.0011524541,0.0010831954,0.047262747],"category_scores_gemma":[0.008363669,0.0007373323,0.00038004338,0.0034073666,0.00138918,0.0018506558,0.0015063279,0.0018415475,0.009919518],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019351337,0.00054314913,0.014957631,0.003889013,0.00013443665,0.010062837,0.045409102,0.00644786,0.032915514,0.06951451,0.39944443,0.41474643],"study_design_scores_gemma":[0.00028411506,0.0001824005,0.027009869,0.0005165873,0.0000959659,0.005261573,0.008233287,0.009245707,0.028404806,0.009970329,0.91062826,0.00016718931],"about_ca_topic_score_codex":0.023034979,"about_ca_topic_score_gemma":0.040778145,"teacher_disagreement_score":0.047262747,"about_ca_system_score_codex":0.0019422282,"about_ca_system_score_gemma":0.0026126613,"threshold_uncertainty_score":0.15810966},"labels":[],"label_agreement":null},{"id":"W2322334754","doi":"10.1093/ijl/ecu019","title":"Why Lexical Semantics is Important for E-Lexicography and Why it is Equally Important to Hide its Formal Representations from Users of Dictionaries","year":2014,"lang":"en","type":"article","venue":"International Journal of Lexicography","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Lexicography; Semantics (computer science); Linguistics; Lexical semantics; Computer science; Lexical item; Philosophy; Programming language","score_opus":0.018757546653284624,"score_gpt":0.3139004244293343,"score_spread":0.2951428777760497,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2322334754","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0649869,0.021586658,0.14388,0.55084324,0.0027469508,0.00013984067,0.0006170863,0.0015796466,0.21361963],"genre_scores_gemma":[0.81305385,0.016284721,0.09489547,0.032738127,0.0027134067,0.00015462484,0.0010251956,0.0019945938,0.037140027],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9896017,0.0054001613,0.0008654904,0.00085956714,0.002561985,0.00071114243],"domain_scores_gemma":[0.9596593,0.025031626,0.0021495325,0.004826905,0.0067530177,0.0015795681],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008647271,0.00028748353,0.00041170028,0.0020073624,0.002432884,0.012859866,0.0006870964,0.0027171336,0.012892298],"category_scores_gemma":[0.047465567,0.000735103,0.00039608745,0.0034310648,0.015236933,0.052287586,0.005021357,0.0048046676,0.0077998377],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024184053,0.000083685205,0.0049193087,0.00096994604,0.000033081935,0.00045492745,0.009185362,0.00061610964,0.005300232,0.73615545,0.05845731,0.18358268],"study_design_scores_gemma":[0.000039005725,0.000039477705,0.0032498618,0.00060750166,0.000022151788,0.00095645053,0.011785665,0.0010296501,0.0032730415,0.76323354,0.21569516,0.0000684946],"about_ca_topic_score_codex":0.0048096953,"about_ca_topic_score_gemma":0.003422117,"teacher_disagreement_score":0.012892298,"about_ca_system_score_codex":0.0029639967,"about_ca_system_score_gemma":0.004392639,"threshold_uncertainty_score":0.045731723},"labels":[],"label_agreement":null},{"id":"W2324256469","doi":"10.14195/2182-8830_4-1_3","title":"Sentence-Alignment and Application of Russian-German Multi-Target Parallel Corpora for Linguistic Analysis and Literary Studies","year":2015,"lang":"en","type":"article","venue":"Matlit Revista do Programa de Doutoramento em Materialidades da Literatura","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Université de Montréal","keywords":"Computer science; German; Natural language processing; Sentence; Rule-based machine translation; Artificial intelligence; Set (abstract data type); Linguistics; Corpus linguistics; Parsing; Programming language","score_opus":0.028210826197469768,"score_gpt":0.33515905640226135,"score_spread":0.30694823020479156,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2324256469","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20964463,0.0017416615,0.7058993,0.0010895703,0.0007674057,0.0013577935,0.019583004,0.020543044,0.039373603],"genre_scores_gemma":[0.29160348,0.0008148822,0.6608228,0.0000946092,0.00012337601,0.0012409188,0.032659363,0.0038879842,0.008752567],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99701643,0.0014513329,0.00035212006,0.00066513976,0.00043560588,0.0000793696],"domain_scores_gemma":[0.99648994,0.0017157716,0.00021204769,0.00063940225,0.0008602494,0.00008249224],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002341747,0.00066680746,0.0005545378,0.0040164115,0.0015039322,0.0018164053,0.0006997025,0.0004729536,0.010723804],"category_scores_gemma":[0.008009371,0.0005755907,0.0006111419,0.005216762,0.00047049468,0.0024178973,0.0019537394,0.0008197805,0.0046687555],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012358036,0.0006829362,0.006849896,0.003033928,0.00026706955,0.002766224,0.014945413,0.009174854,0.1021392,0.050923005,0.05157894,0.7564028],"study_design_scores_gemma":[0.00045852974,0.00061412033,0.04061935,0.00059733615,0.00037483504,0.0035988423,0.012066063,0.14627513,0.1777421,0.042135768,0.5752111,0.00030679218],"about_ca_topic_score_codex":0.0018727294,"about_ca_topic_score_gemma":0.0027422379,"teacher_disagreement_score":0.010723804,"about_ca_system_score_codex":0.00065368286,"about_ca_system_score_gemma":0.0010130048,"threshold_uncertainty_score":0.035874665},"labels":[],"label_agreement":null},{"id":"W2325416833","doi":"10.1515/css-2014-0005","title":"New Semiotics: Canada's New Initiatives","year":2014,"lang":"en","type":"article","venue":"Chinese Semiotic Studies","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Toronto; Toronto Metropolitan University","funders":"","keywords":"Semiotics; Semiotics of culture; Social semiotics; Visual semiotics; Sociology; Epistemology; Philosophy","score_opus":0.012913414801087299,"score_gpt":0.2889610163171869,"score_spread":0.2760476015160996,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2325416833","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.041397545,0.07237397,0.026614586,0.37067667,0.004341808,0.00021695813,0.00083443837,0.00040234375,0.48314166],"genre_scores_gemma":[0.81842256,0.047689456,0.024972737,0.012477,0.0009001543,0.00014949,0.0005080036,0.00035717915,0.094523415],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9891376,0.003084298,0.0003667847,0.00073735515,0.004778722,0.0018951959],"domain_scores_gemma":[0.97772855,0.0041999714,0.0005194999,0.0013561249,0.011944573,0.0042512342],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013476192,0.000752613,0.00077708974,0.0067034187,0.01968289,0.016436933,0.0022232006,0.0023218454,0.01001617],"category_scores_gemma":[0.011872085,0.00034913592,0.0006412342,0.009685934,0.03376477,0.006432821,0.0066733113,0.0039170613,0.00063953997],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":true,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00002695053,0.000016000342,0.0007308284,0.00014850001,0.000006223506,0.00010012525,0.0095466785,0.0003341507,0.00017997813,0.9077306,0.036134496,0.045045447],"study_design_scores_gemma":[0.000008426431,0.0000136336785,0.0018127704,0.00041952616,0.000009914778,0.00009665822,0.012963493,0.0007803532,0.00027875783,0.07392313,0.90965134,0.000041979947],"about_ca_topic_score_codex":0.97469395,"about_ca_topic_score_gemma":0.97146857,"teacher_disagreement_score":0.19456223,"about_ca_system_score_codex":0.19456223,"about_ca_system_score_gemma":0.31630266,"threshold_uncertainty_score":0.9341937},"labels":[],"label_agreement":null},{"id":"W2327588828","doi":"10.1515/cllt.2011.008","title":"Safe harbour: Ethics and accessibility in sociolinguistic corpus building","year":2011,"lang":"en","type":"article","venue":"Corpus Linguistics and Linguistic Theory","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Memorial University of Newfoundland","funders":"","keywords":"Globe; Corpus linguistics; Linguistics; Computer science; Applied linguistics; Computational linguistics; Sociology; Data sharing; Natural language processing; Psychology; Philosophy","score_opus":0.03852763765365219,"score_gpt":0.3088898463052056,"score_spread":0.2703622086515534,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2327588828","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02922508,0.0005501023,0.923438,0.010538179,0.00023684009,0.00085865665,0.00020205622,0.00028429923,0.03466664],"genre_scores_gemma":[0.46122473,0.00057052524,0.5246264,0.001000196,0.00021693061,0.0029691278,0.00043591566,0.0005265759,0.008429626],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.88154125,0.10628412,0.0030115442,0.0032406093,0.0050189937,0.0009035373],"domain_scores_gemma":[0.8403632,0.11643987,0.0046057496,0.028344356,0.008282952,0.001963956],"candidate_categories":["metaresearch","research_integrity"],"consensus_categories":[],"category_scores_codex":[0.0769566,0.0004990534,0.00063069287,0.0039945464,0.009099717,0.01130363,0.0025070044,0.0032554946,0.004696524],"category_scores_gemma":[0.14378342,0.0011785951,0.000492123,0.0038503455,0.028363353,0.020114088,0.01597708,0.004823613,0.0010699125],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006835554,0.00006499829,0.0018823007,0.00022157675,0.000015054922,0.00045229314,0.0890399,0.0020701191,0.0011969031,0.8446196,0.003942955,0.056425977],"study_design_scores_gemma":[0.00003393067,0.000054172888,0.0011876147,0.0006198585,0.000017936463,0.00056967436,0.030739145,0.011547205,0.003523467,0.8011368,0.15050516,0.000065071225],"about_ca_topic_score_codex":0.0034418032,"about_ca_topic_score_gemma":0.0049466738,"teacher_disagreement_score":0.9967445,"about_ca_system_score_codex":0.0034694406,"about_ca_system_score_gemma":0.008102502,"threshold_uncertainty_score":0.4069903},"labels":[],"label_agreement":null},{"id":"W2331201454","doi":"10.1016/s0008-4182(05)80001-8","title":"CJO: the way ahead","year":2005,"lang":"en","type":"article","venue":"Canadian Journal of Ophthalmology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science","score_opus":0.02005740573433736,"score_gpt":0.27980684930016736,"score_spread":0.25974944356583,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2331201454","genre_codex":"commentary","genre_gemma":"editorial","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"editorial","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0011883801,0.025447354,0.0063554314,0.7359098,0.16261467,0.00014263995,0.0012942984,0.001973418,0.065074034],"genre_scores_gemma":[0.032924265,0.051263433,0.04182704,0.3845135,0.11472978,0.00030346308,0.0056326315,0.002813375,0.3659924],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9937623,0.00088515464,0.00041195384,0.00045840838,0.0026330093,0.0018491925],"domain_scores_gemma":[0.9471542,0.0044411654,0.0015635698,0.0026282778,0.021464841,0.022747973],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012420434,0.0013043787,0.0012212575,0.003400558,0.0037498998,0.015566817,0.0033846484,0.017548565,0.15009044],"category_scores_gemma":[0.024561696,0.0005476163,0.0014120399,0.0021952633,0.0046213777,0.016039493,0.006422495,0.012266496,0.07359486],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001228644,0.000055767276,0.00020097148,0.00017792355,0.000007898236,0.000101434554,0.00005193414,0.00004949313,0.00021999875,0.013607875,0.91328084,0.07212294],"study_design_scores_gemma":[0.00003830571,0.000040524847,0.00044508753,0.00035115963,0.000010387779,0.00016685085,0.00033658012,0.00016592321,0.0002143051,0.012867721,0.98533034,0.000032878543],"about_ca_topic_score_codex":0.0155270435,"about_ca_topic_score_gemma":0.020069772,"teacher_disagreement_score":0.15009044,"about_ca_system_score_codex":0.0040440867,"about_ca_system_score_gemma":0.02484743,"threshold_uncertainty_score":0.50210255},"labels":[],"label_agreement":null},{"id":"W2332913205","doi":"","title":"Thinking on WH-Movement","year":2016,"lang":"en","type":"article","venue":"Canadian social science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Movement (music); Sentence; Linguistics; Syntax; Minimalist program; Operator (biology); Generalization; Position (finance); Computer science; Philosophy; Epistemology; Aesthetics","score_opus":0.01060123420781041,"score_gpt":0.2590078148669788,"score_spread":0.2484065806591684,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2332913205","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03460341,0.01600817,0.4098905,0.10207506,0.0028220522,0.00010346202,0.0004709821,0.0005580421,0.4334683],"genre_scores_gemma":[0.83007175,0.005012187,0.08191016,0.011043041,0.0017732088,0.0002253867,0.0003668719,0.00043594048,0.06916144],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9989114,0.00065386016,0.000039465656,0.00018878319,0.00012702911,0.000079463076],"domain_scores_gemma":[0.99919933,0.00044499148,0.0000377886,0.00012458951,0.0001449258,0.00004842227],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020896073,0.0006908338,0.0003705851,0.0016241217,0.0027015964,0.0036976982,0.0010072696,0.0017494431,0.0058021415],"category_scores_gemma":[0.0029494155,0.00038242695,0.0006420249,0.0010041522,0.013765457,0.017624497,0.0020875838,0.0031348835,0.0009393302],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000004090121,0.0000018143536,0.00007166481,0.0000140093125,0.0000029975888,0.000024771345,0.0020240464,0.000080117155,0.00011797659,0.9919783,0.0027807904,0.002899596],"study_design_scores_gemma":[0.0000043284467,0.0000060356497,0.00014717842,0.00003644485,0.0000037224972,0.00007332493,0.0012745889,0.00057772815,0.00019772258,0.9268894,0.070780285,0.000009106781],"about_ca_topic_score_codex":0.005215161,"about_ca_topic_score_gemma":0.004128632,"teacher_disagreement_score":0.0058021415,"about_ca_system_score_codex":0.0024063233,"about_ca_system_score_gemma":0.00086740305,"threshold_uncertainty_score":0.019410133},"labels":[],"label_agreement":null},{"id":"W2334738844","doi":"10.3765/bls.v31i2.3432","title":"Interpreting Yoruba bare nouns as generic","year":2005,"lang":"en","type":"article","venue":"Proceedings of the Annual Meeting of the Berkeley Linguistics Society","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Yoruba; Noun; Linguistics; Mathematics; Computer science; Philosophy; Natural language processing","score_opus":0.008058224990254279,"score_gpt":0.2509951541623305,"score_spread":0.2429369291720762,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2334738844","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.38809878,0.000927331,0.106756546,0.004351006,0.0017292727,0.00013979457,0.0008339141,0.0032746645,0.4938887],"genre_scores_gemma":[0.9515214,0.00036915214,0.018036487,0.0004754259,0.0002620368,0.00003195792,0.00052796863,0.0014915173,0.02728409],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9995931,0.00011097265,0.000026611904,0.00012784281,0.000059312486,0.0000822365],"domain_scores_gemma":[0.9995733,0.00009317641,0.00005776529,0.00009492783,0.00015322497,0.000027600767],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00060795137,0.0007696046,0.0006627521,0.0006583027,0.0021709856,0.0023467792,0.00068370166,0.0011865569,0.012139025],"category_scores_gemma":[0.0013892634,0.000730195,0.00055094104,0.0006722791,0.0024381308,0.004616535,0.0021549342,0.0015685682,0.0022551247],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00059465284,0.00008651137,0.0060407887,0.0004664698,0.000064225445,0.002888095,0.018776309,0.00047861008,0.048961855,0.86041075,0.020547053,0.040684734],"study_design_scores_gemma":[0.00017804517,0.00018790725,0.021292744,0.0005173931,0.0003220755,0.0039125658,0.027252996,0.0112793315,0.027307507,0.42533803,0.48220998,0.00020145957],"about_ca_topic_score_codex":0.011522256,"about_ca_topic_score_gemma":0.017395638,"teacher_disagreement_score":0.012139025,"about_ca_system_score_codex":0.0016244357,"about_ca_system_score_gemma":0.0006645924,"threshold_uncertainty_score":0.04060906},"labels":[],"label_agreement":null},{"id":"W2335704156","doi":"10.3115/1596431.1596436","title":"Using selectional profile distance to detect verb alternations","year":2004,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada; University of Sussex; Drexel University; University of Toronto; University of Minnesota","keywords":"WordNet; Computer science; Semantic similarity; Probability distribution; Similarity (geometry); Verb; Artificial intelligence; Natural language processing; Hierarchy; Distance measures; Measure (data warehouse); Mathematics; Data mining; Statistics","score_opus":0.020444490801634907,"score_gpt":0.3077059891086125,"score_spread":0.2872614983069776,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2335704156","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20136598,0.00035447557,0.79158354,0.00020528154,0.000059600916,0.00016288307,0.00051552436,0.0018798815,0.0038727655],"genre_scores_gemma":[0.8029635,0.00012795074,0.19432141,0.00006252275,0.000050412033,0.00014312676,0.00076120073,0.00017969965,0.0013901059],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9976731,0.0004539749,0.00024557512,0.0004469597,0.0010561007,0.00012424838],"domain_scores_gemma":[0.99351114,0.0031309484,0.0010004579,0.0006497637,0.0013958795,0.00031186204],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015579973,0.0005478139,0.0006746726,0.005398173,0.00066313747,0.0013691488,0.0012896133,0.0008041415,0.0016779109],"category_scores_gemma":[0.009847699,0.00027144095,0.0005424351,0.0024410507,0.0008256146,0.0037750134,0.0015353425,0.00090283365,0.00077162846],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00089583825,0.0005183535,0.0650723,0.0003925513,0.0003376959,0.0008865738,0.0013063322,0.02544684,0.058709152,0.0400162,0.0037757677,0.80264235],"study_design_scores_gemma":[0.0001230667,0.00044606606,0.029174674,0.000055328324,0.00011357638,0.0025116464,0.0006181926,0.81985575,0.039060928,0.09809536,0.009770363,0.00017505905],"about_ca_topic_score_codex":0.0016689848,"about_ca_topic_score_gemma":0.0021883713,"teacher_disagreement_score":0.005398173,"about_ca_system_score_codex":0.0006047523,"about_ca_system_score_gemma":0.00083929155,"threshold_uncertainty_score":0.008239567},"labels":[],"label_agreement":null},{"id":"W2336244146","doi":"10.14288/1.0078277","title":"Computers and content-based language learning","year":2010,"lang":"en","type":"article","venue":"cIRcle (University of British Columbia)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Content (measure theory); Natural language processing; Artificial intelligence; Linguistics; Mathematics","score_opus":0.006929201062446343,"score_gpt":0.18145471477152791,"score_spread":0.17452551370908156,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2336244146","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.22203617,0.011540966,0.50503886,0.026440004,0.0003364659,0.00053296296,0.00025791058,0.0025564553,0.23126012],"genre_scores_gemma":[0.83154786,0.0035222878,0.14291267,0.0009917418,0.00017212839,0.0003602563,0.00023216214,0.00037243223,0.019888528],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9948651,0.0035850233,0.00021424236,0.0005086237,0.0006293223,0.0001978427],"domain_scores_gemma":[0.97216785,0.023947656,0.0009315981,0.0014549065,0.0009802991,0.00051761937],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0052582524,0.0004482156,0.0004086757,0.003364439,0.0013558503,0.012648146,0.0014707098,0.001441181,0.005801542],"category_scores_gemma":[0.02199269,0.0004500626,0.00031730483,0.002743182,0.011721086,0.011513391,0.0042844005,0.0016168277,0.001143246],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010480604,0.0002169429,0.004684977,0.0007282103,0.000051605617,0.00036461366,0.087148406,0.0036885026,0.0066463556,0.52060795,0.0062492187,0.36950842],"study_design_scores_gemma":[0.00008847913,0.0001302714,0.0036179062,0.0008764987,0.000053064978,0.0007052275,0.03306081,0.02006652,0.007737649,0.7439203,0.18966186,0.00008140949],"about_ca_topic_score_codex":0.005745197,"about_ca_topic_score_gemma":0.004023917,"teacher_disagreement_score":0.012648146,"about_ca_system_score_codex":0.004607781,"about_ca_system_score_gemma":0.003114283,"threshold_uncertainty_score":0.033431947},"labels":[],"label_agreement":null},{"id":"W2337339290","doi":"","title":"Building Search Engines for Algonquian languages 1","year":2008,"lang":"en","type":"article","venue":"Algonquian Papers - Archive","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Humanities; Political science; Philosophy","score_opus":0.01559519304613614,"score_gpt":0.28747400961454855,"score_spread":0.2718788165684124,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2337339290","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.29077244,0.012987157,0.49062642,0.0023078744,0.0005762713,0.0022411328,0.052232746,0.082666464,0.0655895],"genre_scores_gemma":[0.35799518,0.003932896,0.5575731,0.0005058393,0.000115500516,0.0006424279,0.05130551,0.0025336503,0.025395932],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99928856,0.0001204942,0.00009422094,0.00021053095,0.00015270014,0.00013348383],"domain_scores_gemma":[0.99905795,0.00037975362,0.000058064714,0.00010585334,0.00034311233,0.000055178683],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007344612,0.0013401213,0.0011110647,0.007738719,0.0012002359,0.0031697915,0.0015104882,0.0011402658,0.014940782],"category_scores_gemma":[0.003972868,0.0007158117,0.001369251,0.0069695422,0.00064627256,0.004932831,0.0014248935,0.00065501523,0.007184548],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009098835,0.00029174573,0.01635705,0.003675033,0.00041684328,0.0014896004,0.0020875018,0.0444785,0.028952464,0.051694393,0.07759887,0.7720481],"study_design_scores_gemma":[0.00024375682,0.00038012798,0.011480352,0.0006809115,0.0005830686,0.0016253887,0.004395153,0.62312996,0.04299978,0.06041826,0.25388852,0.00017471575],"about_ca_topic_score_codex":0.03771756,"about_ca_topic_score_gemma":0.063789636,"teacher_disagreement_score":0.03771756,"about_ca_system_score_codex":0.001959151,"about_ca_system_score_gemma":0.0037358871,"threshold_uncertainty_score":0.074995995},"labels":[],"label_agreement":null},{"id":"W2337867146","doi":"","title":"A Data-Based Survey and Foregrounding Analysis on Chinese Chunk Research","year":2016,"lang":"en","type":"article","venue":"Cross-cultural communication","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Foregrounding; China; Data science; Scale (ratio); Statistical analysis; Computer science; Linguistics; History; Geography; Statistics; Cartography; Mathematics","score_opus":0.1222904576472699,"score_gpt":0.4646363634478028,"score_spread":0.3423459058005329,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2337867146","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.97132796,0.0025228984,0.0026311867,0.0004863649,0.000034024764,0.00057917525,0.016714044,0.00009089795,0.0056132837],"genre_scores_gemma":[0.96605444,0.0028491118,0.005094517,0.00032095512,0.000082154205,0.001275736,0.022455877,0.000051373732,0.0018158677],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99195075,0.0011724472,0.0023247523,0.0011381081,0.0026452886,0.00076870987],"domain_scores_gemma":[0.9610287,0.01761271,0.006835313,0.0022702876,0.010747249,0.001505755],"candidate_categories":["bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.0060377875,0.00033645012,0.00075067906,0.029089395,0.0018611808,0.00185967,0.0007946551,0.00051057927,0.0035494433],"category_scores_gemma":[0.024098126,0.00028522112,0.0005360868,0.030934494,0.0010277696,0.0023964173,0.0022004189,0.0003482673,0.0010798014],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00031148424,0.00019516163,0.81881976,0.0046473225,0.000116117866,0.0015606207,0.030418394,0.00022828807,0.007608605,0.0018286126,0.008426368,0.12583926],"study_design_scores_gemma":[0.000016281829,0.00019431692,0.92957467,0.00053428113,0.00018819765,0.00083192374,0.03133312,0.0010408403,0.0037030075,0.00048116516,0.03203255,0.000069678434],"about_ca_topic_score_codex":0.011065364,"about_ca_topic_score_gemma":0.013167117,"teacher_disagreement_score":0.9709106,"about_ca_system_score_codex":0.0016385934,"about_ca_system_score_gemma":0.0041036652,"threshold_uncertainty_score":0.03193122},"labels":[],"label_agreement":null},{"id":"W2338299047","doi":"10.1093/llc/fqv072","title":"Using Models of Lexical Style to Quantify Free Indirect Discourse in Modernist Fiction","year":2016,"lang":"en","type":"article","venue":"Digital Scholarship in the Humanities","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Style (visual arts); Narrative; Linguistics; Character (mathematics); Computer science; Literature; History; Art; Philosophy; Mathematics","score_opus":0.13476240488836103,"score_gpt":0.33901669510694593,"score_spread":0.2042542902185849,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2338299047","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7187298,0.00042873085,0.26248592,0.0004900712,0.000051485196,0.00015836825,0.0010098696,0.0006234997,0.016022298],"genre_scores_gemma":[0.97115505,0.00009453094,0.026278047,0.000031986034,0.000021132608,0.00010689797,0.0008012074,0.000088960835,0.0014221243],"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99871004,0.00059314596,0.00009606339,0.00030862543,0.00019508303,0.00009698012],"domain_scores_gemma":[0.9899317,0.0075779827,0.00087121496,0.0007755703,0.00060832,0.00023515771],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032398396,0.00067879056,0.00045598802,0.0047340835,0.00092523894,0.004738154,0.00061865675,0.00079109264,0.0024171085],"category_scores_gemma":[0.016772589,0.0005037006,0.00096258166,0.0025534334,0.0020176012,0.0046568713,0.0023451676,0.0011546572,0.0008419044],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007391644,0.000579221,0.28980973,0.00047490586,0.00052678847,0.0008932545,0.04285837,0.22095162,0.015904382,0.16522366,0.005365815,0.25667295],"study_design_scores_gemma":[0.00003125553,0.0001108442,0.060199395,0.00007410651,0.000059189846,0.000285704,0.0030618203,0.82336384,0.002089417,0.105406515,0.0052203657,0.000097567085],"about_ca_topic_score_codex":0.006426108,"about_ca_topic_score_gemma":0.009476189,"teacher_disagreement_score":0.006426108,"about_ca_system_score_codex":0.0017420747,"about_ca_system_score_gemma":0.0006660336,"threshold_uncertainty_score":0.01713413},"labels":[],"label_agreement":null},{"id":"W2339045699","doi":"","title":"RHYME ANALYZER: AN ANALYSIS TOOL FOR RAP LYRICS","year":2010,"lang":"de","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Rhyme; Lyrics; Disk formatting; Linguistics; Face (sociological concept); Syllable; Natural language processing; Style (visual arts); Computer science; Speech recognition; Artificial intelligence; Art; Literature; Poetry; Philosophy","score_opus":0.011953155457385572,"score_gpt":0.2976526711350873,"score_spread":0.28569951567770174,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2339045699","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010233222,0.0003276236,0.659864,0.000079734185,0.00010551656,0.0004336394,0.019747477,0.30216664,0.0070420597],"genre_scores_gemma":[0.07022002,0.0003203908,0.85206026,0.000104050094,0.00016872029,0.0011502476,0.02988073,0.032916255,0.0131794205],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99890924,0.00018771568,0.00018754318,0.0002663074,0.00037343756,0.00007579526],"domain_scores_gemma":[0.9976076,0.0010079177,0.00023078827,0.00043282215,0.0006054769,0.000115416755],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011778582,0.0025317855,0.0010470599,0.005272949,0.0007417271,0.0019056676,0.0016529212,0.00087843026,0.050019834],"category_scores_gemma":[0.005272473,0.0009869039,0.0009800988,0.0025068724,0.00041690818,0.0023713163,0.0019202589,0.0011551881,0.030312898],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00090826204,0.00017285538,0.0040473845,0.0014395202,0.00017002464,0.0009892648,0.0013873772,0.0015306452,0.08604059,0.005953396,0.13413721,0.7632235],"study_design_scores_gemma":[0.0004101928,0.0006022467,0.022841359,0.00038606513,0.00022137382,0.004719013,0.0016645241,0.16492471,0.2625799,0.014998053,0.52618325,0.00046944866],"about_ca_topic_score_codex":0.001256992,"about_ca_topic_score_gemma":0.0015179298,"teacher_disagreement_score":0.050019834,"about_ca_system_score_codex":0.0003696959,"about_ca_system_score_gemma":0.0007736676,"threshold_uncertainty_score":0.167333},"labels":[],"label_agreement":null},{"id":"W2339504674","doi":"","title":"Learning Machine Translation from In-domain and Out-of-domain Data","year":2012,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Machine translation; Computer science; Phrase; Natural language processing; Domain (mathematical analysis); Artificial intelligence; Translation (biology); Training set; Test data; Evaluation of machine translation; Language model; Machine learning; Example-based machine translation; Machine translation software usability; Mathematics","score_opus":0.03243895225103998,"score_gpt":0.2932855740325661,"score_spread":0.2608466217815261,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2339504674","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5702111,0.002717834,0.40976173,0.0010075694,0.00018542405,0.00021482866,0.0021146717,0.0061081494,0.0076787528],"genre_scores_gemma":[0.808058,0.00073016045,0.17823245,0.00020720043,0.000089438734,0.00030213682,0.008375671,0.00060733425,0.003397636],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99715143,0.0015018824,0.00023459675,0.0005328969,0.00045126007,0.00012798049],"domain_scores_gemma":[0.9832449,0.011798177,0.0005961705,0.0023601812,0.001825122,0.00017552587],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005517837,0.0013301141,0.0012324267,0.0018411303,0.0007376827,0.0017683954,0.0010855399,0.0013293219,0.0018093822],"category_scores_gemma":[0.026145006,0.00058621645,0.00094195525,0.0024683233,0.0009125376,0.003341201,0.0018403752,0.0020478873,0.002083627],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00088345126,0.00047011426,0.017683819,0.00073755224,0.00040610647,0.00040767144,0.00056392624,0.3696831,0.019483466,0.004758707,0.0057988064,0.5791233],"study_design_scores_gemma":[0.00004845671,0.00022869071,0.005956939,0.000040677976,0.000073641655,0.00019507301,0.00017691425,0.96235996,0.02019176,0.0072054225,0.0034824326,0.0000401322],"about_ca_topic_score_codex":0.0033931357,"about_ca_topic_score_gemma":0.0051140036,"teacher_disagreement_score":0.005517837,"about_ca_system_score_codex":0.00090198626,"about_ca_system_score_gemma":0.0012096098,"threshold_uncertainty_score":0.02918148},"labels":[],"label_agreement":null},{"id":"W2341390813","doi":"","title":"User Delegation in the CLARIN Infrastructure","year":2015,"lang":"en","type":"article","venue":"Data Archiving and Networked Services (DANS)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canarie","funders":"","keywords":"Delegation; Computer science; Implementation; World Wide Web; Computer security; Knowledge management; Software engineering; Political science","score_opus":0.01814205319153569,"score_gpt":0.2663127690327292,"score_spread":0.24817071584119352,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2341390813","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.37264413,0.00038138533,0.53073424,0.0034499483,0.00017896545,0.001016203,0.00028417743,0.023026103,0.06828485],"genre_scores_gemma":[0.90665287,0.0001246368,0.075790204,0.0003745648,0.00004448642,0.00041758298,0.0002362754,0.0010248737,0.015334437],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9775449,0.012937983,0.0017389331,0.0024633182,0.0034796628,0.0018352323],"domain_scores_gemma":[0.9626592,0.017241076,0.0013914681,0.015417543,0.0020178796,0.0012728657],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.016484633,0.0006790448,0.00064690545,0.0010103933,0.0029721472,0.005393861,0.0018158936,0.0015828212,0.006691907],"category_scores_gemma":[0.028675616,0.000645048,0.00061602204,0.000700822,0.0031951563,0.009572862,0.010323299,0.0022435084,0.001656421],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002357685,0.0016275654,0.023898782,0.0010150155,0.000107464395,0.004828815,0.23459075,0.006876947,0.06274074,0.15588693,0.022830728,0.48323858],"study_design_scores_gemma":[0.00025278958,0.000653631,0.013311355,0.0006652298,0.00012874808,0.0040307404,0.045630105,0.09889525,0.094506875,0.057592835,0.68384326,0.0004891648],"about_ca_topic_score_codex":0.004200538,"about_ca_topic_score_gemma":0.0037004484,"teacher_disagreement_score":0.016484633,"about_ca_system_score_codex":0.002594131,"about_ca_system_score_gemma":0.002596948,"threshold_uncertainty_score":0.08718014},"labels":[],"label_agreement":null},{"id":"W2341475035","doi":"10.7603/s40742-015-0001-6","title":"Automatic and Semi-Automatic Test Generation for Introductory Linguistics Courses Using Natural Language Processing Resources and Text Corpora","year":2015,"lang":"en","type":"article","venue":"GSTF Journal on Education","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Saskatchewan; University of Waterloo","funders":"","keywords":"Computer science; Grammar; Natural language processing; Test (biology); Identification (biology); Artificial intelligence; Phrase structure rules; Corpus linguistics; Computational linguistics; Applied linguistics; Linguistics; Natural language; English grammar; Section (typography); Phrase; Generative grammar","score_opus":0.02455342562358284,"score_gpt":0.318540871417785,"score_spread":0.29398744579420216,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2341475035","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.29368278,0.00030136624,0.5832154,0.00040864525,0.00021151091,0.0028219214,0.008793071,0.104398385,0.006166988],"genre_scores_gemma":[0.32294455,0.000097336524,0.6395362,0.00016764484,0.00007189071,0.0018554885,0.026948564,0.003779425,0.0045989132],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9967482,0.0013374857,0.00033604781,0.00094028964,0.0005032182,0.00013481808],"domain_scores_gemma":[0.9768028,0.015906354,0.00090721087,0.0018003137,0.0040181503,0.00056511495],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035092183,0.0012986722,0.0009850257,0.0031957896,0.0006279528,0.0013327727,0.0021697888,0.0009283048,0.010382943],"category_scores_gemma":[0.019108122,0.0006135289,0.0008140179,0.0012952153,0.00056937634,0.0012214838,0.0015246498,0.0009149841,0.0044904198],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008620917,0.00090389274,0.010397621,0.0008523728,0.0001123806,0.001026547,0.0010841516,0.026948014,0.07065533,0.0017406838,0.03276226,0.85265464],"study_design_scores_gemma":[0.00074246473,0.00095312094,0.028237397,0.00019652394,0.00016074287,0.00085598935,0.0011103228,0.72664106,0.19476114,0.004955691,0.041195545,0.00019007162],"about_ca_topic_score_codex":0.004466293,"about_ca_topic_score_gemma":0.0057196827,"teacher_disagreement_score":0.010382943,"about_ca_system_score_codex":0.00091822504,"about_ca_system_score_gemma":0.0014132444,"threshold_uncertainty_score":0.034734428},"labels":[],"label_agreement":null},{"id":"W2345905986","doi":"10.33011/lilt.v10i.1357","title":"Probabilistic Type Theory and Natural Language Semantics","year":2015,"lang":"en","type":"article","venue":"Linguistic Issues in Language Technology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":68,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Göteborgs Universitet; Atomic Energy of Canada Limited; Economic and Social Research Council; Vetenskapsrådet; Wenner-Gren Stiftelserna","keywords":"Probabilistic logic; Computer science; Computational semantics; Semantics (computer science); Operational semantics; Artificial intelligence; Formal semantics (linguistics); Natural language processing; Natural language; Well-founded semantics; Denotational semantics; Programming language","score_opus":0.008523567500049291,"score_gpt":0.3048502520302477,"score_spread":0.2963266845301984,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2345905986","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0056257844,0.007062768,0.95009905,0.008705399,0.0007177234,0.00006641,0.00027702018,0.00031698367,0.027128803],"genre_scores_gemma":[0.46239024,0.0143326325,0.49630326,0.0036268597,0.0055371905,0.00063907675,0.0008530948,0.00037723407,0.015940433],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9913521,0.0040547205,0.000698185,0.0011017122,0.0023537152,0.00043960777],"domain_scores_gemma":[0.98277676,0.012191833,0.0010883994,0.0018754528,0.0016273063,0.00044025356],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011017269,0.0010182026,0.0017158729,0.0043289526,0.002610335,0.0070451545,0.003098661,0.0033823554,0.0051143453],"category_scores_gemma":[0.019402783,0.0010523294,0.0027318357,0.00557476,0.014234163,0.02060414,0.0035325948,0.0053428654,0.0011248115],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000049310956,0.00000353519,0.00006475757,0.000031217427,0.000008425605,0.000025024792,0.00009269613,0.0012170867,0.00003181568,0.9946666,0.0004898362,0.0033640547],"study_design_scores_gemma":[0.000003950834,0.000003387672,0.000024034855,0.000017434026,0.0000038339026,0.000036043788,0.000023881086,0.002324055,0.000032632983,0.9938752,0.0036491696,0.000006265156],"about_ca_topic_score_codex":0.0028022085,"about_ca_topic_score_gemma":0.0014642543,"teacher_disagreement_score":0.011017269,"about_ca_system_score_codex":0.0042873425,"about_ca_system_score_gemma":0.0027954918,"threshold_uncertainty_score":0.058265567},"labels":[],"label_agreement":null},{"id":"W2349546778","doi":"","title":"An Analysis of the NP_1+(NP_2+VP) Pattern","year":2002,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"CAE (Canada)","funders":"","keywords":"Possessive; Predicate (mathematical logic); Linguistics; Pragmatics; Grammar; Focus (optics); Computer science; Natural language processing; Programming language; Philosophy","score_opus":0.0127527393876662,"score_gpt":0.2587269524751525,"score_spread":0.2459742130874863,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2349546778","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7472463,0.00044838426,0.1204858,0.0025922281,0.00019968624,0.0001789196,0.004306895,0.0012024043,0.12333933],"genre_scores_gemma":[0.9483233,0.00015627126,0.035318114,0.00022841612,0.000050263334,0.00004808967,0.0017219856,0.00024888956,0.013904602],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9997063,0.00005548052,0.000021464028,0.00006057051,0.000116034535,0.000040116298],"domain_scores_gemma":[0.99915326,0.0003917358,0.00009251871,0.00012990923,0.00019148161,0.000041081246],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00022015668,0.00014532573,0.00013637428,0.0006646538,0.0008252756,0.00077330496,0.0003235133,0.00044336825,0.006758279],"category_scores_gemma":[0.0016781794,0.00012669824,0.00026653658,0.00116208,0.00073055975,0.0011599566,0.00050145644,0.00054573576,0.0009549023],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00094711175,0.00026226958,0.05162783,0.0006529203,0.000112307134,0.013251949,0.012220047,0.00212181,0.13985513,0.49549854,0.043728147,0.2397219],"study_design_scores_gemma":[0.00011325132,0.00032570952,0.15421543,0.00016023616,0.00022931132,0.026091503,0.009569414,0.04888097,0.09691547,0.28135785,0.38199466,0.00014627689],"about_ca_topic_score_codex":0.002682451,"about_ca_topic_score_gemma":0.0039906744,"teacher_disagreement_score":0.006758279,"about_ca_system_score_codex":0.00034463432,"about_ca_system_score_gemma":0.00035550387,"threshold_uncertainty_score":0.022608697},"labels":[],"label_agreement":null},{"id":"W2359014521","doi":"10.3765/sp.9.6","title":"How similar is similar enough?","year":2016,"lang":"en","type":"article","venue":"Semantics and Pragmatics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Counterfactual conditional; Counterfactual thinking; Triviality; Assertion; Context (archaeology); Similarity (geometry); Computer science; Consistency (knowledge bases); Statement (logic); Set (abstract data type); Possible world; Interpretation (philosophy); Truth condition; Utterance; Linguistics; Natural language processing; Epistemology; Theoretical computer science; Mathematical economics; Mathematics; Artificial intelligence; Philosophy; Pure mathematics","score_opus":0.010354558840069528,"score_gpt":0.23645759333230923,"score_spread":0.2261030344922397,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2359014521","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3968289,0.0037357348,0.46346664,0.037019193,0.0007869608,0.00029160568,0.0009458465,0.00041798363,0.0965071],"genre_scores_gemma":[0.96318287,0.00041206952,0.03263681,0.001048235,0.00019217201,0.00007523225,0.00027437214,0.000071975635,0.0021062708],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9883409,0.0067546917,0.0006148486,0.0029167081,0.0009302461,0.00044248017],"domain_scores_gemma":[0.9664665,0.02589583,0.002715812,0.0022907958,0.0018075152,0.0008235751],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009472286,0.00043469825,0.0007561208,0.002125765,0.0030734725,0.0048725866,0.0015257376,0.002485598,0.008932292],"category_scores_gemma":[0.050480835,0.0003408819,0.00085263123,0.0019833501,0.00615396,0.014771667,0.0042300406,0.0021219067,0.0009799149],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00039377538,0.00009862266,0.011465682,0.0003106292,0.00016051851,0.00072110613,0.017584821,0.00086173834,0.002861036,0.9149264,0.0045596245,0.046055976],"study_design_scores_gemma":[0.000019346375,0.00007299651,0.0043052565,0.00012817005,0.00012496482,0.00073976634,0.011895348,0.008290535,0.0027262478,0.94902647,0.022604039,0.00006692636],"about_ca_topic_score_codex":0.0020874229,"about_ca_topic_score_gemma":0.0023390795,"teacher_disagreement_score":0.009472286,"about_ca_system_score_codex":0.0017273395,"about_ca_system_score_gemma":0.0008277648,"threshold_uncertainty_score":0.050094783},"labels":[],"label_agreement":null},{"id":"W237500786","doi":"10.1163/9789401206884_003","title":"I haven’t drank in weeks: the use of past tense forms as past participles in English corpora","year":2011,"lang":"en","type":"book-chapter","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":32,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Phenomenon; Past tense; Participle; Modal verb; Verb; Linguistics; Present tense; Present perfect; History; Computer science; Psychology; Artificial intelligence; Philosophy","score_opus":0.05434846318847414,"score_gpt":0.2514001615852677,"score_spread":0.19705169839679354,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W237500786","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8740529,0.009760476,0.038179558,0.0016055384,0.00014076762,0.00012747796,0.0068063247,0.00048118448,0.068845876],"genre_scores_gemma":[0.9587613,0.0033238146,0.02429287,0.00026399508,0.00008041442,0.00013717989,0.0077541685,0.0005145728,0.004871645],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99853015,0.00072293414,0.00018602,0.00023730275,0.000280483,0.000043100674],"domain_scores_gemma":[0.9823691,0.014022494,0.0009628593,0.0017332295,0.0007697279,0.0001425343],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021309888,0.00024340197,0.0002932761,0.0026455212,0.0011449112,0.002220094,0.0004779315,0.00033153011,0.0024206154],"category_scores_gemma":[0.010388848,0.00040339903,0.00019928989,0.0052077495,0.0015583298,0.0035104891,0.0011882161,0.000875801,0.00068166817],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005821932,0.00038558248,0.08407922,0.0040230067,0.00017692099,0.002613206,0.17437296,0.0029568188,0.054531783,0.0951467,0.0457086,0.53542304],"study_design_scores_gemma":[0.0001139801,0.00020357709,0.35387155,0.0017763618,0.0002628345,0.008470532,0.037333664,0.01986837,0.033212893,0.041444067,0.50314397,0.00029821307],"about_ca_topic_score_codex":0.003693437,"about_ca_topic_score_gemma":0.014109282,"teacher_disagreement_score":0.003693437,"about_ca_system_score_codex":0.00051808864,"about_ca_system_score_gemma":0.0005066736,"threshold_uncertainty_score":0.011269867},"labels":[],"label_agreement":null},{"id":"W2377001998","doi":"","title":"Robust Character-Based Chinese Spoken Language Understanding with Domain Information","year":2010,"lang":"en","type":"article","venue":"Microcomputer applications","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Robustness (evolution); Exploit; Natural language processing; Artificial intelligence; Spoken language; Language model; Character (mathematics); Test set; Speech recognition; Word (group theory); Domain (mathematical analysis); Set (abstract data type); Linguistics","score_opus":0.008692664868227845,"score_gpt":0.2299964082547704,"score_spread":0.22130374338654257,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2377001998","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12248005,0.00040782432,0.8690953,0.00014871082,0.00004199528,0.00013995638,0.00032390593,0.0050067324,0.0023555502],"genre_scores_gemma":[0.7373708,0.00023581328,0.25509784,0.00013326843,0.00003832085,0.00017535406,0.0018272407,0.00028661537,0.0048347483],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99901485,0.00024645007,0.00007938692,0.00037987452,0.00018611243,0.00009320973],"domain_scores_gemma":[0.99846613,0.0006238943,0.000108991924,0.0003406555,0.0004099766,0.000050397233],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008419549,0.0009933198,0.00086060463,0.0005845569,0.0002798028,0.0008678148,0.00090917555,0.00070764375,0.002162252],"category_scores_gemma":[0.0026529827,0.00028475322,0.00049868156,0.0005825265,0.00038655542,0.0021266604,0.0011925692,0.00084286154,0.0014081098],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005618907,0.00015351543,0.0018030215,0.0003320615,0.00013927015,0.0004477304,0.00071270973,0.07207589,0.22200876,0.002814776,0.0027552242,0.69619524],"study_design_scores_gemma":[0.00002948558,0.00016146974,0.00207193,0.000010014355,0.000072517534,0.0001960208,0.00018807141,0.89009804,0.10255651,0.002236928,0.0023265963,0.000052389067],"about_ca_topic_score_codex":0.0034526405,"about_ca_topic_score_gemma":0.003414054,"teacher_disagreement_score":0.0034526405,"about_ca_system_score_codex":0.0003454392,"about_ca_system_score_gemma":0.0007754239,"threshold_uncertainty_score":0.007233441},"labels":[],"label_agreement":null},{"id":"W2382556257","doi":"10.1016/j.physio.2015.03.3696","title":"Our journey through a knowledge translation project","year":2015,"lang":"en","type":"article","venue":"Physiotherapy","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; Université de Montréal; Centre for Interdisciplinary Research in Rehabilitation","funders":"","keywords":"Knowledge translation; Knowledge management; Medicine; Medical education; Computer science","score_opus":0.1416903611267047,"score_gpt":0.42535138767951153,"score_spread":0.28366102655280684,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2382556257","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11053309,0.02611583,0.3259152,0.34734046,0.0070816926,0.0009731101,0.0022441857,0.0032364572,0.17655997],"genre_scores_gemma":[0.43262872,0.018007157,0.42503625,0.022474376,0.001522972,0.0008470704,0.007253435,0.002221083,0.09000892],"study_design_codex":"design_other","study_design_gemma":"qualitative","domain_scores_codex":[0.98531544,0.009662603,0.000774778,0.0009687275,0.0025039257,0.0007746447],"domain_scores_gemma":[0.9793596,0.009875411,0.00058977905,0.0027926092,0.004887786,0.002494754],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.018557938,0.0006062421,0.0007612163,0.0017822116,0.005356359,0.011525558,0.0021502892,0.0046280287,0.014684766],"category_scores_gemma":[0.026780702,0.00048679428,0.00074958534,0.0029777451,0.0050484356,0.012915961,0.007405422,0.0045761294,0.005249381],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00033202968,0.0010309482,0.0019121425,0.0012192401,0.00006434369,0.0010407438,0.03750696,0.0020393357,0.0062154834,0.19910467,0.08930882,0.6602252],"study_design_scores_gemma":[0.000088810186,0.00032654745,0.001928608,0.0013528396,0.00005782103,0.0011840262,0.032612942,0.005017654,0.006287783,0.13458985,0.8164494,0.00010368685],"about_ca_topic_score_codex":0.006589391,"about_ca_topic_score_gemma":0.0046666614,"teacher_disagreement_score":0.018557938,"about_ca_system_score_codex":0.0027306604,"about_ca_system_score_gemma":0.013602588,"threshold_uncertainty_score":0.09814495},"labels":[],"label_agreement":null},{"id":"W2385055279","doi":"","title":"New Multilingual Information Retrieval Model Based on Latent Interlingua Semantics","year":2010,"lang":"en","type":"article","venue":"Journal of Chinese Computer Systems","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Cross-language information retrieval; Exploit; Semantics (computer science); Parallel corpora; Information retrieval; Space (punctuation); Machine translation; Programming language","score_opus":0.008407938687764304,"score_gpt":0.2667925029411465,"score_spread":0.2583845642533822,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2385055279","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014089987,0.0016468985,0.9770364,0.0008358122,0.00015568567,0.00015330386,0.0006598512,0.0018654054,0.0035567086],"genre_scores_gemma":[0.39096093,0.0022320189,0.58748835,0.0008145704,0.00045137168,0.00081464107,0.0039292183,0.00050166954,0.012807233],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99854505,0.00042761725,0.00013068842,0.0004440894,0.00035159674,0.000100918776],"domain_scores_gemma":[0.9991443,0.00031461194,0.0000805831,0.00012258849,0.00029701594,0.0000407402],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017044595,0.0009104555,0.0013842337,0.003355717,0.000797976,0.0021598192,0.0015311342,0.0009779787,0.0032136037],"category_scores_gemma":[0.0030285777,0.0004205255,0.0014166659,0.0034485955,0.00072213233,0.0072526424,0.0015052073,0.0012803986,0.0019104752],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00085188745,0.00067830866,0.0037731607,0.0012059418,0.000679314,0.0006339637,0.0013347567,0.08530878,0.022205736,0.15321925,0.031925913,0.69818306],"study_design_scores_gemma":[0.00012953313,0.00019121877,0.000921529,0.00005554958,0.0002221498,0.00041997278,0.0001796927,0.9041,0.0050888825,0.07137462,0.017219966,0.00009686347],"about_ca_topic_score_codex":0.00616193,"about_ca_topic_score_gemma":0.006771999,"teacher_disagreement_score":0.00616193,"about_ca_system_score_codex":0.0012777086,"about_ca_system_score_gemma":0.0016535249,"threshold_uncertainty_score":0.012252152},"labels":[],"label_agreement":null},{"id":"W2385262300","doi":"","title":"Primary Research on the Proofreading of Part-of-speech Tagging of Uighur Words","year":2006,"lang":"en","type":"article","venue":"Microcomputer applications","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Proofreading; Natural language processing; Task (project management); Consistency (knowledge bases); Correctness; Artificial intelligence; Part of speech; Word (group theory); Process (computing); Speech recognition; Linguistics","score_opus":0.03335553315933602,"score_gpt":0.32095380129334616,"score_spread":0.2875982681340101,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2385262300","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.042362884,0.006329725,0.93582445,0.0010653974,0.00064263353,0.00028505435,0.00019591686,0.0020114684,0.011282433],"genre_scores_gemma":[0.22651012,0.0046664965,0.7528914,0.0007410763,0.0008171049,0.00022272357,0.0008966503,0.0012375473,0.012016801],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9781516,0.010013205,0.0019277302,0.0050342423,0.004353468,0.0005197264],"domain_scores_gemma":[0.87861735,0.062417235,0.0062765386,0.03002345,0.021860378,0.00080507086],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.016733158,0.001003014,0.0014232431,0.0064705485,0.0020885128,0.003831406,0.002988218,0.0018710892,0.006264898],"category_scores_gemma":[0.07069727,0.0007954263,0.0012166249,0.0049839746,0.00529039,0.011126645,0.0019167317,0.0020429338,0.0035846482],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00038501868,0.00015609497,0.0068829553,0.0020522184,0.000119444085,0.0007081794,0.0048346934,0.0024982605,0.07471018,0.08250609,0.003946161,0.8212006],"study_design_scores_gemma":[0.00008768403,0.00078965013,0.012971614,0.0010181022,0.0004142017,0.0072681974,0.003818901,0.07551977,0.6209641,0.102523595,0.17424853,0.000375574],"about_ca_topic_score_codex":0.0018536681,"about_ca_topic_score_gemma":0.0013993938,"teacher_disagreement_score":0.016733158,"about_ca_system_score_codex":0.0016157612,"about_ca_system_score_gemma":0.0027123697,"threshold_uncertainty_score":0.08849442},"labels":[],"label_agreement":null},{"id":"W2386889609","doi":"","title":"The Progress of the Chinese Sentence Grou Pprocessing","year":2009,"lang":"en","type":"article","venue":"Microcomputer applications","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Sentence; Computer science; Natural language processing; Parsing; Context (archaeology); Artificial intelligence; Sentence processing; Quality (philosophy); Group (periodic table)","score_opus":0.004296609887668316,"score_gpt":0.2611654958513796,"score_spread":0.25686888596371127,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2386889609","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07444404,0.0046086465,0.90739125,0.0016525198,0.00047493205,0.00030615163,0.0007152447,0.0035946877,0.0068126186],"genre_scores_gemma":[0.22491322,0.0034969887,0.76123834,0.00040537832,0.0006379735,0.00025824932,0.0016532835,0.00075863313,0.006637878],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99676955,0.00088788784,0.00027769723,0.0009395183,0.0009681223,0.00015721783],"domain_scores_gemma":[0.994842,0.0015789147,0.0002895888,0.00093090034,0.0022334387,0.00012505785],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003839579,0.0011844091,0.0010245182,0.003163999,0.0011947334,0.0019305357,0.0015068651,0.00066998583,0.0056213373],"category_scores_gemma":[0.008463259,0.0005957488,0.0012515206,0.0031163623,0.001236319,0.0047125462,0.0010888807,0.0018599301,0.0014694331],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022108552,0.000049209102,0.002899393,0.0011100696,0.000076531855,0.0002799977,0.0017403311,0.0029996787,0.0581437,0.023681823,0.008151394,0.9006468],"study_design_scores_gemma":[0.00022687524,0.0010548405,0.030271238,0.0003694625,0.0010594721,0.001974041,0.0024149073,0.23865713,0.44355956,0.10221994,0.17753254,0.0006600025],"about_ca_topic_score_codex":0.0051265783,"about_ca_topic_score_gemma":0.0040448993,"teacher_disagreement_score":0.0056213373,"about_ca_system_score_codex":0.00091791805,"about_ca_system_score_gemma":0.002076146,"threshold_uncertainty_score":0.020305872},"labels":[],"label_agreement":null},{"id":"W2392705514","doi":"","title":"Transformational Analysis in Ma′s Grammar","year":2003,"lang":"en","type":"article","venue":"Nanjing Linye Daxue xuebao","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Discovery Air (Canada)","funders":"","keywords":"Transformational grammar; Relational grammar; Rhetoric; Transformational leadership; Grammar; Linguistics; Computer science; Phrase structure rules; Emergent grammar; Natural language processing; Psychology; Philosophy","score_opus":0.009087900058525855,"score_gpt":0.2564190571619433,"score_spread":0.24733115710341744,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2392705514","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02571434,0.0023910294,0.78981644,0.004635746,0.00034371653,0.00013534816,0.00026242906,0.000748956,0.17595209],"genre_scores_gemma":[0.69984275,0.0016019909,0.25375327,0.00083413953,0.0003066235,0.00022895257,0.00038086102,0.00040933164,0.042642135],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99862444,0.0005812011,0.00010674776,0.00033006608,0.0002587775,0.00009877244],"domain_scores_gemma":[0.9990939,0.00044055982,0.00007249095,0.00015027967,0.00021141335,0.000031324755],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012895672,0.00045229818,0.00039222252,0.0020593363,0.0017713509,0.002160214,0.0005916088,0.0006032725,0.004917258],"category_scores_gemma":[0.0023365482,0.0003238734,0.0013044626,0.0015645347,0.006167992,0.0044456394,0.0015457894,0.0018726224,0.00089768437],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000047516683,0.0000034650252,0.000085882355,0.00001991208,0.000006034691,0.000069511785,0.0010423776,0.00042114177,0.00039449372,0.9869185,0.0011021435,0.009931802],"study_design_scores_gemma":[0.000004549759,0.000007986469,0.00020022065,0.000013831425,0.000010779972,0.00009171203,0.00021225263,0.0034970425,0.0006927123,0.9582592,0.036999952,0.00000970035],"about_ca_topic_score_codex":0.0033466711,"about_ca_topic_score_gemma":0.0027558699,"teacher_disagreement_score":0.004917258,"about_ca_system_score_codex":0.0022375463,"about_ca_system_score_gemma":0.0011502436,"threshold_uncertainty_score":0.016449869},"labels":[],"label_agreement":null},{"id":"W2394256012","doi":"","title":"Construction of Specialized Corpus and Its Application in Machine Translation","year":2008,"lang":"en","type":"article","venue":"Microcomputer applications","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Machine translation; Natural language processing; Machine translation software usability; Example-based machine translation; Artificial intelligence; Translation (biology); Process (computing); Set (abstract data type); Automation; Parallel corpora; Corpus linguistics; Text corpus; Rule-based machine translation; Computer-assisted translation; Programming language","score_opus":0.012461785397234302,"score_gpt":0.250429558333703,"score_spread":0.2379677729364687,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2394256012","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.034473192,0.0007007307,0.9451128,0.00031509914,0.00022730658,0.00063011533,0.00081662444,0.0021756887,0.015548413],"genre_scores_gemma":[0.12470548,0.00076147146,0.86392874,0.00009427884,0.00012484689,0.0009152745,0.0037763487,0.0005226304,0.005170949],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99840873,0.0006281652,0.00018867722,0.0003839115,0.00030668563,0.00008381049],"domain_scores_gemma":[0.9987224,0.00038923416,0.000053646865,0.0003158548,0.00046280219,0.000056080007],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016145877,0.0005816106,0.0007793947,0.003930866,0.0019011879,0.0012463607,0.00088901795,0.0006980208,0.007196987],"category_scores_gemma":[0.004358898,0.0005531652,0.00064888137,0.0049335826,0.0009063524,0.0020899046,0.0022733654,0.0008984456,0.002336942],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00030285833,0.00016258107,0.0038285558,0.0011981912,0.000054626988,0.0017279857,0.0023948802,0.018394182,0.060350202,0.14539082,0.018193431,0.74800175],"study_design_scores_gemma":[0.00019347275,0.0004918027,0.008402053,0.00037923263,0.0002420905,0.0037640794,0.0020045955,0.38931024,0.14388052,0.1340061,0.31711224,0.00021353494],"about_ca_topic_score_codex":0.0019273657,"about_ca_topic_score_gemma":0.0016588263,"teacher_disagreement_score":0.007196987,"about_ca_system_score_codex":0.00049035513,"about_ca_system_score_gemma":0.0019133367,"threshold_uncertainty_score":0.024076343},"labels":[],"label_agreement":null},{"id":"W2395052721","doi":"10.1080/10489223.2016.1187616","title":"Indirect positive evidence in the acquisition of a subset grammar","year":2016,"lang":"en","type":"article","venue":"Language Acquisition","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Social Sciences and Humanities Research Council of Canada; Fonds de Recherche du Québec-Société et Culture","keywords":"Grammar; Linguistics; Psychology; Phrase structure rules; Language acquisition; Second-language acquisition; Syntax; Indo-European languages; Philosophy","score_opus":0.014630873125853658,"score_gpt":0.2799997069526599,"score_spread":0.26536883382680626,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2395052721","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.91364604,0.000248385,0.047543738,0.000565775,0.000016246659,0.000047219484,0.00006611025,0.00014595661,0.037720583],"genre_scores_gemma":[0.9874051,0.00009998038,0.011140414,0.00007280501,0.0000059904296,0.000023677138,0.000050338393,0.00003450488,0.0011672319],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9983828,0.0005333985,0.00009539089,0.00028543168,0.00062151434,0.00008156375],"domain_scores_gemma":[0.98095036,0.014189681,0.0010853505,0.0020513085,0.001279579,0.00044379054],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002626209,0.0003597025,0.0003210023,0.0008526552,0.0004731335,0.0017135884,0.00071654335,0.0005941596,0.0032477074],"category_scores_gemma":[0.018769521,0.00058739574,0.00033908497,0.00031366848,0.0024232215,0.004010501,0.0021423674,0.0015333958,0.00037753032],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009259338,0.0005148947,0.2140344,0.00091551605,0.00014543317,0.0069363876,0.035662983,0.00359823,0.16922383,0.19257738,0.0010197805,0.37444523],"study_design_scores_gemma":[0.00020619997,0.002358439,0.28103223,0.0007663126,0.00031845117,0.026227696,0.011792236,0.040335763,0.17423192,0.42549708,0.036905043,0.0003286627],"about_ca_topic_score_codex":0.00053286226,"about_ca_topic_score_gemma":0.0009455979,"teacher_disagreement_score":0.0032477074,"about_ca_system_score_codex":0.00030134118,"about_ca_system_score_gemma":0.0005630962,"threshold_uncertainty_score":0.0138888955},"labels":[],"label_agreement":null},{"id":"W2395479133","doi":"10.1093/oxfordhb/9780199276349.013.0023","title":"Sublanguages and Controlled Languages","year":2012,"lang":"en","type":"book","venue":"Oxford University Press eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":35,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"Entomological Society of America","keywords":"Sublanguage; Computer science; Natural language processing; Natural language; Linguistics; Artificial intelligence","score_opus":0.01082141644802223,"score_gpt":0.22050031947987966,"score_spread":0.20967890303185743,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2395479133","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.038406983,0.039102934,0.44047087,0.007654363,0.0014975521,0.00019451023,0.00048144962,0.0017205671,0.47047073],"genre_scores_gemma":[0.7027849,0.018661922,0.1546326,0.0038192149,0.0018950684,0.000858772,0.0010652682,0.0015160468,0.11476615],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99803597,0.0006994612,0.0001536174,0.00053459324,0.0004398143,0.00013656059],"domain_scores_gemma":[0.99735236,0.0015861024,0.00018188101,0.00051637564,0.00025707568,0.0001062401],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013013689,0.00055123575,0.0004417923,0.0015388957,0.0009989119,0.0044142082,0.0009132465,0.00084126124,0.0050929952],"category_scores_gemma":[0.003210196,0.00040323776,0.0007607985,0.00093562953,0.0098751495,0.009940102,0.0026761757,0.00265158,0.0015397129],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000007436989,0.0000033992274,0.000040724863,0.000058533617,0.0000024972949,0.000040680687,0.0008162368,0.00015411117,0.0004107707,0.98517597,0.0012205268,0.012069095],"study_design_scores_gemma":[0.000008379815,0.000020966057,0.00009108004,0.00008026977,0.000005339809,0.00022701506,0.00021983462,0.00089039106,0.0008302005,0.8625565,0.13505732,0.000012637092],"about_ca_topic_score_codex":0.0008852202,"about_ca_topic_score_gemma":0.00048146036,"teacher_disagreement_score":0.0050929952,"about_ca_system_score_codex":0.001524724,"about_ca_system_score_gemma":0.0012058963,"threshold_uncertainty_score":0.017037809},"labels":[],"label_agreement":null},{"id":"W2395734114","doi":"","title":"Studying the Human Translation Process through the TransSearch Log-Files.","year":2005,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Process (computing); Translation (biology); Quality (philosophy); Machine translation system; Natural language processing; Information retrieval; Machine translation; Programming language","score_opus":0.06057285747528785,"score_gpt":0.3607767419651321,"score_spread":0.30020388448984425,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2395734114","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8217047,0.0029646999,0.11971212,0.0019445878,0.00035251465,0.0008731068,0.022183083,0.0050952723,0.025169866],"genre_scores_gemma":[0.8648624,0.0010113892,0.09144081,0.00028718455,0.00013370793,0.00086928083,0.026300687,0.0008552074,0.014239299],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9942673,0.0035698197,0.00032949523,0.00069142034,0.0009776042,0.00016439863],"domain_scores_gemma":[0.9284521,0.05685665,0.003830007,0.0057770023,0.0045423848,0.0005419038],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004386354,0.00071707513,0.00051651866,0.0040835897,0.0013179943,0.001944744,0.0007219469,0.0011349429,0.0061921272],"category_scores_gemma":[0.032188825,0.0003605206,0.00034963188,0.004224238,0.0009687477,0.0038322757,0.0013039631,0.0012030206,0.0039821207],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0050912476,0.0024139334,0.09794421,0.0041500246,0.0005666612,0.0025103758,0.061401322,0.0073301946,0.07440015,0.011123527,0.049727153,0.6833412],"study_design_scores_gemma":[0.00051575963,0.003229235,0.29295847,0.00094544084,0.0005179986,0.007754659,0.051504973,0.07234213,0.19307102,0.021919535,0.35431105,0.0009298471],"about_ca_topic_score_codex":0.0030869613,"about_ca_topic_score_gemma":0.0051401057,"teacher_disagreement_score":0.0061921272,"about_ca_system_score_codex":0.00056480983,"about_ca_system_score_gemma":0.001081125,"threshold_uncertainty_score":0.023197532},"labels":[],"label_agreement":null},{"id":"W2395793795","doi":"","title":"Imperfect Querying through Womb Grammars plus Ontologies.","year":2015,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Parsing; Natural language processing; Artificial intelligence; Rule-based machine translation; Grammar; Imperfect; Mistake; Semantics (computer science); Parsing expression grammar; Programming language; Linguistics; L-attributed grammar; Context-free grammar","score_opus":0.047549743976644805,"score_gpt":0.30945797984230317,"score_spread":0.26190823586565837,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2395793795","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008852805,0.0001228004,0.9787815,0.00076847005,0.00005632892,0.00014439345,0.00053224247,0.006131091,0.0046103587],"genre_scores_gemma":[0.24030063,0.00033330178,0.7461159,0.00087294856,0.00008566015,0.00021727906,0.0017161784,0.0026633418,0.0076947925],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9932787,0.002157782,0.0005952446,0.0013370459,0.00209217,0.0005390539],"domain_scores_gemma":[0.987644,0.0053510237,0.00071476965,0.0048357663,0.0012209652,0.00023350757],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0071505844,0.00094185333,0.00084416603,0.0016685979,0.0012523606,0.0044364585,0.0027309007,0.001614654,0.0044616605],"category_scores_gemma":[0.025351172,0.0013253185,0.0021409222,0.0017465042,0.0039630737,0.012994326,0.006690715,0.0029978813,0.0013163998],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000233507,0.00009962727,0.0043876027,0.00051488884,0.00017790298,0.0012361434,0.0035752184,0.04605974,0.012210228,0.74595904,0.012702266,0.17284378],"study_design_scores_gemma":[0.000044299733,0.00005361531,0.0007058028,0.00017613279,0.00014146649,0.00069917174,0.0006814894,0.25955978,0.024826137,0.60317785,0.109815806,0.000118369106],"about_ca_topic_score_codex":0.007583748,"about_ca_topic_score_gemma":0.009846992,"teacher_disagreement_score":0.007583748,"about_ca_system_score_codex":0.0012780738,"about_ca_system_score_gemma":0.0026949432,"threshold_uncertainty_score":0.037816405},"labels":[],"label_agreement":null},{"id":"W2396074258","doi":"","title":"CLaC Labs: Processing Modality and Negation. Working Notes for QA4MRE Pilot Task at CLEF 2012.","year":2012,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Negation; Clef; Modality (human–computer interaction); Computer science; Heuristics; Natural language processing; Task (project management); Artificial intelligence; Robustness (evolution); Linguistics; Programming language; Engineering","score_opus":0.03902656298056066,"score_gpt":0.2966618935835809,"score_spread":0.25763533060302024,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2396074258","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.027589329,0.006576364,0.42393017,0.01384235,0.0022962533,0.0023257595,0.20801191,0.22056921,0.09485867],"genre_scores_gemma":[0.14943025,0.0009808439,0.43976197,0.0032474457,0.000669709,0.00247209,0.34392172,0.024951503,0.034564447],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99488974,0.0025184918,0.00035011096,0.00070431293,0.0011978596,0.0003394693],"domain_scores_gemma":[0.98881173,0.0058537396,0.0003322571,0.0016057421,0.0027773895,0.0006191018],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0068802936,0.002605659,0.0014805568,0.003114108,0.0021272942,0.0029238167,0.003044542,0.0031955193,0.09604894],"category_scores_gemma":[0.020772986,0.0017011807,0.0015031941,0.0015265225,0.0012139498,0.0062983185,0.003738701,0.0038814647,0.045057364],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005292867,0.0001758751,0.0004504812,0.00091475993,0.00005902764,0.00021143047,0.00034166456,0.001474918,0.0074390103,0.005448455,0.9082991,0.074656144],"study_design_scores_gemma":[0.0018220776,0.00035893006,0.006046003,0.00061040226,0.00016707517,0.0016684821,0.0012255622,0.067142814,0.05529136,0.031751156,0.83350635,0.0004098524],"about_ca_topic_score_codex":0.015706988,"about_ca_topic_score_gemma":0.021548893,"teacher_disagreement_score":0.09604894,"about_ca_system_score_codex":0.0019805564,"about_ca_system_score_gemma":0.0027096006,"threshold_uncertainty_score":0.32131577},"labels":[],"label_agreement":null},{"id":"W2396216261","doi":"","title":"UQAM's System Description for the NTCIR-10 Japanese and English PatentMT Evaluation Tasks.","year":2013,"lang":"en","type":"article","venue":"NTCIR","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Computer science; Natural language processing; Machine translation; Artificial intelligence; Domain (mathematical analysis); Translation (biology); Linguistics","score_opus":0.025567362361244093,"score_gpt":0.2600134725910629,"score_spread":0.2344461102298188,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2396216261","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02233527,0.0066615813,0.42173642,0.002777863,0.0024388751,0.0119020175,0.27467713,0.14991276,0.10755811],"genre_scores_gemma":[0.06121739,0.0016698262,0.35307974,0.0015207957,0.0003545659,0.011241419,0.5047929,0.008637343,0.05748603],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99728465,0.0007473238,0.00039308518,0.00041437018,0.0008966816,0.00026399377],"domain_scores_gemma":[0.9970323,0.00040896563,0.00016341744,0.00054259005,0.0016331214,0.0002196484],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034343386,0.0017615176,0.001267271,0.0034264892,0.0015413491,0.0020041715,0.0020074542,0.0013916828,0.032857966],"category_scores_gemma":[0.0061446377,0.00070017343,0.0006568399,0.0029266067,0.00044555572,0.0021885848,0.0019082024,0.001401537,0.04091974],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004957732,0.00022901452,0.0033018517,0.0020378837,0.00013216112,0.00046640792,0.0005957224,0.0032778785,0.047070775,0.0053750873,0.6989241,0.23809335],"study_design_scores_gemma":[0.000301198,0.0004903073,0.009945541,0.0002708638,0.00013618951,0.0014004274,0.0003687586,0.01950816,0.038460046,0.0029275815,0.9259445,0.00024632673],"about_ca_topic_score_codex":0.03235619,"about_ca_topic_score_gemma":0.03076643,"teacher_disagreement_score":0.032857966,"about_ca_system_score_codex":0.0015440343,"about_ca_system_score_gemma":0.004518042,"threshold_uncertainty_score":0.1099208},"labels":[],"label_agreement":null},{"id":"W2396331625","doi":"","title":"Merging Different Languages in a Single Document Collection.","year":2002,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Search engine indexing; Natural language processing; Information retrieval; Set (abstract data type); Artificial intelligence; Document retrieval; Programming language","score_opus":0.013111845119339868,"score_gpt":0.25457936522130536,"score_spread":0.24146752010196548,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2396331625","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.42820427,0.006805506,0.4944704,0.0018299756,0.00076762855,0.0032582376,0.011216366,0.029567609,0.023879984],"genre_scores_gemma":[0.26655346,0.0006756429,0.69581175,0.00059337093,0.00016848819,0.00081200147,0.022143517,0.002587998,0.01065384],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9915181,0.00329913,0.0007810693,0.0015255342,0.0024529411,0.00042321085],"domain_scores_gemma":[0.9862212,0.0060296054,0.00070543936,0.0030556216,0.003549788,0.00043838503],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008436723,0.0011359364,0.0014629206,0.0048281997,0.002090945,0.0037445556,0.0017643186,0.0011661141,0.0051951567],"category_scores_gemma":[0.020030262,0.0008025591,0.0018079317,0.006080648,0.0011058811,0.0077842474,0.0038133764,0.0014041282,0.003475351],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0029758413,0.0011869064,0.010010117,0.0026106648,0.00080446195,0.0011921326,0.0045471354,0.012145722,0.11343289,0.011324886,0.029553426,0.8102159],"study_design_scores_gemma":[0.0011142518,0.0027939654,0.020295473,0.0005539978,0.0026335635,0.004520455,0.0069018085,0.14423895,0.5105448,0.03587261,0.26973632,0.0007938091],"about_ca_topic_score_codex":0.0032514234,"about_ca_topic_score_gemma":0.0058030905,"teacher_disagreement_score":0.008436723,"about_ca_system_score_codex":0.0015940777,"about_ca_system_score_gemma":0.002059448,"threshold_uncertainty_score":0.04461819},"labels":[],"label_agreement":null},{"id":"W2396347845","doi":"","title":"UB.dmirg: Learning Textual Entailment Relationships Using Lexical Semantic Features.","year":2010,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Textual entailment; Natural language processing; WordNet; Computer science; Artificial intelligence; Logical consequence; FrameNet; Parsing; Semantic role labeling; Sentence","score_opus":0.012768951969041192,"score_gpt":0.2786167734804698,"score_spread":0.2658478215114286,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2396347845","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.041306816,0.005267583,0.7588359,0.0015068089,0.00051308563,0.0017751536,0.046000704,0.11656893,0.028224984],"genre_scores_gemma":[0.17075384,0.0014406986,0.6976109,0.0009359975,0.00023648374,0.0011193336,0.11387932,0.0024288187,0.011594619],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99853766,0.00046638862,0.00010736451,0.00042678337,0.00036346965,0.00009842893],"domain_scores_gemma":[0.99827266,0.00096453377,0.0001322607,0.00030432135,0.00026005568,0.00006626542],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019114572,0.0014713241,0.0009225119,0.0043001594,0.0011254632,0.0017046852,0.0023754272,0.0017097221,0.026272558],"category_scores_gemma":[0.009794572,0.0006164608,0.001236528,0.0020598117,0.0005966099,0.0051343767,0.00270958,0.0015390993,0.010570758],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007622285,0.0004904623,0.003259885,0.0018567366,0.00025909612,0.0006933673,0.00048310854,0.008177694,0.012854158,0.018387021,0.16259407,0.7901822],"study_design_scores_gemma":[0.00055193424,0.0007899562,0.0075305863,0.0005186277,0.00047024034,0.0018973725,0.0009736705,0.54208416,0.055606317,0.12889712,0.26051843,0.00016154883],"about_ca_topic_score_codex":0.005742442,"about_ca_topic_score_gemma":0.009644354,"teacher_disagreement_score":0.026272558,"about_ca_system_score_codex":0.0012217683,"about_ca_system_score_gemma":0.0013735214,"threshold_uncertainty_score":0.087890506},"labels":[],"label_agreement":null},{"id":"W2396584413","doi":"10.63317/3x3m982keb5s","title":"A Compact Arabic Lexical Semantics Language Resource Based on the Theory of Semantic Fields","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Saskatchewan","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Semantics (computer science); Set (abstract data type); Abstraction; Word (group theory); Linguistics; Programming language","score_opus":0.020702214919912133,"score_gpt":0.2616753241121271,"score_spread":0.24097310919221493,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2396584413","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.032538727,0.0012595528,0.8730379,0.0010348883,0.0005200879,0.00069292454,0.017658483,0.040934604,0.032322884],"genre_scores_gemma":[0.21524596,0.0008931089,0.7384132,0.00043409417,0.0002912224,0.0010278822,0.025103303,0.0029717647,0.015619508],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9993129,0.00013708574,0.000118698255,0.0001559209,0.00021229827,0.00006318504],"domain_scores_gemma":[0.99852884,0.0004643884,0.00010446642,0.0003443237,0.0004262557,0.00013171442],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008131454,0.00082080025,0.0016071122,0.005425178,0.0015937763,0.0031415215,0.0013576942,0.0006952835,0.025562825],"category_scores_gemma":[0.0035592269,0.0005255986,0.00084995566,0.0044090142,0.0009006204,0.0075081433,0.002928895,0.0010738648,0.010246576],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011324151,0.0003108448,0.000985065,0.001019532,0.00009613126,0.0011568278,0.0010501863,0.0055255997,0.031423427,0.37653753,0.08866274,0.49209964],"study_design_scores_gemma":[0.00029163752,0.00029842285,0.001320724,0.0003399108,0.00019771686,0.0017245928,0.00095166295,0.091499366,0.033227894,0.44351363,0.42633712,0.00029736123],"about_ca_topic_score_codex":0.0025228614,"about_ca_topic_score_gemma":0.0029430112,"teacher_disagreement_score":0.025562825,"about_ca_system_score_codex":0.0007932977,"about_ca_system_score_gemma":0.002117354,"threshold_uncertainty_score":0.085516155},"labels":[],"label_agreement":null},{"id":"W2396598687","doi":"","title":"Effects of Using Simple Semantic Similarity on Textual Entailment Recognition.","year":2011,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Textual entailment; WordNet; Logical consequence; Natural language processing; Computer science; Artificial intelligence; Semantic similarity; Task (project management); Similarity (geometry); Simple (philosophy); Baseline (sea); Image (mathematics)","score_opus":0.018851084307852834,"score_gpt":0.2685057563310064,"score_spread":0.24965467202315353,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2396598687","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8814238,0.009393794,0.08679167,0.00083647884,0.0009914858,0.0008370845,0.001538696,0.009689138,0.008497831],"genre_scores_gemma":[0.89732605,0.0009940476,0.0956672,0.00033502633,0.0002442173,0.0002068771,0.0029619085,0.0004220864,0.0018425186],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.984937,0.0069048475,0.0019526883,0.0027964488,0.0028361422,0.0005728778],"domain_scores_gemma":[0.88170785,0.098771095,0.003538037,0.0093156155,0.005285816,0.0013816537],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011407025,0.0027765448,0.0022495482,0.002783939,0.0012323135,0.0024599405,0.002287974,0.0028959794,0.0034089976],"category_scores_gemma":[0.09354513,0.00075239217,0.001233325,0.003136204,0.0013955774,0.008833133,0.00388875,0.002467619,0.0015406961],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.01498201,0.0042593684,0.02092097,0.003748017,0.0022482078,0.00061259663,0.00078009546,0.045504604,0.109371305,0.0019099215,0.008684102,0.78697884],"study_design_scores_gemma":[0.0016476606,0.018275833,0.050074242,0.00029390043,0.0024000085,0.002501434,0.001662213,0.6668011,0.233283,0.012999094,0.009465406,0.0005960987],"about_ca_topic_score_codex":0.005375099,"about_ca_topic_score_gemma":0.0075469767,"teacher_disagreement_score":0.011407025,"about_ca_system_score_codex":0.00088933995,"about_ca_system_score_gemma":0.001209759,"threshold_uncertainty_score":0.060326874},"labels":[],"label_agreement":null},{"id":"W2397110197","doi":"","title":"Terminologie et paramètres expérimentaux pour l'évaluation des résumés automatiques","year":2007,"lang":"fr","type":"article","venue":"Trait. Autom. des Langues","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec en Outaouais","funders":"","keywords":"Terminology; Computer science; Humanities; Philosophy; Linguistics","score_opus":0.09946210459983239,"score_gpt":0.3835909856200303,"score_spread":0.2841288810201979,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2397110197","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8189963,0.007742052,0.153052,0.0008719607,0.00060837815,0.003695524,0.002796432,0.00302056,0.00921684],"genre_scores_gemma":[0.8766131,0.0011217649,0.108063936,0.00028641513,0.00034156523,0.006475267,0.0042808084,0.0008991922,0.0019179325],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.8631859,0.10648327,0.010474044,0.006242163,0.0122382175,0.0013764009],"domain_scores_gemma":[0.6000905,0.34436053,0.010687703,0.01988847,0.023151057,0.0018218579],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.05033829,0.0019271574,0.0014434885,0.0028148848,0.0014740756,0.0028489367,0.0014106339,0.0022988357,0.0036868493],"category_scores_gemma":[0.24809289,0.000869485,0.0011482456,0.0024768596,0.0017609424,0.004037332,0.0014936912,0.0019075804,0.0011685659],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.031830393,0.0103063965,0.04915723,0.011151116,0.0029440166,0.0004957384,0.0064877574,0.047869183,0.1816848,0.00960158,0.0136311045,0.63484067],"study_design_scores_gemma":[0.0071359966,0.045149263,0.31204048,0.0018180626,0.004818983,0.002199281,0.007476664,0.13383001,0.40929887,0.021111397,0.05330985,0.0018112517],"about_ca_topic_score_codex":0.0019780376,"about_ca_topic_score_gemma":0.0020713343,"teacher_disagreement_score":0.05033829,"about_ca_system_score_codex":0.0015641222,"about_ca_system_score_gemma":0.0011032893,"threshold_uncertainty_score":0.26621747},"labels":[],"label_agreement":null},{"id":"W2397197605","doi":"10.1111/coin.12094","title":"Improving Shift‐Reduce Phrase‐Structure Parsing with Constituent Boundary Information","year":2016,"lang":"en","type":"article","venue":"Computational Intelligence","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Novelis (Canada)","funders":"National Natural Science Foundation of China; Ministry of Education - Singapore","keywords":"Parsing; Computer science; Parser combinator; Artificial intelligence; Classifier (UML); Top-down parsing; Phrase; Bottom-up parsing; Natural language processing; Boundary (topology); LR parser; Mathematics","score_opus":0.011009476967291787,"score_gpt":0.2611931464031279,"score_spread":0.25018366943583614,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2397197605","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.032228272,0.0007175133,0.9394493,0.00052720914,0.00014164603,0.00018128849,0.0011871911,0.020104863,0.0054626786],"genre_scores_gemma":[0.17987916,0.00054451375,0.79989403,0.00064584956,0.00012966829,0.00025342993,0.0073898206,0.004008116,0.007255393],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99863213,0.00029071618,0.00008551662,0.00043360272,0.00044242342,0.00011565126],"domain_scores_gemma":[0.99617934,0.0016483421,0.00018666516,0.0009766527,0.0009400031,0.00006895832],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016250182,0.0018095537,0.0013357353,0.0022820225,0.0011793952,0.001643591,0.0017091841,0.0014744421,0.0059219],"category_scores_gemma":[0.0063598542,0.00097599276,0.0020127678,0.0026354627,0.0010050485,0.004469861,0.0025594912,0.0025524439,0.00552641],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00031351653,0.0003365258,0.004019264,0.00058094645,0.00017082613,0.0005506768,0.0009496248,0.04556454,0.059206925,0.022483872,0.05486299,0.8109604],"study_design_scores_gemma":[0.000103344944,0.0001918821,0.0038886266,0.00008391859,0.0003100406,0.0007112233,0.00038925983,0.8091412,0.08087538,0.061308194,0.042845078,0.00015192303],"about_ca_topic_score_codex":0.0055918335,"about_ca_topic_score_gemma":0.008839694,"teacher_disagreement_score":0.0059219,"about_ca_system_score_codex":0.0008084552,"about_ca_system_score_gemma":0.0028230948,"threshold_uncertainty_score":0.019810736},"labels":[],"label_agreement":null},{"id":"W2397199043","doi":"","title":"Summarizing with Roget's and with FrameNet.","year":2009,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"FrameNet; Computer science; Pairwise comparison; Natural language processing; Semantic similarity; Thesaurus; Information retrieval; Sentence; Artificial intelligence; Graph; Similarity (geometry); Semantic property; Task (project management); Image (mathematics); Theoretical computer science","score_opus":0.003971854845075968,"score_gpt":0.22995664322202847,"score_spread":0.2259847883769525,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2397199043","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.29277495,0.0025675946,0.32996538,0.0030834493,0.0035138347,0.0032328034,0.054374177,0.24934788,0.061139833],"genre_scores_gemma":[0.397932,0.00037961156,0.45050836,0.0005593824,0.00040349443,0.0017922528,0.11590942,0.009537503,0.022977943],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99684924,0.0012852496,0.0002621918,0.0007992361,0.00054846937,0.00025573347],"domain_scores_gemma":[0.990184,0.0047009178,0.0002900105,0.0020175693,0.0022817522,0.00052578864],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005395959,0.001582938,0.0012660561,0.0026091037,0.001161967,0.002564014,0.0018955789,0.0020837372,0.014872582],"category_scores_gemma":[0.030287873,0.0007184302,0.0012373514,0.0020232892,0.0005581593,0.0055900575,0.002191008,0.0017126977,0.0059636096],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0042137946,0.00094786834,0.005986535,0.0032054416,0.00069286185,0.00067927013,0.006088956,0.019281609,0.015097347,0.011627556,0.33115983,0.60101897],"study_design_scores_gemma":[0.0019994206,0.0031271176,0.020538667,0.00046235925,0.00083037606,0.00078805373,0.004246342,0.39502606,0.046875563,0.03518509,0.4904867,0.00043424193],"about_ca_topic_score_codex":0.010672601,"about_ca_topic_score_gemma":0.017432526,"teacher_disagreement_score":0.014872582,"about_ca_system_score_codex":0.0009825844,"about_ca_system_score_gemma":0.0014514503,"threshold_uncertainty_score":0.049753726},"labels":[],"label_agreement":null},{"id":"W2397269111","doi":"","title":"Generalizing from Freebase and Patterns using Cluster-Based Distant Supervision for TAC KBP Slotfilling 2012.","year":2012,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Sentence; Classifier (UML); Word (group theory); Training set; Set (abstract data type); Natural language processing; Test set; Artificial intelligence; Information retrieval; Context (archaeology); Mathematics","score_opus":0.01750765833137411,"score_gpt":0.28086724392896895,"score_spread":0.2633595855975948,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2397269111","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07819864,0.00066373317,0.77046746,0.00107619,0.00023409812,0.00074669527,0.008318857,0.13267167,0.0076226164],"genre_scores_gemma":[0.29553217,0.000250797,0.6513539,0.00044329706,0.00010996822,0.0005287014,0.039731,0.0038157697,0.008234385],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99833316,0.00041798502,0.000122451,0.00054140587,0.00048938824,0.00009562915],"domain_scores_gemma":[0.9958793,0.0017268164,0.00017246776,0.0013143498,0.00077952415,0.00012754943],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022242852,0.0012015358,0.0009943992,0.0019989156,0.00091881934,0.0013447515,0.0024638202,0.0015285929,0.0053916485],"category_scores_gemma":[0.011455269,0.0007353514,0.0008695766,0.002143955,0.0005112937,0.005274262,0.0026676077,0.0019895043,0.0058118072],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005924228,0.0004923985,0.004279644,0.00090921187,0.00022901647,0.0005705504,0.0017100106,0.02557194,0.03305556,0.0057564145,0.12279105,0.80404174],"study_design_scores_gemma":[0.00018521631,0.00022605348,0.0038116241,0.00009115722,0.00013036118,0.0006492995,0.0008374189,0.84770405,0.043588944,0.02935398,0.07330603,0.00011581192],"about_ca_topic_score_codex":0.012644377,"about_ca_topic_score_gemma":0.021745412,"teacher_disagreement_score":0.012644377,"about_ca_system_score_codex":0.0007684767,"about_ca_system_score_gemma":0.0021369888,"threshold_uncertainty_score":0.025141537},"labels":[],"label_agreement":null},{"id":"W2397430370","doi":"","title":"Using a Distributional Neighbourhood Graph to Enrich Semantic Frames in the Field of the Environment.","year":2015,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Distributional semantics; Computer science; Artificial intelligence; Natural language processing; Neighbourhood (mathematics); Lexical semantics; Exploit; Graph; Semantics (computer science); Semantic similarity; Lexical item; Mathematics; Theoretical computer science","score_opus":0.02159050998028564,"score_gpt":0.29173362413761017,"score_spread":0.2701431141573245,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2397430370","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08769069,0.0008083415,0.8926126,0.00042007596,0.00018314274,0.0003201709,0.0020029293,0.0015167112,0.014445339],"genre_scores_gemma":[0.4473507,0.00039079806,0.54414344,0.00012222529,0.000054106957,0.00029449313,0.0031306376,0.0002542851,0.0042592995],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99911803,0.0003325223,0.000053227617,0.00028014113,0.00017648711,0.000039496867],"domain_scores_gemma":[0.9978315,0.0013107592,0.00022258416,0.0002609726,0.0002985191,0.00007563366],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007575225,0.0004944081,0.00030039175,0.00379071,0.00082768075,0.0010137073,0.00064743654,0.0006918919,0.0032322295],"category_scores_gemma":[0.005645947,0.00025366084,0.0006594829,0.0023970318,0.0010541244,0.0038567577,0.001412756,0.000695644,0.00095359754],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006294957,0.00021540558,0.014130158,0.0012535444,0.00023037921,0.0010402178,0.0074747833,0.01235805,0.059292965,0.22951195,0.0129754655,0.6608876],"study_design_scores_gemma":[0.000112187125,0.00032464555,0.031869598,0.00041465458,0.00030960114,0.002435851,0.0064638183,0.23450622,0.032280117,0.50882936,0.18218654,0.0002674588],"about_ca_topic_score_codex":0.00515122,"about_ca_topic_score_gemma":0.012310737,"teacher_disagreement_score":0.00515122,"about_ca_system_score_codex":0.000814523,"about_ca_system_score_gemma":0.0007121663,"threshold_uncertainty_score":0.010812938},"labels":[],"label_agreement":null},{"id":"W2397481291","doi":"","title":"A Procedural Definition of Multi-word Lexical Units","year":2015,"lang":"en","type":"article","venue":"Recent Advances in Natural Language Processing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; WordNet; Intuition; Artificial intelligence; Natural language processing; Decision tree; Machine learning","score_opus":0.036234424091415286,"score_gpt":0.3289595941006193,"score_spread":0.292725170009204,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2397481291","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0044724387,0.0005816506,0.9712195,0.0018947044,0.00055519486,0.00026778068,0.00053983444,0.000489648,0.019979201],"genre_scores_gemma":[0.14609542,0.00044653934,0.8411563,0.0015084049,0.00037649012,0.0013564993,0.0007008374,0.000602998,0.007756425],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9949673,0.0021247808,0.00070715236,0.001212503,0.0007601455,0.00022814369],"domain_scores_gemma":[0.99357903,0.0025062286,0.0006357891,0.001791612,0.0012529601,0.00023433975],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0049678446,0.0009005097,0.00082938524,0.002809675,0.0027613607,0.0061667617,0.0031491076,0.0021170639,0.010086911],"category_scores_gemma":[0.016169272,0.00081609393,0.001000706,0.003047774,0.010144661,0.012842801,0.0038588706,0.0053085736,0.004493249],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000027979584,0.000015739011,0.00024334798,0.0000934429,0.000009699956,0.000055652858,0.0010986617,0.00032465506,0.001842315,0.9720196,0.002738281,0.021530565],"study_design_scores_gemma":[0.000017327397,0.000055844674,0.0005041983,0.00018715177,0.000019165069,0.0004704044,0.0007104517,0.0062531824,0.0032658533,0.815102,0.17336805,0.00004634921],"about_ca_topic_score_codex":0.0007153494,"about_ca_topic_score_gemma":0.001068117,"teacher_disagreement_score":0.010086911,"about_ca_system_score_codex":0.0010390803,"about_ca_system_score_gemma":0.0015732604,"threshold_uncertainty_score":0.033744097},"labels":[],"label_agreement":null},{"id":"W2397608897","doi":"","title":"Update Summary Update.","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Headline; Computer science; Information retrieval; Semantic similarity; Variety (cybernetics); Task (project management); Natural language processing; Thesaurus; Heuristics; Similarity (geometry); Representation (politics); Artificial intelligence; Linguistics","score_opus":0.012764729480326106,"score_gpt":0.247688647055197,"score_spread":0.23492391757487088,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2397608897","genre_codex":"software","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12400345,0.0025575247,0.05844706,0.0023625293,0.005007195,0.0033044035,0.16047284,0.60174656,0.04209843],"genre_scores_gemma":[0.22780666,0.00062151166,0.1289598,0.0010324656,0.0008249632,0.002012214,0.5584215,0.03694942,0.04337153],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9936178,0.0017260797,0.0007625952,0.0016452165,0.001676637,0.0005717929],"domain_scores_gemma":[0.97252005,0.0095131425,0.0008632301,0.0064442973,0.008792096,0.0018671794],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0065573687,0.003387413,0.002278887,0.0030907968,0.0015244293,0.0033387488,0.002699172,0.002118404,0.0297607],"category_scores_gemma":[0.032003917,0.0014071809,0.0012927837,0.0023579474,0.0004593234,0.0041731345,0.0024838499,0.0026339479,0.028836261],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0031170212,0.00075828825,0.004675102,0.0016268776,0.00034974012,0.0007353936,0.0009864402,0.0034127089,0.011125238,0.0008433805,0.7782072,0.19416262],"study_design_scores_gemma":[0.0023776917,0.002381242,0.028212273,0.00021086838,0.0005596007,0.0011893341,0.0017312684,0.12787941,0.05884987,0.0031018225,0.7730622,0.0004445086],"about_ca_topic_score_codex":0.010667779,"about_ca_topic_score_gemma":0.017526118,"teacher_disagreement_score":0.9702393,"about_ca_system_score_codex":0.001082177,"about_ca_system_score_gemma":0.0017781603,"threshold_uncertainty_score":0.09955943},"labels":[],"label_agreement":null},{"id":"W2398210540","doi":"","title":"AutoSummENG and MeMoG in Evaluating Guided Summaries.","year":2011,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":38,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Preprocessor; Point (geometry); Artificial intelligence; Data mining; Information retrieval; Data science; Machine learning; Mathematics","score_opus":0.029940231717058317,"score_gpt":0.30740245757568846,"score_spread":0.2774622258586301,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2398210540","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.37654665,0.04027855,0.48184112,0.0029463707,0.0020625547,0.0016850407,0.016389439,0.046467885,0.031782378],"genre_scores_gemma":[0.6539653,0.0019266655,0.31527355,0.00062007323,0.00045762555,0.00065029896,0.01582447,0.0011711153,0.010110841],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99398226,0.0034456996,0.00039301245,0.000586777,0.0013965382,0.0001956898],"domain_scores_gemma":[0.985233,0.010243824,0.00078591116,0.0017967148,0.0015347552,0.00040580673],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005267464,0.001344496,0.0011393559,0.005150988,0.0008297944,0.0018106385,0.0013964276,0.0020541912,0.0049462044],"category_scores_gemma":[0.0271874,0.00026510668,0.0005150494,0.0032643264,0.00059162796,0.0034860736,0.0018015149,0.000998468,0.002276008],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0030167792,0.00040207058,0.006369248,0.0019535366,0.00063227466,0.00023184787,0.00045474092,0.022873102,0.010550326,0.0055974303,0.04388236,0.9040363],"study_design_scores_gemma":[0.0010122987,0.004879189,0.022175172,0.0005285527,0.0007911383,0.0018080407,0.0020433462,0.7370158,0.06511996,0.06389555,0.100366555,0.00036441538],"about_ca_topic_score_codex":0.0026588647,"about_ca_topic_score_gemma":0.008126496,"teacher_disagreement_score":0.005267464,"about_ca_system_score_codex":0.0007943166,"about_ca_system_score_gemma":0.0011221049,"threshold_uncertainty_score":0.027857363},"labels":[],"label_agreement":null},{"id":"W2398318312","doi":"","title":"GDUFS at Slot Filling TAC-KBP 2012.","year":2012,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Construct (python library); Task (project management); Computer science; Selection (genetic algorithm); Matching (statistics); Baseline (sea); Information retrieval; Artificial intelligence; Mathematics; Engineering; Geology; Programming language; Statistics","score_opus":0.009141144134649531,"score_gpt":0.26115961596455867,"score_spread":0.2520184718299091,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2398318312","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029492427,0.003957404,0.568578,0.021096438,0.0046192803,0.0023039388,0.091980554,0.20941918,0.06855285],"genre_scores_gemma":[0.086504415,0.0010615,0.6948319,0.0024072188,0.00062496617,0.0012150055,0.14863962,0.016610555,0.048104756],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99357295,0.0022193573,0.0004476891,0.0011938615,0.0020974632,0.00046861832],"domain_scores_gemma":[0.9857913,0.005035719,0.0003576933,0.0034372334,0.0044839494,0.00089404173],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008230806,0.0016587386,0.0016245525,0.004490876,0.0026727035,0.0044647837,0.0031886932,0.002757195,0.03525437],"category_scores_gemma":[0.03032737,0.0012385614,0.0013053804,0.004240967,0.0012128269,0.010690963,0.004673087,0.0044555864,0.03187264],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00059030915,0.00027091676,0.0008246369,0.0008369367,0.00007328487,0.00058387883,0.0013352291,0.0023656972,0.009754924,0.020214764,0.66203207,0.30111724],"study_design_scores_gemma":[0.0002640642,0.0002013978,0.0012408948,0.00017695816,0.00005727059,0.0007706489,0.0012436309,0.04443604,0.019817512,0.023875117,0.90778244,0.00013400806],"about_ca_topic_score_codex":0.020164091,"about_ca_topic_score_gemma":0.018927136,"teacher_disagreement_score":0.03525437,"about_ca_system_score_codex":0.0034352348,"about_ca_system_score_gemma":0.0043375525,"threshold_uncertainty_score":0.117937565},"labels":[],"label_agreement":null},{"id":"W2398392921","doi":"","title":"Multilingual Interface Usage.","year":2011,"lang":"fr","type":"article","venue":"Ingénierie des systèmes d information","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Interface (matter); Computer science; Human–computer interaction; World Wide Web; Operating system","score_opus":0.034231222081750154,"score_gpt":0.27738685533954127,"score_spread":0.2431556332577911,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2398392921","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1384986,0.00940396,0.17943603,0.0028065091,0.0018523057,0.000666319,0.031101687,0.08412343,0.55211115],"genre_scores_gemma":[0.6568307,0.003894559,0.08569357,0.0023713538,0.0005690893,0.0006631705,0.03854693,0.021056307,0.19037424],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99575555,0.0011993825,0.000519873,0.00072745525,0.0014334153,0.00036420958],"domain_scores_gemma":[0.9936935,0.0014420744,0.000309334,0.0018617602,0.0023826153,0.00031077638],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023386835,0.0010452395,0.0006147452,0.0030441931,0.0010320059,0.004631575,0.0015326192,0.0009635754,0.06642199],"category_scores_gemma":[0.013766325,0.00036730003,0.00061701215,0.0030339397,0.0005403625,0.00755076,0.005111389,0.0011185026,0.040204],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00094348314,0.00030608804,0.008876473,0.0016890309,0.000103827944,0.0009770445,0.0052595246,0.0002928148,0.02622401,0.015282763,0.12629898,0.81374586],"study_design_scores_gemma":[0.00007125406,0.00021062446,0.0098066125,0.0006148411,0.00018673217,0.004408559,0.003067615,0.00474524,0.03934186,0.006924028,0.93050504,0.00011760261],"about_ca_topic_score_codex":0.0026742253,"about_ca_topic_score_gemma":0.003685991,"teacher_disagreement_score":0.06642199,"about_ca_system_score_codex":0.00079238584,"about_ca_system_score_gemma":0.0008360883,"threshold_uncertainty_score":0.22220367},"labels":[],"label_agreement":null},{"id":"W2398409858","doi":"","title":"Wikipedia Search as Effective Entity Linking Algorithm.","year":2013,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Task (project management); Entity linking; F1 score; Context (archaeology); Information retrieval; Natural language processing; Artificial intelligence; Knowledge base; Engineering","score_opus":0.004715212997515526,"score_gpt":0.2634789772103846,"score_spread":0.25876376421286906,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2398409858","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06330055,0.007178045,0.8270307,0.0010406248,0.001254244,0.0010285728,0.012747807,0.043216575,0.043202847],"genre_scores_gemma":[0.19469845,0.0011155487,0.7577703,0.00040858853,0.00022617595,0.0005180837,0.024994163,0.0018746953,0.018394029],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99648416,0.0015320552,0.00029255686,0.0006410798,0.0008866915,0.0001635218],"domain_scores_gemma":[0.99595094,0.0020261358,0.00022483365,0.0006665986,0.001024606,0.00010678268],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00331307,0.0010653406,0.0008991914,0.008322112,0.000975396,0.0015823367,0.0012759367,0.0014132897,0.0063609206],"category_scores_gemma":[0.009432939,0.000342873,0.0008982439,0.004853003,0.00029179844,0.0029742634,0.0014170598,0.00065182865,0.005420762],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00051347713,0.00043032988,0.004434812,0.0013479647,0.00050260883,0.00034204283,0.00037274745,0.019945817,0.013041824,0.015510305,0.11532472,0.8282334],"study_design_scores_gemma":[0.0002087048,0.0005454902,0.007993967,0.0003404924,0.00046194793,0.00198632,0.00064445386,0.61843514,0.057515178,0.04952392,0.26217622,0.00016817606],"about_ca_topic_score_codex":0.0027185667,"about_ca_topic_score_gemma":0.0055491617,"teacher_disagreement_score":0.008322112,"about_ca_system_score_codex":0.0005729145,"about_ca_system_score_gemma":0.0010445436,"threshold_uncertainty_score":0.021279395},"labels":[],"label_agreement":null},{"id":"W2398499392","doi":"","title":"Using Payoff-Similarity to Speed Up Search","year":2012,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Nearest neighbor search; Transposition (logic); Pruning; Similarity (geometry); Metric (unit); Speedup; Stochastic game; Neighbourhood (mathematics); Algorithm; Theoretical computer science; State (computer science); Exploit; Mathematics; Data mining; Artificial intelligence; Engineering","score_opus":0.09732299386884227,"score_gpt":0.3771492480848052,"score_spread":0.27982625421596297,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2398499392","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07808181,0.00057171925,0.90848964,0.00033327422,0.00011641652,0.00017978865,0.00020820217,0.004179496,0.007839781],"genre_scores_gemma":[0.44007227,0.0002120055,0.5544342,0.00013176973,0.00006377621,0.00013145982,0.0005178143,0.00037613453,0.0040605804],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988827,0.00028048785,0.00008366041,0.00017408846,0.00044648256,0.0001324447],"domain_scores_gemma":[0.99682236,0.0018819133,0.00021474082,0.00063159916,0.00034543095,0.00010400591],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013835173,0.00075210544,0.0015240953,0.0014996257,0.00058774627,0.0013312477,0.0019460318,0.0013701551,0.007171437],"category_scores_gemma":[0.009233439,0.000396703,0.00073911605,0.0016272801,0.000778672,0.003594166,0.00214177,0.0011729364,0.001782635],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00043940224,0.00028406933,0.0040533263,0.0004035205,0.0001332697,0.0003002806,0.00034422975,0.2657696,0.015751671,0.097400896,0.0100559015,0.60506386],"study_design_scores_gemma":[0.000057586887,0.00009968205,0.00046026582,0.00002156193,0.000033495926,0.00015182486,0.000036999252,0.94990844,0.0052672178,0.041150976,0.0027966204,0.00001539707],"about_ca_topic_score_codex":0.0033437314,"about_ca_topic_score_gemma":0.0074495007,"teacher_disagreement_score":0.007171437,"about_ca_system_score_codex":0.0010546377,"about_ca_system_score_gemma":0.0014234498,"threshold_uncertainty_score":0.02399081},"labels":[],"label_agreement":null},{"id":"W2398518647","doi":"10.63317/5n9i64yqu75i","title":"Evaluating Variants of the Lesk Approach for Disambiguating Words","year":2004,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":94,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"WordNet; Computer science; Word-sense disambiguation; Artificial intelligence; Natural language processing; Precision and recall","score_opus":0.0577276454700606,"score_gpt":0.3548098605602581,"score_spread":0.2970822150901975,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2398518647","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7311136,0.010192372,0.23030941,0.0008153979,0.00071641954,0.0009576623,0.003371291,0.013582359,0.008941487],"genre_scores_gemma":[0.6283516,0.0017764174,0.3588426,0.00029137978,0.00017492686,0.0003447689,0.006037953,0.001034948,0.0031453385],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9773653,0.009873688,0.003255199,0.0034081135,0.0054764757,0.0006211826],"domain_scores_gemma":[0.94400024,0.039128017,0.0018253338,0.006336569,0.0076896264,0.0010202578],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01875475,0.0023963193,0.0023332404,0.007575628,0.00186215,0.0040992126,0.0030355472,0.002379322,0.0019821236],"category_scores_gemma":[0.052200723,0.00081738085,0.0008179156,0.0061598043,0.0019438483,0.008755153,0.0045437454,0.0014570936,0.0020684032],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.007176415,0.0008407765,0.04095194,0.0031705769,0.001907169,0.00054190215,0.0033450236,0.04318712,0.029168976,0.0047916006,0.008270525,0.85664797],"study_design_scores_gemma":[0.0013583872,0.00511739,0.060779784,0.00055342703,0.0016301633,0.005223096,0.010357493,0.72960514,0.13371895,0.014130052,0.036551658,0.0009744613],"about_ca_topic_score_codex":0.0060788067,"about_ca_topic_score_gemma":0.011668728,"teacher_disagreement_score":0.01875475,"about_ca_system_score_codex":0.0015453647,"about_ca_system_score_gemma":0.0016865677,"threshold_uncertainty_score":0.099185765},"labels":[],"label_agreement":null},{"id":"W2398981024","doi":"","title":"SINAI at RTE-7: Integrating Personalized Page Rank Vectors into EDITS.","year":2011,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Rank (graph theory); Computer science; Textual entailment; Similarity (geometry); Information retrieval; Natural language processing; Logical consequence; Artificial intelligence; Term (time); Value (mathematics); Machine learning; Mathematics; Combinatorics","score_opus":0.01040798563329431,"score_gpt":0.25325116611363413,"score_spread":0.24284318048033982,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2398981024","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.043171156,0.0021675031,0.8687486,0.0011687573,0.0007027085,0.0007815285,0.013889983,0.058448166,0.010921588],"genre_scores_gemma":[0.14015174,0.0007439673,0.8014347,0.00029396752,0.00033842298,0.00041519673,0.033118617,0.0022187117,0.02128469],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99740124,0.00085739547,0.00016899928,0.00062204245,0.00083493284,0.00011546311],"domain_scores_gemma":[0.99534273,0.0017527881,0.00036232919,0.001346848,0.0009576745,0.00023759363],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031460896,0.0011226411,0.0009991779,0.0026325702,0.0005752652,0.0022185906,0.00114808,0.0012925275,0.009312669],"category_scores_gemma":[0.017393209,0.00051165523,0.00091406354,0.0017542869,0.00039062285,0.0042212806,0.0015145944,0.0014517896,0.006719628],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00074103125,0.0004627646,0.0042754957,0.000468186,0.00017403673,0.00018900269,0.0002862919,0.014183756,0.011684744,0.013232135,0.07344421,0.88085836],"study_design_scores_gemma":[0.00012017229,0.0007127928,0.007109988,0.00009299124,0.00011640079,0.00077226886,0.00026779855,0.7847987,0.038203873,0.048397083,0.11927054,0.00013745994],"about_ca_topic_score_codex":0.0032582083,"about_ca_topic_score_gemma":0.00835828,"teacher_disagreement_score":0.009312669,"about_ca_system_score_codex":0.000590586,"about_ca_system_score_gemma":0.00090718735,"threshold_uncertainty_score":0.031153977},"labels":[],"label_agreement":null},{"id":"W2399228482","doi":"","title":"Context Derivation Sets and Context-Free Normal Forms.","year":2002,"lang":"en","type":"article","venue":"DCFS","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Terminal and nonterminal symbols; Context-free grammar; Symbol (formal); Context-sensitive grammar; Tree-adjoining grammar; Mathematics; Grammar; Context (archaeology); Mildly context-sensitive grammar formalism; Computer science; Indexed grammar; Rule-based machine translation; Natural language processing; Linguistics; Generative grammar; Artificial intelligence; Phrase structure rules; Programming language","score_opus":0.01591446828257888,"score_gpt":0.23801898765661814,"score_spread":0.22210451937403924,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2399228482","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008100898,0.0011498681,0.9522653,0.00057937286,0.00018972834,0.00016338081,0.00030824158,0.0017338829,0.035509344],"genre_scores_gemma":[0.18095669,0.0017917677,0.7905689,0.00059131725,0.0003133823,0.000533822,0.0015862378,0.0007676315,0.022890335],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99703753,0.00070632267,0.00026704962,0.00062685035,0.001219771,0.0001424793],"domain_scores_gemma":[0.9971349,0.0014638763,0.00015735693,0.0006233358,0.0005167,0.00010371172],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017044442,0.0008572045,0.0004837271,0.0019258811,0.0011819719,0.0030630685,0.0013217202,0.0011905766,0.0079431515],"category_scores_gemma":[0.0070999973,0.0006549887,0.0011132965,0.0017293544,0.003764802,0.0060204766,0.0024791486,0.0033017131,0.0035444987],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000018479699,0.000017599414,0.00013746686,0.00008386944,0.000009630247,0.00013722507,0.000301615,0.0021590271,0.001298973,0.92124176,0.0022098962,0.07238446],"study_design_scores_gemma":[0.000012227328,0.000012614442,0.00007375302,0.0000435595,0.00000998232,0.0002918686,0.000058320747,0.008086578,0.0027155145,0.9518455,0.03683452,0.000015539308],"about_ca_topic_score_codex":0.0013383445,"about_ca_topic_score_gemma":0.0014827633,"teacher_disagreement_score":0.0079431515,"about_ca_system_score_codex":0.0013336291,"about_ca_system_score_gemma":0.0012493761,"threshold_uncertainty_score":0.026572466},"labels":[],"label_agreement":null},{"id":"W2399424026","doi":"","title":"(Almost) Total Recall - SYDNEY CMCRC at TAC 2012.","year":2012,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Recall; Computer science; Psychology; Cognitive psychology","score_opus":0.007795056345747168,"score_gpt":0.25697437273844453,"score_spread":0.24917931639269736,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2399424026","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1071772,0.012000788,0.21229935,0.047726896,0.026641846,0.0013706195,0.15119933,0.07958014,0.36200395],"genre_scores_gemma":[0.33976,0.0017289887,0.11654964,0.003154159,0.0054912595,0.00095453777,0.1328792,0.010452539,0.38902974],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98890674,0.0031176987,0.0005803453,0.002233319,0.0041877893,0.00097413926],"domain_scores_gemma":[0.98367614,0.0022771603,0.00034555246,0.0046101813,0.0076821703,0.0014087653],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01104695,0.0024788089,0.0020405925,0.0045921025,0.0029309555,0.0065313024,0.0028117865,0.0027040257,0.052060414],"category_scores_gemma":[0.026420085,0.000988652,0.0016291513,0.004412233,0.00092776987,0.0048537357,0.0034737892,0.0035233465,0.037756648],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010079854,0.00027813888,0.001487534,0.0002113208,0.00013593466,0.00018559629,0.0002695871,0.004567007,0.0019612852,0.013286764,0.85276455,0.12384426],"study_design_scores_gemma":[0.00068366586,0.0006092607,0.012948045,0.00030923422,0.00028483997,0.0005703184,0.00063747395,0.08672188,0.025081087,0.10380802,0.7680643,0.00028197747],"about_ca_topic_score_codex":0.025452519,"about_ca_topic_score_gemma":0.04556337,"teacher_disagreement_score":0.052060414,"about_ca_system_score_codex":0.0054583712,"about_ca_system_score_gemma":0.004528864,"threshold_uncertainty_score":0.17415941},"labels":[],"label_agreement":null},{"id":"W2400099212","doi":"","title":"ITNLP Entity Linking System at TAC 2013.","year":2013,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Ranking (information retrieval); Cluster analysis; Entity linking; Information retrieval; Set (abstract data type); Knowledge base; Task (project management); Rank (graph theory); Population; Artificial intelligence; Process (computing); Hierarchical clustering; Data mining; Natural language processing; Mathematics; Engineering","score_opus":0.005544463551831476,"score_gpt":0.233324954191289,"score_spread":0.22778049063945754,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2400099212","genre_codex":"software","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019898945,0.0010377758,0.27741787,0.0025835373,0.0017507239,0.0015813237,0.27583894,0.31286025,0.10703061],"genre_scores_gemma":[0.036934815,0.00030201292,0.20032234,0.00069023896,0.00018345895,0.0009281712,0.72218156,0.011351242,0.027106192],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9963768,0.00085992983,0.0003017455,0.0007463569,0.0013951963,0.0003200832],"domain_scores_gemma":[0.9938765,0.0012163103,0.00031305768,0.0016926993,0.0023974506,0.0005040704],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0056442847,0.0019021372,0.0013072486,0.004627741,0.0028196026,0.00425929,0.0036732873,0.0019282803,0.041269083],"category_scores_gemma":[0.010832408,0.0009318583,0.00095892564,0.0056207906,0.0005373097,0.007883014,0.0032793044,0.0033969241,0.041463256],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00045385092,0.00040222946,0.0012069675,0.00049955066,0.000121661185,0.0006990307,0.0005339573,0.003339499,0.005962744,0.0127146505,0.87496346,0.09910233],"study_design_scores_gemma":[0.0002558146,0.00015702611,0.0021885254,0.00011044416,0.0001422764,0.0005665934,0.00040243773,0.043320056,0.022135047,0.014202975,0.91638845,0.00013035686],"about_ca_topic_score_codex":0.018992659,"about_ca_topic_score_gemma":0.019612454,"teacher_disagreement_score":0.041269083,"about_ca_system_score_codex":0.00217081,"about_ca_system_score_gemma":0.00416701,"threshold_uncertainty_score":0.1380589},"labels":[],"label_agreement":null},{"id":"W2400327768","doi":"","title":"A Comparative Study for Query Translation using Linear Combination and Confidence Measure","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Measure (data warehouse); Translation (biology); Cross-language information retrieval; Confidence interval; Machine translation; Artificial intelligence; Linear model; Natural language processing; Data mining; Machine learning; Statistics; Mathematics","score_opus":0.11932504943678805,"score_gpt":0.35821299665932554,"score_spread":0.23888794722253748,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2400327768","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1600762,0.01635446,0.8072543,0.0008882066,0.00022877491,0.0004002413,0.00040843224,0.0055362424,0.0088531105],"genre_scores_gemma":[0.5877868,0.001943266,0.40644968,0.00020074754,0.00027080378,0.00023662561,0.0009942662,0.0005962205,0.0015216285],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9707211,0.015952848,0.0018545325,0.0025730294,0.008375938,0.00052253203],"domain_scores_gemma":[0.89312,0.09070643,0.0026334263,0.004792029,0.008175178,0.00057304883],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.025377193,0.0021013191,0.002309161,0.0069704363,0.0010231375,0.0032572593,0.002340992,0.0022032002,0.0021614209],"category_scores_gemma":[0.08308986,0.0010182108,0.0020575712,0.009253969,0.0012815051,0.008333977,0.002051332,0.0018801544,0.001091844],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.003974073,0.00057194754,0.011625657,0.0012581792,0.0015145777,0.00022400696,0.0005725168,0.061719235,0.010204244,0.00575326,0.0038976017,0.89868474],"study_design_scores_gemma":[0.00023144997,0.0017683182,0.007124991,0.000085235544,0.0007794986,0.0010573138,0.0002859304,0.9597147,0.020414751,0.005115127,0.0032224837,0.00020017319],"about_ca_topic_score_codex":0.0029494015,"about_ca_topic_score_gemma":0.0017828301,"teacher_disagreement_score":0.025377193,"about_ca_system_score_codex":0.0014747861,"about_ca_system_score_gemma":0.0010178025,"threshold_uncertainty_score":0.13420904},"labels":[],"label_agreement":null},{"id":"W2400415464","doi":"","title":"Using SSM for Enhancing Summarization.","year":2011,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Automatic summarization; Computer science; Information retrieval; Ontology; Thesaurus; Task (project management); Process (computing); Multi-document summarization; Natural language processing; Artificial intelligence; Programming language","score_opus":0.023331459692503678,"score_gpt":0.28417965034786075,"score_spread":0.2608481906553571,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2400415464","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017104033,0.0027408022,0.96094537,0.0004006975,0.00029657627,0.00033688996,0.00074328214,0.0120449895,0.0053872815],"genre_scores_gemma":[0.12043914,0.0010290103,0.8648192,0.0003195234,0.00038236362,0.0003187415,0.0032489689,0.00071128615,0.008731706],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9982779,0.0005471835,0.00018789926,0.00032291524,0.00059806765,0.00006601976],"domain_scores_gemma":[0.99580055,0.0015610949,0.00045932407,0.00068935344,0.0014108416,0.0000788933],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023199078,0.0009509755,0.0006674814,0.0029266512,0.0006727582,0.0013341055,0.0010470719,0.0006998339,0.005391173],"category_scores_gemma":[0.0071111196,0.000317828,0.0007192509,0.0021393562,0.00034098234,0.002577537,0.0010864073,0.00075880304,0.004275857],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002162264,0.00008377418,0.0010293797,0.0010006087,0.00014207112,0.00015698778,0.00083285273,0.00397948,0.08090322,0.006019619,0.017068826,0.888567],"study_design_scores_gemma":[0.00018628819,0.0015775823,0.008757867,0.00029029022,0.0008526048,0.0018082346,0.0011791462,0.2660349,0.321576,0.032692228,0.36483446,0.00021036164],"about_ca_topic_score_codex":0.001404952,"about_ca_topic_score_gemma":0.002611045,"teacher_disagreement_score":0.005391173,"about_ca_system_score_codex":0.00042191907,"about_ca_system_score_gemma":0.00050132023,"threshold_uncertainty_score":0.018035293},"labels":[],"label_agreement":null},{"id":"W2400453854","doi":"","title":"Learning Transformation Rules for Semantic Role Labeling","year":2004,"lang":"en","type":"article","venue":"Conference on Computational Natural Language Learning","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Thomson Reuters (Canada)","funders":"","keywords":"Novelty; Brill; Transformation (genetics); Computer science; Semantic role labeling; Natural language processing; Artificial intelligence; Verb; Information retrieval; Philosophy; Chemistry","score_opus":0.011673565048578817,"score_gpt":0.2795745289702728,"score_spread":0.267900963921694,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2400453854","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0072623887,0.00011641333,0.9835988,0.0001901174,0.0000634755,0.0002143594,0.0006207104,0.0065280916,0.0014055211],"genre_scores_gemma":[0.13382544,0.00027348832,0.85475063,0.00025692096,0.000078781006,0.00038937124,0.0065554557,0.0011226438,0.0027472256],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99430084,0.0021513598,0.00044540473,0.0015124881,0.0013820358,0.00020779288],"domain_scores_gemma":[0.9821698,0.012366676,0.0007134,0.0026459985,0.0018757294,0.0002284436],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0045767874,0.0018454811,0.0010065952,0.002932678,0.0010876143,0.0022887853,0.003076636,0.0015324112,0.0057253316],"category_scores_gemma":[0.022319557,0.00076284586,0.002207122,0.0018547527,0.0013474097,0.004509947,0.002246158,0.004137244,0.0037775785],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003107856,0.0005707838,0.002285413,0.0005878366,0.00013248416,0.00038475663,0.000721515,0.055930767,0.012948154,0.030027209,0.0159649,0.88013536],"study_design_scores_gemma":[0.000094664545,0.00014456855,0.00052793405,0.00011703543,0.00009618024,0.00031097248,0.0003236163,0.7975647,0.036435474,0.14153543,0.022783145,0.000066257526],"about_ca_topic_score_codex":0.0029687649,"about_ca_topic_score_gemma":0.0053218543,"teacher_disagreement_score":0.0057253316,"about_ca_system_score_codex":0.0012575424,"about_ca_system_score_gemma":0.0019334761,"threshold_uncertainty_score":0.024204671},"labels":[],"label_agreement":null},{"id":"W2400799260","doi":"","title":"PKUTM participation at TAC 2011 Summarization Track.","year":2011,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Automatic summarization; Track (disk drive); Computer science; Information retrieval; Operating system","score_opus":0.01538329077243138,"score_gpt":0.26339571304893034,"score_spread":0.24801242227649895,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2400799260","genre_codex":"dataset","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021358948,0.00614879,0.2257447,0.03255219,0.025572931,0.0053115794,0.38114467,0.1230427,0.17912352],"genre_scores_gemma":[0.032328296,0.0011555043,0.12043767,0.0033223552,0.0032440068,0.002434307,0.5499439,0.0106704645,0.27646345],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9942403,0.0019670483,0.00031519233,0.0010762342,0.0019845776,0.0004167473],"domain_scores_gemma":[0.9840896,0.0025698647,0.00041015656,0.0025017443,0.008667891,0.0017608217],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0076275053,0.0025353115,0.0022490858,0.0027868263,0.0037228828,0.006553701,0.0046812133,0.0032036037,0.13838772],"category_scores_gemma":[0.018445535,0.0006159554,0.00111024,0.0029691164,0.000572897,0.007679569,0.0038565374,0.0027125296,0.1096397],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020891473,0.000116542564,0.0001245391,0.00031877088,0.00003421365,0.000047841284,0.00009632866,0.00024624908,0.0021126894,0.00078466587,0.95701075,0.038898416],"study_design_scores_gemma":[0.00032177533,0.00032857741,0.001408578,0.00012741901,0.00009377235,0.00015842772,0.0004354812,0.010174632,0.00796538,0.0058904253,0.97302544,0.000070139344],"about_ca_topic_score_codex":0.00946585,"about_ca_topic_score_gemma":0.021512486,"teacher_disagreement_score":0.13838772,"about_ca_system_score_codex":0.0019182238,"about_ca_system_score_gemma":0.00351141,"threshold_uncertainty_score":0.46295303},"labels":[],"label_agreement":null},{"id":"W2400983492","doi":"","title":"The University of British Columbia at TAC 2008","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Automatic summarization; Task (project management); Context (archaeology); Computer science; Columbia university; Work (physics); Library science; Data science; Information retrieval; World Wide Web; Media studies; History; Sociology; Engineering","score_opus":0.006903095408431809,"score_gpt":0.18693848057395462,"score_spread":0.1800353851655228,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2400983492","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010807619,0.017463187,0.01886724,0.028587589,0.0122078005,0.00033691086,0.05120044,0.014380225,0.8461491],"genre_scores_gemma":[0.01912031,0.0036684435,0.009558306,0.001152722,0.00048328025,0.00007931156,0.019895181,0.0015514455,0.94449097],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9983909,0.00017801225,0.00007624727,0.00049027265,0.00064809376,0.00021651936],"domain_scores_gemma":[0.99528885,0.00032262394,0.000111132045,0.000484515,0.002951948,0.0008408916],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0015406719,0.0011673594,0.0011945376,0.0023275919,0.0032717334,0.007088239,0.0012288047,0.0014311983,0.3022874],"category_scores_gemma":[0.0041951044,0.0005017976,0.0004461567,0.0028004115,0.0007033197,0.0024442933,0.0015460625,0.0015405919,0.20066224],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011270796,0.00004029506,0.0005451688,0.00014364599,0.000008792522,0.00011397249,0.00012798554,0.000212375,0.0011932015,0.004266675,0.8876307,0.10560444],"study_design_scores_gemma":[0.000018347077,0.000013981727,0.0014099843,0.00011401844,0.000007178353,0.0000619667,0.00015826934,0.0007256451,0.0007479748,0.0015955288,0.9951233,0.000023806502],"about_ca_topic_score_codex":0.24005306,"about_ca_topic_score_gemma":0.34134096,"teacher_disagreement_score":0.3022874,"about_ca_system_score_codex":0.0073064426,"about_ca_system_score_gemma":0.007861692,"threshold_uncertainty_score":0.99520236},"labels":[],"label_agreement":null},{"id":"W2401226416","doi":"","title":"Text Generation for Abstractive Summarization.","year":2010,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Automatic summarization; Computer science; Sentence; Redundancy (engineering); Natural language processing; Parsing; Artificial intelligence; Focus (optics); Natural language generation; Text generation; Natural language","score_opus":0.01459659402011844,"score_gpt":0.2815096071218646,"score_spread":0.2669130131017462,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2401226416","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.002069834,0.00065481407,0.9829584,0.0003594802,0.00019577524,0.00034478088,0.00051275943,0.009822882,0.0030813871],"genre_scores_gemma":[0.041331083,0.0005455787,0.94792014,0.000240353,0.00022188987,0.00043962823,0.0032764159,0.0010699775,0.0049549625],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9979348,0.0010017263,0.00017459964,0.0003479648,0.00045748136,0.00008336004],"domain_scores_gemma":[0.99602973,0.001992669,0.00035136455,0.00069574564,0.0008323023,0.000098141834],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029568048,0.0012968737,0.00068197167,0.0017257491,0.00068090565,0.0016753537,0.00132415,0.00095329416,0.0137576405],"category_scores_gemma":[0.009170879,0.00043463858,0.001000157,0.0011548019,0.000577751,0.0026441237,0.0012394878,0.001117774,0.007569263],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028178171,0.00016618415,0.0007784376,0.0016584623,0.00013278147,0.0003705014,0.0007093997,0.010923269,0.05376159,0.045542043,0.061503503,0.8241721],"study_design_scores_gemma":[0.00024987748,0.00070819317,0.0019959626,0.00041172656,0.00032877884,0.001513557,0.000515238,0.4110091,0.11882538,0.1413132,0.32294592,0.00018303642],"about_ca_topic_score_codex":0.0007625922,"about_ca_topic_score_gemma":0.0011260905,"teacher_disagreement_score":0.0137576405,"about_ca_system_score_codex":0.0006334631,"about_ca_system_score_gemma":0.0006444416,"threshold_uncertainty_score":0.046023905},"labels":[],"label_agreement":null},{"id":"W2401233896","doi":"10.21437/interspeech.2013-514","title":"Using text and acoustic features to diagnose progressive aphasia and its subtypes","year":2013,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":62,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Aphasia; Computer science; Natural language processing; Speech recognition; Artificial intelligence; Linguistics; Psychology; Cognitive psychology; Philosophy","score_opus":0.014068475382300197,"score_gpt":0.2917986623752805,"score_spread":0.2777301869929803,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2401233896","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9752794,0.0005231602,0.019908365,0.00024347183,0.00010109356,0.00019611875,0.00079855253,0.0013417797,0.0016079851],"genre_scores_gemma":[0.9474588,0.00017376826,0.049304545,0.00008661183,0.00007468073,0.000085251726,0.001759132,0.00006294028,0.0009943343],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9984409,0.00053188426,0.00019549938,0.0004387711,0.00022353658,0.00016935267],"domain_scores_gemma":[0.9888995,0.009385365,0.00035808302,0.0003165978,0.0007570965,0.00028328292],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021540585,0.0011635099,0.0009202654,0.001673044,0.000567409,0.0008327926,0.0006931099,0.0015993668,0.0012990414],"category_scores_gemma":[0.010453666,0.00026248072,0.00064240687,0.0006452722,0.0003367089,0.0014628884,0.0006022605,0.000836088,0.0010817384],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004458254,0.0013737729,0.101308435,0.0006151191,0.00031752975,0.0015742364,0.001269872,0.009231844,0.1038254,0.00020643976,0.002998356,0.7728207],"study_design_scores_gemma":[0.00075937423,0.00498278,0.23598796,0.00020554996,0.0015141721,0.008788073,0.003909177,0.545753,0.18772684,0.0028025291,0.007214093,0.00035640525],"about_ca_topic_score_codex":0.0047765775,"about_ca_topic_score_gemma":0.004632217,"teacher_disagreement_score":0.0047765775,"about_ca_system_score_codex":0.00028538294,"about_ca_system_score_gemma":0.00051675184,"threshold_uncertainty_score":0.011391878},"labels":[],"label_agreement":null},{"id":"W2401870333","doi":"","title":"Using Textual Entailment with Variables for KBP Slot Filling Task.","year":2012,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Logical consequence; Textual entailment; Computer science; Natural language processing; Task (project management); Artificial intelligence; Process (computing); Population; Programming language","score_opus":0.015766613893334346,"score_gpt":0.28171643372207594,"score_spread":0.2659498198287416,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2401870333","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029437719,0.00061969855,0.95316935,0.0007518959,0.00011086013,0.00047173759,0.002203068,0.00886168,0.0043740226],"genre_scores_gemma":[0.23657686,0.0003337408,0.749603,0.0002898142,0.00012439933,0.00033356802,0.008761352,0.0005307383,0.003446626],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99668807,0.0015573932,0.00029480405,0.00073570013,0.0005859055,0.00013826085],"domain_scores_gemma":[0.99428624,0.0037385747,0.0005257386,0.0007471233,0.00057204603,0.00013022861],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037279548,0.00089083216,0.0006724643,0.002051995,0.0010351618,0.0023916524,0.0018332212,0.001273793,0.0071064644],"category_scores_gemma":[0.015887229,0.0004711901,0.0010713342,0.0014712617,0.0008489928,0.006394068,0.0025250572,0.0015693718,0.0032378137],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010413383,0.00040350313,0.0049225963,0.001599983,0.00025118855,0.0006833902,0.0019592335,0.010778232,0.05034495,0.033700086,0.028227983,0.86608756],"study_design_scores_gemma":[0.0002403875,0.00089169986,0.0073831687,0.0003472935,0.00047204533,0.0019433118,0.002767131,0.5602199,0.16459326,0.1433894,0.11748581,0.0002666068],"about_ca_topic_score_codex":0.0024840352,"about_ca_topic_score_gemma":0.004084922,"teacher_disagreement_score":0.0071064644,"about_ca_system_score_codex":0.0007002415,"about_ca_system_score_gemma":0.0014656117,"threshold_uncertainty_score":0.023773432},"labels":[],"label_agreement":null},{"id":"W2402290736","doi":"10.1609/socs.v3i1.18259","title":"A* Variants for Optimal Multi-Agent Pathfinding","year":2021,"lang":"en","type":"article","venue":"Proceedings of the International Symposium on Combinatorial Search","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Israel Science Foundation","keywords":"Pathfinding; Computer science; Theoretical computer science; Shortest path problem","score_opus":0.02463526732612297,"score_gpt":0.2993281981942744,"score_spread":0.27469293086815144,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2402290736","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012448549,0.0004730635,0.9768515,0.0003039456,0.00017538841,0.00015674991,0.00028408386,0.00091791595,0.008388766],"genre_scores_gemma":[0.12803514,0.00046606656,0.86414677,0.0002783706,0.000104604835,0.00062554324,0.0007199378,0.00033083314,0.0052927854],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985701,0.00045993345,0.00015359967,0.00028889976,0.0003919878,0.00013547369],"domain_scores_gemma":[0.9961339,0.0018571357,0.00029526433,0.0009879064,0.0005445107,0.00018143721],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002001838,0.0010936362,0.0008244884,0.0014504623,0.00097195565,0.0015395135,0.0034463692,0.0016972455,0.008762071],"category_scores_gemma":[0.0060301516,0.0006295014,0.0016472349,0.0021770927,0.0011331505,0.0028078612,0.0027632024,0.0024079941,0.0018005617],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004808336,0.00042971416,0.0010216247,0.0005113241,0.00017291185,0.00026356327,0.00021421433,0.2886663,0.008866867,0.33376637,0.020240672,0.3453656],"study_design_scores_gemma":[0.00011934854,0.00032894005,0.00031887437,0.000056887122,0.0000561697,0.00036389756,0.00007859472,0.78941107,0.0047880127,0.180657,0.023773931,0.00004725896],"about_ca_topic_score_codex":0.0021005394,"about_ca_topic_score_gemma":0.0024517926,"teacher_disagreement_score":0.008762071,"about_ca_system_score_codex":0.0007530836,"about_ca_system_score_gemma":0.0012791558,"threshold_uncertainty_score":0.029312015},"labels":[],"label_agreement":null},{"id":"W2402725029","doi":"","title":"Cross-Lingual Cross-Document Coreference with Entity Linking.","year":2011,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":45,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Coreference; Computer science; Entity linking; Natural language processing; Artificial intelligence; Knowledge base; Base (topology); Translation (biology); Resolution (logic); Information retrieval; Mathematics; Chemistry","score_opus":0.014335775863261893,"score_gpt":0.29113051875988594,"score_spread":0.27679474289662404,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2402725029","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0036259664,0.0006535935,0.98971283,0.0002474481,0.00009880811,0.00009215059,0.00014096627,0.000991441,0.004436758],"genre_scores_gemma":[0.08968031,0.00088181923,0.8996462,0.00066485116,0.0001850653,0.00022223344,0.0013354731,0.00048650155,0.006897602],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98953235,0.0054396796,0.000489307,0.002179862,0.0020433427,0.0003154823],"domain_scores_gemma":[0.98163694,0.009114963,0.0011490816,0.0055408874,0.0023556182,0.00020242126],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009653134,0.0007204095,0.0010341903,0.004484728,0.0028346665,0.004680776,0.0041675284,0.0024176638,0.007351747],"category_scores_gemma":[0.026781617,0.00059872924,0.001445245,0.0066569797,0.0019643856,0.00918943,0.008643144,0.0025331574,0.005293156],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014591568,0.00025323394,0.0042410283,0.000981178,0.00043486909,0.0006474027,0.0033267182,0.011616168,0.018272944,0.11025328,0.01903018,0.8307971],"study_design_scores_gemma":[0.000045912293,0.00018543318,0.0045828545,0.00045655924,0.0005388828,0.004982789,0.003456324,0.23618089,0.09949283,0.3633178,0.28653157,0.00022820312],"about_ca_topic_score_codex":0.001292439,"about_ca_topic_score_gemma":0.0025033774,"teacher_disagreement_score":0.009653134,"about_ca_system_score_codex":0.0010229689,"about_ca_system_score_gemma":0.0019315978,"threshold_uncertainty_score":0.05105126},"labels":[],"label_agreement":null},{"id":"W2402772263","doi":"","title":"A Formal Framework for Stringology.","year":2015,"lang":"en","type":"article","venue":"Prague Stringology Conference","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"","keywords":"Computer science; Programming language","score_opus":0.05683039625031779,"score_gpt":0.3251004874468627,"score_spread":0.26827009119654494,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2402772263","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0018519388,0.002998941,0.9539691,0.0033766155,0.00077376456,0.00008375159,0.00044275424,0.00076005375,0.03574301],"genre_scores_gemma":[0.19155073,0.0049309307,0.7655622,0.0033029225,0.0022808919,0.0006397471,0.0021363744,0.0010496435,0.028546587],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99638534,0.0015449501,0.00041387847,0.00069801544,0.00071561383,0.000242203],"domain_scores_gemma":[0.9952974,0.0024523053,0.00031515217,0.0008745807,0.00075578154,0.00030484278],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0047976845,0.0010565429,0.00084305793,0.0041468465,0.002980242,0.0073530483,0.0029633814,0.0023812514,0.014673958],"category_scores_gemma":[0.009527232,0.001053049,0.0026076138,0.003864239,0.010288743,0.018477371,0.0055607837,0.005809077,0.0052565597],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000040505806,0.0000054696325,0.000036094036,0.000036904024,0.000004426061,0.000044037402,0.00015034236,0.00021025725,0.000110139656,0.9934684,0.0016837745,0.0042461082],"study_design_scores_gemma":[0.0000065007102,0.0000058864944,0.000035286477,0.000044458702,0.0000059158992,0.00009277352,0.00009411531,0.0016752112,0.0001802856,0.96570563,0.0321458,0.000008080479],"about_ca_topic_score_codex":0.002768504,"about_ca_topic_score_gemma":0.0024313927,"teacher_disagreement_score":0.014673958,"about_ca_system_score_codex":0.002448746,"about_ca_system_score_gemma":0.0022332466,"threshold_uncertainty_score":0.049089253},"labels":[],"label_agreement":null},{"id":"W2403014101","doi":"","title":"Model Fusion Experiments for the Cross Language Speech Retrieval Task at CLEF 2007.","year":2007,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Ottawa","funders":"","keywords":"Clef; Computer science; Query expansion; Information retrieval; Search engine indexing; Task (project management); Natural language processing; Artificial intelligence; Relevance feedback; Thesaurus; Query language; Image retrieval","score_opus":0.025040452838383322,"score_gpt":0.34771407710168145,"score_spread":0.32267362426329815,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2403014101","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9243261,0.0054484806,0.035085224,0.0017030824,0.00071579806,0.0013083165,0.0086855255,0.014950567,0.0077768904],"genre_scores_gemma":[0.87358385,0.00067639386,0.08417909,0.00065701484,0.00023660497,0.000887039,0.0324913,0.0011211047,0.0061676437],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9924885,0.0043065334,0.0006342872,0.0011579929,0.0010109835,0.00040175198],"domain_scores_gemma":[0.9876847,0.008374544,0.00030005255,0.0013488311,0.0018893123,0.0004025656],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013029526,0.0023384797,0.002328973,0.0016482014,0.0017234292,0.0009859814,0.0017250256,0.0029072438,0.0041752714],"category_scores_gemma":[0.017545752,0.00064505445,0.0017945383,0.0019770316,0.0008441254,0.0028068386,0.0016447745,0.0020536839,0.002908802],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.020609086,0.00867534,0.012043617,0.004436325,0.0038421487,0.00201657,0.003999184,0.19655813,0.11292714,0.0016507951,0.12379506,0.5094467],"study_design_scores_gemma":[0.0036459216,0.009265523,0.038285255,0.0001480219,0.0016023522,0.001420399,0.0024368323,0.8032385,0.114082396,0.002568258,0.022621745,0.0006847163],"about_ca_topic_score_codex":0.03925617,"about_ca_topic_score_gemma":0.03610284,"teacher_disagreement_score":0.03925617,"about_ca_system_score_codex":0.0019098378,"about_ca_system_score_gemma":0.001417489,"threshold_uncertainty_score":0.07805532},"labels":[],"label_agreement":null},{"id":"W2403380339","doi":"","title":"Cross-Language Entity Linking in Maryland during a Hurricane.","year":2011,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Task (project management); Natural language processing; Machine translation; Artificial intelligence; Transliteration; Software; Measure (data warehouse); Information retrieval; Data mining; Programming language","score_opus":0.00806790300983532,"score_gpt":0.2635963014563376,"score_spread":0.25552839844650227,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2403380339","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8353914,0.001573279,0.027142648,0.006984077,0.0028318723,0.0005486695,0.034406763,0.011788469,0.079332806],"genre_scores_gemma":[0.83457047,0.00046291167,0.034697127,0.0013124013,0.00032113274,0.00035205213,0.06482637,0.0010789548,0.062378608],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99876344,0.0004128589,0.00007548154,0.00031282948,0.00025329908,0.00018211872],"domain_scores_gemma":[0.99741036,0.00088536035,0.00020456103,0.0003761409,0.0007998019,0.0003238387],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026055325,0.00044349546,0.00032688733,0.0011278625,0.0030231273,0.0014125811,0.000556816,0.0013505316,0.0066983947],"category_scores_gemma":[0.006086201,0.00024256333,0.00027915344,0.0023708933,0.0003940366,0.0019133021,0.0023855814,0.0009142486,0.0045587765],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0028937585,0.000863973,0.0656947,0.0011005717,0.00025093462,0.007928758,0.019939154,0.010589275,0.032039735,0.0054406966,0.43292734,0.42033118],"study_design_scores_gemma":[0.00014037626,0.00043400502,0.16713193,0.00015227443,0.00011678125,0.0017643225,0.019813301,0.015892515,0.056607146,0.004362261,0.7333888,0.00019631776],"about_ca_topic_score_codex":0.022993188,"about_ca_topic_score_gemma":0.0687058,"teacher_disagreement_score":0.022993188,"about_ca_system_score_codex":0.0014378793,"about_ca_system_score_gemma":0.0017983802,"threshold_uncertainty_score":0.04571867},"labels":[],"label_agreement":null},{"id":"W2403479003","doi":"","title":"Next-Generation Summarization: Contrastive, Focused, and Update Summaries","year":2007,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Automatic summarization; Computer science; Information retrieval; Focus (optics); Set (abstract data type); Context (archaeology); Task (project management); Multi-document summarization; Representation (politics); Natural language processing; Artificial intelligence; Engineering","score_opus":0.023570881410870326,"score_gpt":0.2581904780667015,"score_spread":0.23461959665583115,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2403479003","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.076292515,0.0059817866,0.89262,0.0011918647,0.00038042295,0.0008262737,0.0030483406,0.010188334,0.00947047],"genre_scores_gemma":[0.33931682,0.0010123794,0.64536154,0.0002454788,0.00028143692,0.00040730994,0.005746635,0.00055745855,0.0070709353],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9985715,0.0006115608,0.00012761402,0.00026730288,0.00034515033,0.00007687553],"domain_scores_gemma":[0.9954196,0.0017463084,0.0005020455,0.00095335016,0.0012362236,0.00014241629],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019516988,0.0008439927,0.00061469426,0.0024419622,0.00066310394,0.0015056588,0.0010901765,0.00089918246,0.003450807],"category_scores_gemma":[0.009362522,0.00027712615,0.0004858887,0.0016192495,0.00033293228,0.0023130982,0.0009928542,0.0007867399,0.0015679192],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00078888645,0.00027086312,0.0031545605,0.0007776772,0.00019509744,0.00040710493,0.0012399651,0.02204126,0.03145249,0.010305139,0.028370522,0.9009965],"study_design_scores_gemma":[0.0006349292,0.00215966,0.018684534,0.00038202043,0.0009530653,0.0026476774,0.0012811143,0.60928714,0.12749273,0.043521456,0.19259673,0.0003589537],"about_ca_topic_score_codex":0.002086543,"about_ca_topic_score_gemma":0.0045393514,"teacher_disagreement_score":0.003450807,"about_ca_system_score_codex":0.0005466062,"about_ca_system_score_gemma":0.00066905166,"threshold_uncertainty_score":0.011544108},"labels":[],"label_agreement":null},{"id":"W2403558166","doi":"","title":"Using Ontology Alignment for the TAC RTE Challenge.","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Ontology; Task (project management); Textual entailment; Logical consequence; Natural language processing; Ontology alignment; Artificial intelligence; Fragment (logic); Information retrieval; Upper ontology; Semantic Web; Programming language","score_opus":0.08163474028519087,"score_gpt":0.32948832827929353,"score_spread":0.24785358799410268,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2403558166","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04508735,0.0012835285,0.89812124,0.004582447,0.000954412,0.0011104597,0.0029121956,0.02085749,0.025090845],"genre_scores_gemma":[0.28629506,0.0005605443,0.6816792,0.0015231975,0.0003593959,0.00079042144,0.015432796,0.0013370126,0.012022412],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9905385,0.004444684,0.00062205445,0.001564968,0.0023652567,0.00046451038],"domain_scores_gemma":[0.9883483,0.005104415,0.0011295135,0.0026699107,0.0022135077,0.0005344106],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007015811,0.0012327143,0.0009214542,0.0025471516,0.0022050238,0.0030849034,0.0017402116,0.002904279,0.005908738],"category_scores_gemma":[0.023860259,0.00048500695,0.0010812016,0.0015829109,0.0012783307,0.008650636,0.005503595,0.0031283633,0.0048884875],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007728805,0.0011902654,0.0062498786,0.0012376879,0.0003268786,0.0017177071,0.0035587277,0.009213671,0.064177826,0.031989843,0.11637675,0.7631879],"study_design_scores_gemma":[0.00029719394,0.00063474244,0.009764131,0.00032877724,0.00021832707,0.0051687113,0.005671578,0.3959657,0.10318689,0.11790229,0.3604829,0.0003786833],"about_ca_topic_score_codex":0.0037981237,"about_ca_topic_score_gemma":0.005388783,"teacher_disagreement_score":0.007015811,"about_ca_system_score_codex":0.0010288096,"about_ca_system_score_gemma":0.002651793,"threshold_uncertainty_score":0.037103593},"labels":[],"label_agreement":null},{"id":"W2403748832","doi":"","title":"Experimenting with phrase-based statistical translation within the IWSLT 2004 Chinese-to-English shared translation task.","year":2004,"lang":"en","type":"article","venue":"IWSLT","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Phrase; Computer science; Translation (biology); Natural language processing; Task (project management); Machine translation; Artificial intelligence; Frame (networking); Engineering","score_opus":0.013950822586931685,"score_gpt":0.27638879681530537,"score_spread":0.26243797422837367,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2403748832","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6087358,0.0032422678,0.27400658,0.0036510525,0.0025284614,0.0032207568,0.013265651,0.056052364,0.035297073],"genre_scores_gemma":[0.67731744,0.0004980491,0.25695458,0.00094545475,0.00032360802,0.0010364581,0.045405205,0.0020302136,0.015488882],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99028325,0.005363546,0.00071520894,0.001724414,0.0013771802,0.000536463],"domain_scores_gemma":[0.9897188,0.005013775,0.0003143514,0.002123853,0.002245464,0.0005838419],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01030706,0.0014106923,0.0015798054,0.001206438,0.001905643,0.002189496,0.0018049078,0.0019787853,0.0055595445],"category_scores_gemma":[0.024590028,0.0006142854,0.00082908734,0.0026473526,0.00101377,0.0056689433,0.0031400628,0.0026413337,0.005291594],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.003959778,0.00412813,0.0118795335,0.002849476,0.001448515,0.0016740648,0.0040221433,0.03328361,0.09837772,0.009162374,0.18460941,0.64460534],"study_design_scores_gemma":[0.0028029792,0.00612189,0.02069306,0.00018314247,0.00111271,0.0025801614,0.0049287127,0.59405637,0.24068849,0.016748419,0.10950535,0.0005788355],"about_ca_topic_score_codex":0.013556665,"about_ca_topic_score_gemma":0.015465216,"teacher_disagreement_score":0.013556665,"about_ca_system_score_codex":0.0010338962,"about_ca_system_score_gemma":0.002591423,"threshold_uncertainty_score":0.05450958},"labels":[],"label_agreement":null},{"id":"W2404363504","doi":"","title":"Detecting Textual Entailment with Conditions on Directional Text Relatedness Scores Revisited.","year":2010,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"WordNet; Computer science; Natural language processing; Textual entailment; Artificial intelligence; Context (archaeology); Word (group theory); Logical consequence; Mathematics","score_opus":0.005102462888377609,"score_gpt":0.25479140690947943,"score_spread":0.24968894402110184,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2404363504","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2186906,0.0003686036,0.7569267,0.00042349778,0.000064581356,0.00067402073,0.0015532288,0.0174697,0.003829],"genre_scores_gemma":[0.53511435,0.000087401204,0.45980337,0.00014969741,0.000064778964,0.00026866593,0.0025653145,0.0006277925,0.0013186414],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99431854,0.0016737044,0.0006071371,0.0014714334,0.0015715036,0.000357727],"domain_scores_gemma":[0.98042405,0.009789328,0.0019215759,0.002991496,0.0042086756,0.0006648621],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004829298,0.00094396225,0.0014345974,0.004311035,0.0007490444,0.0025716287,0.0016829626,0.0020246704,0.0044092825],"category_scores_gemma":[0.03230312,0.00050754077,0.00088185654,0.0022763081,0.0015443352,0.005091002,0.0023656006,0.001378375,0.0023111987],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002969569,0.00080762716,0.030083569,0.0012497328,0.00032840975,0.0010807947,0.001135435,0.010569232,0.25725988,0.02871848,0.009809592,0.6559877],"study_design_scores_gemma":[0.0003680674,0.001054037,0.025673395,0.00014499313,0.00032147125,0.0028629424,0.00094750227,0.5755512,0.33678132,0.04703244,0.008948022,0.0003145951],"about_ca_topic_score_codex":0.001858925,"about_ca_topic_score_gemma":0.0026130988,"teacher_disagreement_score":0.004829298,"about_ca_system_score_codex":0.0008330843,"about_ca_system_score_gemma":0.0014397826,"threshold_uncertainty_score":0.025540113},"labels":[],"label_agreement":null},{"id":"W2404416160","doi":"","title":"The MSR Systems for Entity Linking and Temporal Slot Filling at TAC 2013.","year":2013,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Task (project management); Relation (database); Architecture; Natural language processing; Artificial intelligence; Programming language; Data mining; Systems engineering; Engineering; Archaeology","score_opus":0.008702228969991848,"score_gpt":0.24929455051790578,"score_spread":0.24059232154791393,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2404416160","genre_codex":"software","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.21804085,0.005827842,0.25778452,0.0036288046,0.005902097,0.008255852,0.064794175,0.32870898,0.10705686],"genre_scores_gemma":[0.31445923,0.000702415,0.41218945,0.0014066728,0.00068385515,0.0040784907,0.189712,0.01151362,0.065254204],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99025154,0.0030497424,0.000631424,0.002333111,0.0026907285,0.0010434217],"domain_scores_gemma":[0.98671895,0.0028066747,0.00046201856,0.0031622597,0.005458762,0.0013913299],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012244647,0.0025700477,0.0016586068,0.0032354218,0.0024831574,0.0033612587,0.003341398,0.0030506877,0.013723282],"category_scores_gemma":[0.02247222,0.0013191402,0.0017846997,0.001736545,0.0008273421,0.0061704507,0.0041479375,0.003919034,0.022454867],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0023263376,0.0018488207,0.0051834723,0.0016668404,0.0005647181,0.0008643096,0.00316616,0.007254566,0.036205187,0.004255107,0.44138947,0.49527505],"study_design_scores_gemma":[0.0019531944,0.0038118998,0.031710453,0.00042513726,0.0006939834,0.002530302,0.0034090036,0.15418148,0.11672054,0.009228279,0.6745202,0.00081547024],"about_ca_topic_score_codex":0.022062542,"about_ca_topic_score_gemma":0.026625901,"teacher_disagreement_score":0.022062542,"about_ca_system_score_codex":0.0021206532,"about_ca_system_score_gemma":0.0039045417,"threshold_uncertainty_score":0.06475663},"labels":[],"label_agreement":null},{"id":"W2404784473","doi":"","title":"Lexical based two-way RTE System at RTE-5.","year":2009,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Bigram; Computer science; Natural language processing; WordNet; Artificial intelligence; Textual entailment; Task (project management); Identification (biology); Matching (statistics); Logical consequence; Statistics; Mathematics","score_opus":0.006424281169561039,"score_gpt":0.25647355852220927,"score_spread":0.2500492773526482,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2404784473","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04313369,0.0007129348,0.7705725,0.00060658174,0.00034131078,0.0016850545,0.010882255,0.14985396,0.022211725],"genre_scores_gemma":[0.17526688,0.00018172855,0.761092,0.0004153746,0.00010677026,0.00111216,0.025469407,0.0044212462,0.031934407],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9980867,0.00045244448,0.00025139365,0.0006580473,0.00041607485,0.000135338],"domain_scores_gemma":[0.99594206,0.0010076688,0.0002366166,0.001428747,0.001205717,0.00017918562],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026160097,0.00077496516,0.0009684127,0.0016737522,0.00060961046,0.0021385713,0.0015813947,0.001568256,0.051088378],"category_scores_gemma":[0.00648893,0.00067480473,0.000872085,0.0008104657,0.00040516842,0.0044500018,0.002470555,0.00097105553,0.03429633],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019241343,0.00051342376,0.0052293385,0.001068599,0.00021138071,0.0013472033,0.0012319768,0.0016605274,0.19995227,0.0118585,0.07461367,0.7003889],"study_design_scores_gemma":[0.0007309711,0.001840041,0.013386062,0.00016345324,0.0002927421,0.0076023005,0.0012518716,0.12210319,0.42530754,0.018280232,0.40859526,0.00044634216],"about_ca_topic_score_codex":0.0007215958,"about_ca_topic_score_gemma":0.0008410938,"teacher_disagreement_score":0.051088378,"about_ca_system_score_codex":0.00048474394,"about_ca_system_score_gemma":0.0009742141,"threshold_uncertainty_score":0.17090768},"labels":[],"label_agreement":null},{"id":"W2404951684","doi":"","title":"ARPANI@BIT_DURG: KBP English Slot-filling Task Challenge.","year":2013,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Vocabulary; Knowledge base; Task (project management); Entity linking; Natural language processing; Context (archaeology); Information retrieval; Robustness (evolution); Artificial intelligence; World Wide Web; Linguistics","score_opus":0.006654326583959316,"score_gpt":0.2356791206221886,"score_spread":0.22902479403822928,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2404951684","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10285209,0.0034677624,0.13945076,0.035308193,0.010157419,0.0033073733,0.36727858,0.07459811,0.2635798],"genre_scores_gemma":[0.19978711,0.00065557775,0.09699412,0.0034043198,0.000956322,0.0029691474,0.47525603,0.0069398666,0.2130375],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9977881,0.00076324254,0.00013824194,0.0005062195,0.00057076686,0.00023336396],"domain_scores_gemma":[0.994456,0.0022946156,0.00011551521,0.0007333398,0.001627202,0.0007734275],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0046138843,0.001118394,0.0011786378,0.0012183575,0.0024136158,0.003380413,0.0020329857,0.0027491366,0.06794083],"category_scores_gemma":[0.011082675,0.0005396231,0.0005165007,0.0015135312,0.0007074478,0.004936844,0.0031796591,0.0023209895,0.056258406],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00045132608,0.00019583417,0.00048537418,0.00046594813,0.00001948196,0.00035042758,0.0007883617,0.00050554797,0.003608237,0.0035535211,0.9371028,0.052473146],"study_design_scores_gemma":[0.0002669746,0.00023974279,0.0037062394,0.00012118005,0.00001889015,0.0005816981,0.0022322608,0.016694754,0.008881514,0.008242853,0.95891875,0.000095134186],"about_ca_topic_score_codex":0.015414113,"about_ca_topic_score_gemma":0.019438798,"teacher_disagreement_score":0.06794083,"about_ca_system_score_codex":0.0012461833,"about_ca_system_score_gemma":0.001713899,"threshold_uncertainty_score":0.22728473},"labels":[],"label_agreement":null},{"id":"W2405029006","doi":"","title":"Summarization System Evaluation Variations Based on N-Gram Graphs.","year":2010,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Automatic summarization; Computer science; Granularity; Graph; Theoretical computer science; Hierarchy; Data mining; Artificial intelligence; Programming language","score_opus":0.0066088679547011775,"score_gpt":0.26149843591290806,"score_spread":0.2548895679582069,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2405029006","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.34126306,0.011436798,0.5822994,0.001038092,0.0012255544,0.0019383756,0.0074698934,0.026816797,0.026512083],"genre_scores_gemma":[0.7228449,0.0011165501,0.25633472,0.0001910959,0.00025328388,0.00052670715,0.010626802,0.0008484056,0.0072575836],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9953003,0.002046903,0.00035966942,0.0005977327,0.0015371534,0.00015837375],"domain_scores_gemma":[0.9935587,0.0029582032,0.0004778414,0.0006688186,0.0021459984,0.00019046702],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028733162,0.0015652596,0.0009685465,0.0046027987,0.0008831654,0.0018677677,0.00094794505,0.0010950534,0.003744266],"category_scores_gemma":[0.010041584,0.00019680728,0.0006437829,0.0030396234,0.00036988957,0.0024446864,0.0011770165,0.0006848991,0.0021319846],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021254355,0.00047166226,0.008161229,0.0016254737,0.0009391998,0.00026366572,0.00035395613,0.019229634,0.08305384,0.0036832108,0.02155184,0.8585409],"study_design_scores_gemma":[0.00039381796,0.004799511,0.04147065,0.00018094639,0.0011031618,0.0015802358,0.0014534397,0.6574911,0.23876977,0.01294532,0.039497834,0.00031423938],"about_ca_topic_score_codex":0.002358949,"about_ca_topic_score_gemma":0.006153368,"teacher_disagreement_score":0.0046027987,"about_ca_system_score_codex":0.000886265,"about_ca_system_score_gemma":0.00062133954,"threshold_uncertainty_score":0.015195727},"labels":[],"label_agreement":null},{"id":"W2405168890","doi":"","title":"Toward a New Language Engineering.","year":2011,"lang":"en","type":"article","venue":"The Florida AI Research Society","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal; Université du Québec à Trois-Rivières","funders":"","keywords":"Computer science; Chain (unit); Process (computing); Set (abstract data type); Architecture; Perspective (graphical); Programming language; Theoretical computer science; Software engineering; Artificial intelligence","score_opus":0.0906200283212725,"score_gpt":0.3481879933884929,"score_spread":0.25756796506722035,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2405168890","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0044806544,0.006056128,0.8886614,0.03531557,0.0010489114,0.000100956684,0.00020224051,0.00083008513,0.06330406],"genre_scores_gemma":[0.078274585,0.0052408427,0.8608899,0.010405404,0.0019022227,0.00034169416,0.00055752596,0.00062921393,0.041758623],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99563426,0.0018983461,0.00025057798,0.0010027202,0.0010019602,0.00021215998],"domain_scores_gemma":[0.99319893,0.0030476341,0.00027769117,0.0017833449,0.0013286698,0.00036378237],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0055314917,0.0008460855,0.0006569852,0.00296448,0.0017184354,0.006348008,0.002435577,0.002802343,0.008900408],"category_scores_gemma":[0.010617716,0.0008492764,0.0018270706,0.0013967761,0.010119969,0.020197095,0.0065847947,0.0061756084,0.0044931215],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000010902852,0.000034515786,0.00018658632,0.000112335896,0.000011120903,0.00004857888,0.00042956928,0.0006715029,0.0005447576,0.9530026,0.009094779,0.035852727],"study_design_scores_gemma":[0.0000068954337,0.000010361693,0.0000830166,0.000068197915,0.000004455395,0.00012454034,0.00019907243,0.0039096735,0.0004799254,0.9043343,0.09076786,0.000011682274],"about_ca_topic_score_codex":0.0012598525,"about_ca_topic_score_gemma":0.0015604099,"teacher_disagreement_score":0.008900408,"about_ca_system_score_codex":0.0024823288,"about_ca_system_score_gemma":0.0023361705,"threshold_uncertainty_score":0.029774845},"labels":[],"label_agreement":null},{"id":"W2405297130","doi":"","title":"UWaterloo at NTCIR-9: Intent discovery with anchor text","year":2011,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Hyperlink; Information retrieval; Task (project management); Natural language processing; World Wide Web; Graph; Representation (politics); Artificial intelligence; Web page","score_opus":0.017677302626016248,"score_gpt":0.22119518853480838,"score_spread":0.20351788590879213,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2405297130","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08164526,0.011188672,0.31134716,0.016296994,0.01116472,0.010102759,0.27583367,0.07567332,0.20674746],"genre_scores_gemma":[0.13941322,0.002469914,0.25166085,0.0022204625,0.0010177044,0.004213366,0.5277959,0.0052680713,0.065940455],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9899609,0.0050980467,0.00074849074,0.0010312396,0.0024212038,0.00074025616],"domain_scores_gemma":[0.98559225,0.0061804783,0.00037381751,0.0016956375,0.0049936445,0.0011642248],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00890281,0.002232761,0.002770391,0.0046529765,0.0038801723,0.0051224874,0.002713439,0.0029160702,0.032669563],"category_scores_gemma":[0.022679243,0.0011021564,0.0010273713,0.0030859094,0.0018500692,0.0061430945,0.004214905,0.0028756696,0.02439948],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012137303,0.0004778278,0.0018997699,0.003214356,0.00008951001,0.0010020575,0.002444961,0.0036804723,0.027498282,0.011845977,0.83294666,0.11368641],"study_design_scores_gemma":[0.0008693718,0.0007914724,0.0061310944,0.00063860544,0.00018928986,0.0011144617,0.0035870739,0.047286067,0.04710468,0.012815098,0.8789511,0.00052172533],"about_ca_topic_score_codex":0.11659066,"about_ca_topic_score_gemma":0.12787747,"teacher_disagreement_score":0.11659066,"about_ca_system_score_codex":0.0036513244,"about_ca_system_score_gemma":0.007957894,"threshold_uncertainty_score":0.23182404},"labels":[],"label_agreement":null},{"id":"W2405321746","doi":"","title":"ICL Participation at RTE-7.","year":2011,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science","score_opus":0.016670766218261836,"score_gpt":0.2728597701427493,"score_spread":0.25618900392448746,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2405321746","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0024269035,0.0011988024,0.023540562,0.010400289,0.008616219,0.0005196949,0.028624868,0.016096411,0.9085762],"genre_scores_gemma":[0.0152040245,0.00035314745,0.011047062,0.0012206032,0.0008528038,0.00028344835,0.024917861,0.004312501,0.94180846],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99710494,0.00056291115,0.00009790143,0.00061379507,0.0010828823,0.000537602],"domain_scores_gemma":[0.9935929,0.00059354387,0.0001411041,0.0011790472,0.0025097248,0.0019837157],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.00416092,0.0013102364,0.0012640993,0.0017663504,0.0020109678,0.0051618535,0.002869952,0.0028889328,0.6784187],"category_scores_gemma":[0.00799703,0.00034890696,0.00092515873,0.0019933328,0.0006791438,0.0039570234,0.0048186188,0.0019060132,0.59441733],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029351749,0.00011511061,0.00019682635,0.00022808055,0.000012541727,0.000101762096,0.00016774263,0.00016083906,0.00271056,0.022796663,0.8996274,0.073589005],"study_design_scores_gemma":[0.000038353726,0.000047056146,0.00016788798,0.0000583704,0.0000068488066,0.00007746307,0.00010259166,0.00052740576,0.0016275116,0.004051719,0.9932835,0.000011332573],"about_ca_topic_score_codex":0.002281305,"about_ca_topic_score_gemma":0.0032200941,"teacher_disagreement_score":0.6784187,"about_ca_system_score_codex":0.0020434444,"about_ca_system_score_gemma":0.0031663973,"threshold_uncertainty_score":0.45869666},"labels":[],"label_agreement":null},{"id":"W2405616480","doi":"10.63317/2fsyaei3eytq","title":"An Approach to Lexical Development for Inflectional Languages","year":2000,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Lexicon; Natural language processing; Artificial intelligence; Base (topology); Lexical database; Lemmatisation; Mathematics; WordNet","score_opus":0.018531864037999498,"score_gpt":0.31221550238491913,"score_spread":0.29368363834691963,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2405616480","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0007058486,0.00012632126,0.99421895,0.00013717143,0.00006108783,0.00015440096,0.00014427041,0.0017901281,0.002661836],"genre_scores_gemma":[0.007546586,0.000109539964,0.98865384,0.00008952993,0.000041224455,0.00024572352,0.00038559685,0.00038842883,0.0025395134],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9977354,0.00065991597,0.00029194422,0.0005537571,0.0006777821,0.000081097576],"domain_scores_gemma":[0.9963971,0.0016472777,0.00023082962,0.0007396191,0.0008579772,0.00012732024],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001966624,0.0011924192,0.0008914843,0.0056577325,0.0021830706,0.004754304,0.0025918174,0.0015171885,0.016916627],"category_scores_gemma":[0.007725512,0.0013859847,0.0018266801,0.0031605347,0.0024741385,0.0059670447,0.0042409105,0.0030821927,0.009941397],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008217223,0.00016544486,0.0011012908,0.00089669233,0.00009576328,0.0006212282,0.0024750442,0.0033032533,0.02332905,0.2557626,0.016450057,0.69571745],"study_design_scores_gemma":[0.00008917984,0.00013177235,0.0013604573,0.0003684425,0.00012418703,0.0024747786,0.0009769717,0.053982373,0.026991889,0.41602275,0.49730733,0.00016980508],"about_ca_topic_score_codex":0.0010801882,"about_ca_topic_score_gemma":0.002476292,"teacher_disagreement_score":0.016916627,"about_ca_system_score_codex":0.00089792523,"about_ca_system_score_gemma":0.0022689623,"threshold_uncertainty_score":0.05659175},"labels":[],"label_agreement":null},{"id":"W2405740971","doi":"","title":"LIA at TAC KBP 2012 English Entity Linking track.","year":2012,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Track (disk drive); Computer science; Natural language processing; Operating system","score_opus":0.007490476861205684,"score_gpt":0.2546019468647975,"score_spread":0.2471114700035918,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2405740971","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014538774,0.0032651178,0.04798003,0.007187416,0.0038014082,0.00092424365,0.7025508,0.11441801,0.10533415],"genre_scores_gemma":[0.018600581,0.00053724396,0.038157772,0.0008642209,0.0003865374,0.0005474422,0.8880193,0.005358969,0.047527894],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99559706,0.0011520277,0.00039186707,0.00092434033,0.0015512507,0.0003835183],"domain_scores_gemma":[0.9858796,0.0030627719,0.0006689801,0.0034935013,0.005822036,0.0010731616],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005683317,0.0031383215,0.0026012894,0.009261964,0.004551877,0.0075419177,0.0044175265,0.003223181,0.1592972],"category_scores_gemma":[0.02013479,0.001356666,0.0010148769,0.0073291487,0.00073871994,0.015229945,0.004731732,0.00303815,0.13687947],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017210508,0.0001221528,0.00032057718,0.00034117242,0.000042831514,0.00015144581,0.0001041871,0.000249561,0.0010898267,0.0018227028,0.9752162,0.020367213],"study_design_scores_gemma":[0.0004258257,0.00010204605,0.003229708,0.00023515076,0.00011664821,0.0003499228,0.0005110167,0.013691822,0.006332393,0.008463651,0.96642095,0.000120918194],"about_ca_topic_score_codex":0.045980368,"about_ca_topic_score_gemma":0.065299444,"teacher_disagreement_score":0.1592972,"about_ca_system_score_codex":0.0021843517,"about_ca_system_score_gemma":0.0038590059,"threshold_uncertainty_score":0.53290224},"labels":[],"label_agreement":null},{"id":"W2406121106","doi":"","title":"Semantic smoothing and fabrication of phrase pairs for SMT","year":2011,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Phrase; Smoothing; Computer science; Natural language processing; Artificial intelligence; Machine translation; Translation (biology); Speech recognition","score_opus":0.02594067486342347,"score_gpt":0.25905598960058596,"score_spread":0.2331153147371625,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2406121106","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010497061,0.00018971664,0.98648876,0.00009329141,0.000046916357,0.000050881004,0.00008957129,0.0018060581,0.00073770527],"genre_scores_gemma":[0.151296,0.00017856197,0.84523225,0.00011404612,0.00006645435,0.00014374741,0.00069175,0.00086148654,0.0014157046],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9970962,0.0013447292,0.00017183495,0.0005931065,0.00069888355,0.00009525179],"domain_scores_gemma":[0.9939877,0.002597507,0.0004064674,0.0022511587,0.0006665808,0.000090578105],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035471944,0.0013530982,0.00096765364,0.00091668835,0.0014172238,0.0010095215,0.0016634285,0.001516823,0.003961958],"category_scores_gemma":[0.013036022,0.0006850655,0.0011312093,0.0019539115,0.00167832,0.002938498,0.002622481,0.0027903484,0.0042202403],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008483705,0.00016329314,0.0019649635,0.00043998085,0.00014989238,0.00028091797,0.0007105426,0.08884539,0.101049356,0.050313424,0.0059937546,0.74924016],"study_design_scores_gemma":[0.00006778773,0.0003158468,0.0014636781,0.00004156529,0.00007693914,0.000562934,0.0001849854,0.7594293,0.095852755,0.12882395,0.013076992,0.00010331023],"about_ca_topic_score_codex":0.0010115934,"about_ca_topic_score_gemma":0.0021280402,"teacher_disagreement_score":0.003961958,"about_ca_system_score_codex":0.00065952126,"about_ca_system_score_gemma":0.00080687174,"threshold_uncertainty_score":0.018759549},"labels":[],"label_agreement":null},{"id":"W2406241794","doi":"","title":"Singular Referring Expressions in Conjunctive Query Answers: the case for a CFD DL Dialect.","year":2015,"lang":"en","type":"article","venue":"Description Logics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Conjunctive query; Computer science; Context (archaeology); Noun phrase; Expression (computer science); Constant (computer programming); Phrase; Base (topology); Theoretical computer science; Natural language processing; Information retrieval; Programming language; Noun; Mathematics; Relational database","score_opus":0.10156480869624833,"score_gpt":0.31122075541662503,"score_spread":0.20965594672037668,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2406241794","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.028515017,0.00078351557,0.9402848,0.004999277,0.0001957239,0.00031458054,0.00066415366,0.0020803113,0.02216267],"genre_scores_gemma":[0.40914324,0.0007460704,0.57130075,0.0026536512,0.00022045829,0.00043096798,0.0010016388,0.00077126507,0.013731947],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.98842657,0.004783439,0.0013794367,0.001950627,0.002757181,0.0007027747],"domain_scores_gemma":[0.97389835,0.015912125,0.0012143375,0.0045576137,0.003804798,0.00061276776],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011970716,0.00055593334,0.001092369,0.0015586644,0.0024653866,0.0070095914,0.0021387446,0.0025907366,0.0069969245],"category_scores_gemma":[0.030760774,0.0010512302,0.0018776453,0.0026683765,0.0066504274,0.018915612,0.008276426,0.0037043376,0.0017717058],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019253808,0.00003625064,0.001134571,0.0002444158,0.000034857476,0.0008781628,0.0061302,0.0014489066,0.0031279328,0.95250946,0.005876388,0.028386327],"study_design_scores_gemma":[0.00007956515,0.00006677368,0.0007726631,0.00016167117,0.00016184039,0.0025015739,0.0046661757,0.054874495,0.013188926,0.76663965,0.15674184,0.00014483834],"about_ca_topic_score_codex":0.009102272,"about_ca_topic_score_gemma":0.0074520083,"teacher_disagreement_score":0.011970716,"about_ca_system_score_codex":0.0035032837,"about_ca_system_score_gemma":0.001995533,"threshold_uncertainty_score":0.06330794},"labels":[],"label_agreement":null},{"id":"W2406663039","doi":"","title":"Open Information Extraction to KBP Relations in 3 Hours.","year":2013,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Extractor; Computer science; Precision and recall; Set (abstract data type); Extraction (chemistry); Ontology; Open source; Information extraction; Data mining; Information retrieval; Chromatography; Engineering; Programming language; Process engineering; Chemistry","score_opus":0.007067271636542768,"score_gpt":0.2781092129664362,"score_spread":0.27104194132989345,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2406663039","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.28964898,0.0037700906,0.5466806,0.0025877708,0.0011916272,0.0030871977,0.031891715,0.08640699,0.034734935],"genre_scores_gemma":[0.3697834,0.00078774616,0.5574433,0.0008724928,0.0001627019,0.0014121078,0.048078254,0.006012182,0.015447882],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9915092,0.0020812445,0.000880424,0.0020296162,0.0030402986,0.00045919153],"domain_scores_gemma":[0.97501963,0.014277788,0.0009069673,0.00477405,0.004542496,0.00047914073],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0049014944,0.0014643854,0.0010907619,0.002219791,0.001597749,0.0026329746,0.0016440129,0.0010914687,0.008962791],"category_scores_gemma":[0.035891756,0.0009807294,0.0011628978,0.0047732173,0.00069303287,0.0032975979,0.0033561366,0.0017970898,0.0065578935],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011288351,0.00060434063,0.006040093,0.001164706,0.00023316678,0.0009263472,0.0045819655,0.0056685307,0.021177042,0.0035603852,0.06996295,0.88495165],"study_design_scores_gemma":[0.0007868601,0.001476469,0.035995927,0.0005649684,0.0005643723,0.0020164915,0.006342665,0.13104135,0.1596821,0.0419483,0.61923766,0.00034290378],"about_ca_topic_score_codex":0.008129802,"about_ca_topic_score_gemma":0.011458877,"teacher_disagreement_score":0.008962791,"about_ca_system_score_codex":0.0009045502,"about_ca_system_score_gemma":0.0021171586,"threshold_uncertainty_score":0.02998346},"labels":[],"label_agreement":null},{"id":"W2407187509","doi":"","title":"English Slot Filling with the Knowledge Resolver System.","year":2013,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Resolver; Search engine indexing; Inference; Computer science; Task (project management); Interpretation (philosophy); Artificial intelligence; Selection (genetic algorithm); Natural language processing; Information retrieval; Programming language; Engineering; Telecommunications","score_opus":0.0056104343999691915,"score_gpt":0.22712410668090366,"score_spread":0.22151367228093446,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2407187509","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0241275,0.0013559201,0.66575193,0.0014772471,0.0006866241,0.0005675055,0.023062991,0.20931102,0.07365931],"genre_scores_gemma":[0.14526436,0.00067770283,0.7543085,0.0009056428,0.00016122095,0.00027617725,0.05396789,0.0080506615,0.036387768],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99769825,0.00042591634,0.00030635545,0.0005716982,0.0008487116,0.00014898129],"domain_scores_gemma":[0.996606,0.0013541222,0.0002204695,0.00087246206,0.00083984993,0.00010715303],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002153659,0.00089640025,0.00083835993,0.0029042638,0.0011295505,0.0029867226,0.0020696733,0.0012020459,0.03716351],"category_scores_gemma":[0.010744824,0.0007896906,0.001120657,0.002287109,0.0005695861,0.006460315,0.0031510901,0.0017498196,0.028651979],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00042453894,0.00021886721,0.0030918533,0.0017800102,0.00014488814,0.0011034736,0.001621868,0.0028141579,0.018627236,0.036355674,0.20721756,0.72659993],"study_design_scores_gemma":[0.00012549269,0.000134213,0.0033782567,0.00035901956,0.00017018456,0.0027071133,0.001712574,0.06564714,0.06057957,0.04130253,0.82372135,0.00016255688],"about_ca_topic_score_codex":0.0039985706,"about_ca_topic_score_gemma":0.005349106,"teacher_disagreement_score":0.03716351,"about_ca_system_score_codex":0.0007638708,"about_ca_system_score_gemma":0.0019638566,"threshold_uncertainty_score":0.12432438},"labels":[],"label_agreement":null},{"id":"W2407195633","doi":"","title":"A Statistical Model for Measuring Structural Similarity between Webpages","year":2015,"lang":"en","type":"article","venue":"Recent Advances in Natural Language Processing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Similarity (geometry); Computer science; Web page; Statistical model; Task (project management); Measure (data warehouse); Artificial intelligence; Structural similarity; Information retrieval; Natural language processing; Data mining; Image (mathematics); World Wide Web","score_opus":0.039281710599538375,"score_gpt":0.3505640517397068,"score_spread":0.3112823411401684,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2407195633","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.049894128,0.00018720141,0.94625837,0.00023617558,0.000033867535,0.00011482223,0.0006827622,0.00103252,0.0015601994],"genre_scores_gemma":[0.75218236,0.00046265114,0.2375225,0.0003341252,0.00020020257,0.0009045416,0.0032999814,0.0003433567,0.004750206],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99772996,0.0006042411,0.00014588468,0.0006109598,0.000779711,0.00012918786],"domain_scores_gemma":[0.9927382,0.0038325107,0.0011188142,0.0009972011,0.0011360688,0.0001772135],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002785242,0.00079248054,0.0009879731,0.0044171326,0.00053166447,0.0013819173,0.0018104782,0.0015360424,0.0019057818],"category_scores_gemma":[0.014038479,0.00060548796,0.0013734272,0.002715086,0.0012698361,0.0037568714,0.00090857106,0.0015922383,0.002039571],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005636811,0.0008825797,0.04529995,0.00044389095,0.00065853127,0.0008334274,0.00072935736,0.5388482,0.045206774,0.12420349,0.007738075,0.23459198],"study_design_scores_gemma":[0.000010401392,0.00009926474,0.0056829634,0.000013793647,0.00004124791,0.00025432114,0.000033302458,0.95966125,0.0019512137,0.03100012,0.0012160118,0.000036057405],"about_ca_topic_score_codex":0.0035299337,"about_ca_topic_score_gemma":0.0031017074,"teacher_disagreement_score":0.0044171326,"about_ca_system_score_codex":0.0010925423,"about_ca_system_score_gemma":0.0012228696,"threshold_uncertainty_score":0.014729917},"labels":[],"label_agreement":null},{"id":"W2407223942","doi":"","title":"Discourse Relation Recognition by Comparing Various Units of Sentence Expression with Recursive Neural Network","year":2015,"lang":"en","type":"article","venue":"Institutional Repositories DataBase (IRDB)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Atomic Energy of Canada Limited; Institute for Catastrophic Loss Reduction; University of Pennsylvania","keywords":"Sentence; Computer science; Relation (database); Meaning (existential); Natural language processing; Feature (linguistics); Artificial intelligence; Recurrent neural network; Expression (computer science); Word (group theory); Artificial neural network; Speech recognition; Linguistics; Psychology; Data mining","score_opus":0.033816294109018374,"score_gpt":0.2725006151502351,"score_spread":0.23868432104121673,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2407223942","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13084847,0.00059392257,0.85980666,0.00022854864,0.00009565726,0.00020729173,0.00038961798,0.0038638972,0.0039660446],"genre_scores_gemma":[0.5537011,0.0002623661,0.44227934,0.00008853873,0.000055493692,0.00024232999,0.00095306145,0.0001796383,0.0022380902],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9984397,0.00049634295,0.00015040058,0.00053066283,0.00029643194,0.000086299944],"domain_scores_gemma":[0.99857605,0.0006307475,0.00022643375,0.0001480327,0.000377098,0.000041684914],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012166249,0.0009483769,0.0005484988,0.0016498553,0.00038053098,0.0010481791,0.0011238015,0.0005430021,0.001947923],"category_scores_gemma":[0.0049607386,0.00025216292,0.00051320746,0.0011480903,0.0005131828,0.0022286929,0.00074381207,0.0007909897,0.0006733718],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00042460285,0.0001738407,0.0052594445,0.0003049337,0.00012044004,0.0001918729,0.001286957,0.012030522,0.10956006,0.0074933455,0.0022828667,0.8608712],"study_design_scores_gemma":[0.00004672716,0.00029526194,0.011185535,0.000066519366,0.00017657819,0.00021142975,0.00053945265,0.8820424,0.08515041,0.01368932,0.0065022586,0.00009417718],"about_ca_topic_score_codex":0.003098019,"about_ca_topic_score_gemma":0.0035110912,"teacher_disagreement_score":0.003098019,"about_ca_system_score_codex":0.00059647515,"about_ca_system_score_gemma":0.00060487824,"threshold_uncertainty_score":0.006516397},"labels":[],"label_agreement":null},{"id":"W2407423171","doi":"","title":"Domination in lexicographic product digraphs.","year":2011,"lang":"en","type":"article","venue":"Ars Combinatoria","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Lexicographical order; Mathematics; Product (mathematics); Combinatorics; Geometry","score_opus":0.01552132551322981,"score_gpt":0.2427508621659651,"score_spread":0.22722953665273526,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2407423171","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.21098687,0.003568878,0.5325556,0.002968115,0.00031059197,0.00025850374,0.0032752613,0.0015857497,0.2444904],"genre_scores_gemma":[0.7607668,0.0025280763,0.1468891,0.00083096506,0.00033154408,0.00027435005,0.0035830997,0.00048173362,0.08431435],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99895096,0.0003285958,0.00011080536,0.00025987774,0.00023204979,0.00011766791],"domain_scores_gemma":[0.99601495,0.0028069625,0.00026404692,0.00036054503,0.00036700035,0.0001865036],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00086380244,0.00036882618,0.00072384346,0.002365396,0.0017789818,0.003985373,0.00071300636,0.00058150745,0.014776478],"category_scores_gemma":[0.0050873086,0.0005709782,0.00072665757,0.0032593498,0.001837304,0.0062925075,0.0016452395,0.0011066272,0.002257235],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000778544,0.000043633056,0.0009605189,0.00023328546,0.000034141147,0.00036025402,0.00065020786,0.0016914066,0.0014334474,0.9403931,0.0065933517,0.047528826],"study_design_scores_gemma":[0.000025846342,0.000021605192,0.00046322885,0.00005833313,0.000033617973,0.00057080045,0.00023659278,0.009006428,0.0016310314,0.95810556,0.02982601,0.000020885962],"about_ca_topic_score_codex":0.0024788857,"about_ca_topic_score_gemma":0.0051663974,"teacher_disagreement_score":0.014776478,"about_ca_system_score_codex":0.0015785919,"about_ca_system_score_gemma":0.0009065588,"threshold_uncertainty_score":0.049432218},"labels":[],"label_agreement":null},{"id":"W2407550762","doi":"10.31234/osf.io/at6y2_v1","title":"The Memory Tesseract: Distributed MINERVA and the Unification of Memory","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University; Carleton University","funders":"","keywords":"Computer science; Associative property; Unification; Judgement; Memory model; Content-addressable memory; Cognitive science; Artificial intelligence; Psychology; Philosophy; Mathematics; Programming language; Artificial neural network; Epistemology; Pure mathematics","score_opus":0.0056610012227606005,"score_gpt":0.2511806737864159,"score_spread":0.2455196725636553,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2407550762","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12370379,0.00067864504,0.84784347,0.0016084061,0.00020197677,0.00004127714,0.00013512083,0.001033254,0.024754036],"genre_scores_gemma":[0.90559804,0.00033775577,0.07718525,0.00033726843,0.00011017589,0.00009772396,0.00010504968,0.00016503499,0.016063547],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9996275,0.00007760066,0.0000216325,0.00010467675,0.00009201061,0.00007657569],"domain_scores_gemma":[0.99877113,0.00035737373,0.000113545844,0.00050918106,0.00015611215,0.000092560804],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00078096933,0.00040016221,0.00060877786,0.00038304066,0.000750805,0.0017685267,0.0022055274,0.0012055724,0.005859336],"category_scores_gemma":[0.0025225948,0.0002963869,0.00087499316,0.00031347925,0.0017901591,0.0047172788,0.0025767842,0.0018107073,0.00072221225],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019483201,0.000057477442,0.00055202167,0.00007407921,0.000038256298,0.00017232345,0.0001784275,0.061109424,0.007574928,0.8940706,0.002040946,0.033936717],"study_design_scores_gemma":[0.000030904037,0.000041866788,0.00012150946,0.000011289569,0.00001281778,0.0001320958,0.000021158337,0.4420075,0.0053074243,0.5489815,0.003316277,0.000015613443],"about_ca_topic_score_codex":0.0008580442,"about_ca_topic_score_gemma":0.0010230868,"teacher_disagreement_score":0.005859336,"about_ca_system_score_codex":0.000769514,"about_ca_system_score_gemma":0.0009315091,"threshold_uncertainty_score":0.019601464},"labels":[],"label_agreement":null},{"id":"W2407557870","doi":"10.63317/2ccdwd2a9gnc","title":"Detection of Domain Specific Terminology Using Corpora Comparison","year":2004,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":59,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Terminology; Computer science; Term (time); Identification (biology); Focus (optics); Natural language processing; Task (project management); Domain (mathematical analysis); Artificial intelligence; Field (mathematics); Lexicography; Selection (genetic algorithm); Information retrieval; Linguistics","score_opus":0.034025952916465395,"score_gpt":0.2916037686252727,"score_spread":0.25757781570880733,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2407557870","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.46977141,0.007643201,0.4699016,0.0008963371,0.00077154953,0.0021684358,0.008463549,0.00830904,0.03207493],"genre_scores_gemma":[0.4543787,0.0020381955,0.5186516,0.000211792,0.00033172523,0.0014209466,0.016307056,0.0013500857,0.0053099534],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.992392,0.0025076666,0.0012134452,0.0018498612,0.0017520299,0.0002849783],"domain_scores_gemma":[0.9762439,0.011417115,0.0023886324,0.0026900163,0.006800422,0.00045988662],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0052426546,0.0009584689,0.0011880939,0.019334052,0.0016245997,0.0035433536,0.0012541452,0.0012540347,0.004282401],"category_scores_gemma":[0.02966842,0.00043799548,0.00091729866,0.017165693,0.00085690565,0.004109313,0.0022346473,0.0009850824,0.0023759943],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00092759985,0.0002612061,0.025932247,0.0023558927,0.00035031931,0.002271064,0.005382216,0.0039871284,0.110284485,0.014685435,0.015792297,0.81777],"study_design_scores_gemma":[0.00057684863,0.0017014779,0.16389696,0.0011256967,0.002065396,0.0130106155,0.013592347,0.14317282,0.26225552,0.047223233,0.35074955,0.0006295039],"about_ca_topic_score_codex":0.0015757574,"about_ca_topic_score_gemma":0.0018854807,"teacher_disagreement_score":0.019334052,"about_ca_system_score_codex":0.00086898473,"about_ca_system_score_gemma":0.0015879922,"threshold_uncertainty_score":0.027726173},"labels":[],"label_agreement":null},{"id":"W2407606031","doi":"","title":"Experimenting with Clause Segmentation for Text Summarization.","year":2008,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Automatic summarization; Computer science; Segmentation; Sentence; Natural language processing; Artificial intelligence; Dependent clause; Selection (genetic algorithm); Heuristic; Baseline (sea)","score_opus":0.010051519858476217,"score_gpt":0.2657650424146443,"score_spread":0.2557135225561681,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2407606031","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.591708,0.0040201903,0.33018827,0.0014713275,0.0010317095,0.0035279933,0.009923428,0.04812807,0.010000958],"genre_scores_gemma":[0.38420948,0.00055007206,0.5771877,0.00054403796,0.00027312207,0.0010375907,0.027959099,0.0017300278,0.0065088705],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9945326,0.0027278804,0.0004899061,0.0012211667,0.0008195665,0.00020885625],"domain_scores_gemma":[0.9740292,0.018915081,0.00092527847,0.0017922064,0.003926256,0.00041202118],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006470132,0.0015101776,0.0011412938,0.0015617515,0.0011857406,0.0015963248,0.0015966241,0.0016161205,0.0046186084],"category_scores_gemma":[0.027591193,0.0004923661,0.0007497268,0.0026536894,0.0004756781,0.0029944954,0.0008219155,0.0017751905,0.0026909718],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0036658833,0.0024008066,0.005764124,0.0041840514,0.0011565593,0.0006688716,0.003438044,0.040191486,0.19473916,0.0014385389,0.038253035,0.7040994],"study_design_scores_gemma":[0.0024232396,0.014038324,0.01754612,0.00015418076,0.0014579169,0.0016705346,0.0033604454,0.39099076,0.49632826,0.004404775,0.06724915,0.0003762449],"about_ca_topic_score_codex":0.0037833252,"about_ca_topic_score_gemma":0.007115012,"teacher_disagreement_score":0.006470132,"about_ca_system_score_codex":0.00060192886,"about_ca_system_score_gemma":0.0007614488,"threshold_uncertainty_score":0.034217715},"labels":[],"label_agreement":null},{"id":"W2407683577","doi":"","title":"University of Washington at TAC 2011.","year":2011,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Automatic summarization; Task (project management); Computer science; Ranking (information retrieval); Selection (genetic algorithm); Sentence; Track (disk drive); Natural language processing; Artificial intelligence; Information retrieval; Engineering; Systems engineering; Operating system","score_opus":0.009497009510668158,"score_gpt":0.2180927104726006,"score_spread":0.20859570096193245,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2407683577","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0046303147,0.0053412896,0.054939017,0.008841515,0.0067984406,0.0004235581,0.07723528,0.075500965,0.76628965],"genre_scores_gemma":[0.023526823,0.0049272478,0.035298787,0.0019169749,0.0009598115,0.00038614337,0.10997112,0.01027049,0.81274253],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99900407,0.00019546007,0.00008530334,0.00030851646,0.00029143193,0.000115278235],"domain_scores_gemma":[0.9979456,0.00025734835,0.00013471588,0.00056690787,0.00073496177,0.00036050673],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0017649083,0.0018852898,0.0014736432,0.0024534937,0.001707261,0.008058652,0.0016836101,0.0020556864,0.44639018],"category_scores_gemma":[0.00363986,0.0009957985,0.0006359594,0.0035260979,0.00038866053,0.0044629783,0.002051497,0.0017543447,0.44533885],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025559514,0.00010987674,0.00032491278,0.00027250676,0.000025579495,0.00017172532,0.00012296707,0.00018941054,0.002029722,0.0067324103,0.8667962,0.122969165],"study_design_scores_gemma":[0.00007250387,0.000032495296,0.0005775619,0.00013614561,0.000025942518,0.00015519428,0.00008483209,0.0010770626,0.0014359958,0.0047980235,0.9915769,0.000027320499],"about_ca_topic_score_codex":0.008396851,"about_ca_topic_score_gemma":0.009917798,"teacher_disagreement_score":0.44639018,"about_ca_system_score_codex":0.0011119245,"about_ca_system_score_gemma":0.0013410674,"threshold_uncertainty_score":0.78965724},"labels":[],"label_agreement":null},{"id":"W2407749301","doi":"","title":"The University of Sheffield System at TAC KBP 2010.","year":2010,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Task (project management); Node (physics); Information retrieval; Economic shortage; Relation (database); Perspective (graphical); Data mining; Artificial intelligence; Linguistics","score_opus":0.0037283611417814216,"score_gpt":0.2088595354385276,"score_spread":0.20513117429674615,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2407749301","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012833037,0.0047122594,0.070004426,0.004299872,0.0008036797,0.0015083012,0.36378607,0.19404581,0.3480065],"genre_scores_gemma":[0.100466266,0.0020897044,0.07877116,0.0012945075,0.0002263714,0.001492297,0.60672504,0.018037587,0.1908971],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99679846,0.0004966888,0.00038225105,0.00083753356,0.001177689,0.00030741168],"domain_scores_gemma":[0.9940807,0.001468732,0.0002744488,0.0013507978,0.0023744327,0.0004508701],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030358299,0.0019870242,0.0014853326,0.0053101396,0.002706076,0.005050096,0.0027670353,0.0023540938,0.19748718],"category_scores_gemma":[0.017002596,0.0014217395,0.0005587952,0.00557365,0.0009815791,0.010718545,0.003910514,0.0016537745,0.12501445],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003519656,0.000046598685,0.00079390156,0.0011001156,0.000051534018,0.00048923754,0.00086154195,0.0005561355,0.004967482,0.0073044165,0.9178807,0.06559634],"study_design_scores_gemma":[0.00020149603,0.000047369907,0.0025338656,0.00032460768,0.000047532416,0.00041583774,0.0006266018,0.0058739944,0.0054072035,0.004362964,0.98002726,0.00013129703],"about_ca_topic_score_codex":0.19922401,"about_ca_topic_score_gemma":0.22934999,"teacher_disagreement_score":0.19922401,"about_ca_system_score_codex":0.0050946916,"about_ca_system_score_gemma":0.0053225937,"threshold_uncertainty_score":0.6606604},"labels":[],"label_agreement":null},{"id":"W2407793169","doi":"","title":"Training a Perceptron with Global and Local Features for Chinese Word Segmentation","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Conditional random field; Segmentation; Artificial intelligence; Word (group theory); Computer science; Character (mathematics); Text segmentation; Natural language processing; Perceptron; Pattern recognition (psychology); Speech recognition; Artificial neural network; Mathematics","score_opus":0.015065324310744838,"score_gpt":0.290412212100892,"score_spread":0.27534688779014715,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2407793169","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15158576,0.00092111423,0.8291192,0.00036192092,0.00023900373,0.00013245053,0.00038307338,0.014081925,0.0031754554],"genre_scores_gemma":[0.6710196,0.0002474257,0.32040253,0.00037467622,0.00012756369,0.00020864166,0.0014580463,0.00045488222,0.0057066143],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9982116,0.0005762361,0.00013786685,0.0006745875,0.00018809692,0.00021153489],"domain_scores_gemma":[0.9973979,0.0016642204,0.00013211544,0.00021897208,0.00044767858,0.00013920318],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037739717,0.00133123,0.0016748608,0.0015050948,0.0009031111,0.0010421934,0.0013680389,0.0015904784,0.003935998],"category_scores_gemma":[0.005022627,0.0009864842,0.0012814959,0.0018290334,0.0006935347,0.0033830197,0.0010821811,0.0019068886,0.0019908918],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014030697,0.00042270438,0.009645985,0.00031267703,0.00040713005,0.00034566846,0.00036508302,0.1732311,0.02388433,0.00248738,0.0081343,0.77936065],"study_design_scores_gemma":[0.0000357865,0.00013253292,0.0013414026,0.000014295112,0.00006840856,0.000057680052,0.0000525493,0.9894013,0.005842485,0.0021137327,0.0009170203,0.000022867087],"about_ca_topic_score_codex":0.0077243615,"about_ca_topic_score_gemma":0.010580174,"teacher_disagreement_score":0.0077243615,"about_ca_system_score_codex":0.00098127,"about_ca_system_score_gemma":0.0014049234,"threshold_uncertainty_score":0.019958913},"labels":[],"label_agreement":null},{"id":"W2408089248","doi":"","title":"CatolicaSC at TAC 2011","year":2011,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Operations research; Mathematics","score_opus":0.011934085099559373,"score_gpt":0.24925781643947167,"score_spread":0.2373237313399123,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2408089248","genre_codex":"software","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014543308,0.0014698028,0.117938764,0.0058621806,0.006436218,0.0012650648,0.06447619,0.65212756,0.13588081],"genre_scores_gemma":[0.10502788,0.0013677983,0.123938344,0.0036338244,0.0014108717,0.0016272769,0.3203263,0.18668827,0.25597936],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99684757,0.0007764969,0.00018339095,0.00050457416,0.0012591428,0.00042882888],"domain_scores_gemma":[0.9951929,0.0010516248,0.00012716284,0.0017263273,0.001381826,0.00052015035],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004389768,0.0023829478,0.001447202,0.0018952074,0.0021171505,0.004832264,0.0035195316,0.002043134,0.15869091],"category_scores_gemma":[0.011797566,0.0010753911,0.0014285281,0.0019741931,0.00077199034,0.007307539,0.0031718097,0.003636785,0.12466025],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007468665,0.00018369427,0.00044429267,0.0002150672,0.00005331222,0.00024308052,0.00013554755,0.00061073323,0.0030429275,0.0072365766,0.93855256,0.048535418],"study_design_scores_gemma":[0.000555563,0.00024333097,0.0013283703,0.00010773595,0.000071640214,0.00032085736,0.00014545511,0.019887805,0.010603229,0.014369431,0.9522421,0.00012453164],"about_ca_topic_score_codex":0.012958109,"about_ca_topic_score_gemma":0.012899005,"teacher_disagreement_score":0.15869091,"about_ca_system_score_codex":0.0020591943,"about_ca_system_score_gemma":0.002241426,"threshold_uncertainty_score":0.530874},"labels":[],"label_agreement":null},{"id":"W2408490492","doi":"","title":"SYDNEY CMCRC at TAC 2013.","year":2013,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science","score_opus":0.005760547107424089,"score_gpt":0.2413774725978181,"score_spread":0.235616925490394,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2408490492","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0041241576,0.006440949,0.019924393,0.012719495,0.014645488,0.0002677621,0.014498661,0.010046832,0.9173322],"genre_scores_gemma":[0.0048056636,0.0007966523,0.0024393594,0.00025718333,0.00035659937,0.000031742355,0.002602962,0.00056195253,0.98814803],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9990395,0.00012066492,0.00004076958,0.0002767474,0.00036471753,0.00015773613],"domain_scores_gemma":[0.99888784,0.00005971734,0.00004695186,0.00018261501,0.00046840124,0.00035453518],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0011603887,0.0012674237,0.0012711206,0.0024171453,0.0022531631,0.0051729307,0.0016551078,0.0017831858,0.64938074],"category_scores_gemma":[0.0018744568,0.0005715006,0.0007345641,0.0025161132,0.000430491,0.0034016797,0.0028526562,0.0016246298,0.52516],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016850962,0.00007536,0.00028110968,0.00012309231,0.000013823617,0.0001651777,0.00012673579,0.00033808846,0.0011012598,0.009375288,0.87280035,0.11543116],"study_design_scores_gemma":[0.00001163158,0.000019138955,0.00043541376,0.00006669185,0.000006671242,0.00007140444,0.00009315488,0.00055453257,0.0005771093,0.0034473443,0.99470514,0.000011765278],"about_ca_topic_score_codex":0.007824436,"about_ca_topic_score_gemma":0.021794155,"teacher_disagreement_score":0.64938074,"about_ca_system_score_codex":0.0020914555,"about_ca_system_score_gemma":0.0023968164,"threshold_uncertainty_score":0.50011575},"labels":[],"label_agreement":null},{"id":"W2408506617","doi":"","title":"Relational Recognition for Information Extraction in Free Text Documents.","year":2005,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Lockheed Martin (Canada)","funders":"","keywords":"Tuple; Computer science; Relationship extraction; Information extraction; Information retrieval; Bridge (graph theory); Domain (mathematical analysis); Extraction (chemistry); Relation (database); Artificial intelligence; Natural language processing; Data mining; Mathematics","score_opus":0.02135414282681454,"score_gpt":0.28662491780161914,"score_spread":0.2652707749748046,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2408506617","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0018545943,0.002094354,0.9696857,0.00087052333,0.00038268723,0.0006156828,0.0027448102,0.015182422,0.006569247],"genre_scores_gemma":[0.016854323,0.0015808493,0.9697528,0.00035768308,0.00014718584,0.00044960846,0.005906989,0.00058405893,0.0043665487],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99629045,0.0010077483,0.00053831976,0.000608391,0.0014181668,0.00013683048],"domain_scores_gemma":[0.99446446,0.002602241,0.0004798223,0.001354855,0.00097187,0.00012676994],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029965304,0.0011490226,0.0010138579,0.0044723568,0.0013342199,0.003668291,0.0018499715,0.0012671835,0.012353536],"category_scores_gemma":[0.012376147,0.00063587935,0.0015714691,0.004976012,0.0010226619,0.005151059,0.0020329936,0.0019454687,0.02011727],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017065508,0.000118620395,0.00086882786,0.0017059123,0.00016051848,0.000665806,0.00076499593,0.0035897386,0.023280527,0.092235394,0.07994225,0.79649675],"study_design_scores_gemma":[0.00009029539,0.00017104948,0.0018905548,0.0006162572,0.00023020626,0.0028954067,0.00072922494,0.12047151,0.08941353,0.17122363,0.6120733,0.0001950665],"about_ca_topic_score_codex":0.0021992163,"about_ca_topic_score_gemma":0.0033452956,"teacher_disagreement_score":0.012353536,"about_ca_system_score_codex":0.0009649134,"about_ca_system_score_gemma":0.0017491749,"threshold_uncertainty_score":0.0413267},"labels":[],"label_agreement":null},{"id":"W2408560832","doi":"","title":"HEXTAC: the Creation of a Manual Extractive Run","year":2009,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":38,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Automatic summarization; Computer science; Word (group theory); Natural language processing; Quality (philosophy); Artificial intelligence; Competition (biology); Information retrieval; Linguistics; Philosophy","score_opus":0.00937550245727826,"score_gpt":0.30011491854418737,"score_spread":0.2907394160869091,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2408560832","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029152902,0.00073416164,0.835021,0.00023864648,0.0004418653,0.0008625495,0.003285147,0.12001294,0.010250759],"genre_scores_gemma":[0.11325786,0.00034959978,0.8392701,0.0004082514,0.0002726105,0.0010819334,0.015477518,0.012871892,0.0170102],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99334574,0.0021923177,0.00073596585,0.0016824078,0.0017060327,0.00033759722],"domain_scores_gemma":[0.97706556,0.008727183,0.0010490487,0.008505707,0.004283706,0.0003687248],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005747461,0.0018919692,0.0013471371,0.0022405963,0.0011450999,0.0035505567,0.002018038,0.0012096721,0.017705148],"category_scores_gemma":[0.018676238,0.0009951139,0.00082004536,0.0012226263,0.0007141676,0.0029201743,0.0022925264,0.001760959,0.013464614],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021826932,0.00039215197,0.0033869492,0.0012747895,0.0003246427,0.00042251454,0.0014262723,0.004649527,0.12473231,0.0051386193,0.056378018,0.7996915],"study_design_scores_gemma":[0.00062389905,0.0026089537,0.018060237,0.00039329176,0.0007328595,0.0022671327,0.0010092928,0.17880099,0.40651238,0.008298751,0.3801545,0.0005377804],"about_ca_topic_score_codex":0.0015951488,"about_ca_topic_score_gemma":0.0036085853,"teacher_disagreement_score":0.017705148,"about_ca_system_score_codex":0.0005852571,"about_ca_system_score_gemma":0.0021076389,"threshold_uncertainty_score":0.059229612},"labels":[],"label_agreement":null},{"id":"W2408742823","doi":"10.2495/dne-v11-n2-88-96","title":"A multi-agent solution for managing complexity in english to sinhala machine translation","year":2016,"lang":"en","type":"article","venue":"International Journal of Design & Nature and Ecodynamics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Syntax; Artificial intelligence; Machine translation; Natural language processing; Semantics (computer science); Sentence; Transfer-based machine translation; Rule-based machine translation; Process (computing); Point (geometry); Ontology; Translation (biology); Example-based machine translation; Programming language","score_opus":0.03216539525153374,"score_gpt":0.30471134460039967,"score_spread":0.27254594934886595,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2408742823","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.050843336,0.00039360745,0.9335625,0.0010987247,0.00013445604,0.00032513752,0.000047058464,0.0008092437,0.012785909],"genre_scores_gemma":[0.47434384,0.00028139475,0.5131823,0.00022024517,0.00004916073,0.0005327832,0.00010075485,0.000087794746,0.01120182],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99935013,0.000267618,0.00005952999,0.00013653425,0.00011872679,0.00006739561],"domain_scores_gemma":[0.99928147,0.00028124332,0.000078104706,0.000079183956,0.00019277068,0.000087276145],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011045934,0.00067258225,0.00060733827,0.0003624869,0.0016562751,0.0017496838,0.0011461152,0.0011121046,0.002367198],"category_scores_gemma":[0.0018598691,0.00037225193,0.0006430997,0.0003484217,0.0006978196,0.0012032441,0.0023182838,0.0011243417,0.00048230786],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00043370682,0.0004399277,0.0026850908,0.00061168877,0.000298844,0.0025364943,0.0031455257,0.55065286,0.031530276,0.119567454,0.008394883,0.27970317],"study_design_scores_gemma":[0.00009812734,0.00016421564,0.00035974997,0.000031955075,0.000073492854,0.00024070553,0.0003828686,0.95580846,0.004261409,0.0206398,0.017898746,0.00004056204],"about_ca_topic_score_codex":0.0037433717,"about_ca_topic_score_gemma":0.004589738,"teacher_disagreement_score":0.0037433717,"about_ca_system_score_codex":0.0007220041,"about_ca_system_score_gemma":0.0018324789,"threshold_uncertainty_score":0.007919073},"labels":[],"label_agreement":null},{"id":"W2414491991","doi":"10.4242/balisagevol3.sperberg-mcqueen01","title":"Formal and informal meaning from documents through skeleton sentences","year":2009,"lang":"en","type":"article","venue":"Balisage series on markup technologies","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Meaning (existential); Sentence; Set (abstract data type); Natural language; Natural (archaeology); Semantics (computer science); Linguistics; Natural language processing; Complement (music); Natural language generation; Artificial intelligence; Epistemology; Programming language; Philosophy; History","score_opus":0.009119940111175638,"score_gpt":0.2504557326011053,"score_spread":0.24133579248992967,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2414491991","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014082254,0.0027158353,0.93002254,0.0040143956,0.00046848672,0.0003418028,0.0016054595,0.0010049492,0.045744207],"genre_scores_gemma":[0.2554068,0.0040929527,0.7151125,0.0015376643,0.00068858213,0.0008706337,0.0046454393,0.0008809838,0.016764509],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9956044,0.0019347469,0.0005004482,0.0006165694,0.0011411713,0.00020262809],"domain_scores_gemma":[0.9922518,0.0048819752,0.0005764014,0.0012971556,0.000823295,0.00016940644],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0047154073,0.0012954762,0.0005505838,0.003304435,0.0019361969,0.0065447814,0.0016401183,0.002352679,0.008925853],"category_scores_gemma":[0.012504252,0.000982642,0.0018228317,0.0031199192,0.007783898,0.021979246,0.0049127317,0.004860996,0.0026089721],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000023735382,0.000012256358,0.00011181855,0.00021690363,0.000011611436,0.00022623409,0.0033514516,0.00077572383,0.0012466416,0.97698385,0.0023307933,0.0147090545],"study_design_scores_gemma":[0.000016260547,0.000025939913,0.00021370898,0.00017333803,0.00002016951,0.00035190495,0.0008820378,0.0038510142,0.0019797003,0.8887333,0.10371776,0.00003486634],"about_ca_topic_score_codex":0.0020458808,"about_ca_topic_score_gemma":0.0017654208,"teacher_disagreement_score":0.008925853,"about_ca_system_score_codex":0.0022038186,"about_ca_system_score_gemma":0.0016131089,"threshold_uncertainty_score":0.02985996},"labels":[],"label_agreement":null},{"id":"W2415313051","doi":"10.1007/978-3-319-22689-7_19","title":"Asymmetry Theory and Asymmetry Based Parsing","year":2015,"lang":"en","type":"book-chapter","venue":"Communications in computer and information science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Asymmetry; Parsing; Computer science; Natural language processing; Physics; Particle physics","score_opus":0.036806962591631284,"score_gpt":0.3166785686922267,"score_spread":0.2798716061005954,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2415313051","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012445585,0.0038479816,0.79965866,0.0053333603,0.0010630192,0.000071158036,0.000567307,0.0010698981,0.17594309],"genre_scores_gemma":[0.55678713,0.00892895,0.30983353,0.002830031,0.0034598692,0.0004481483,0.0022261979,0.0027805455,0.112705596],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99836344,0.0005637932,0.00012169811,0.0002547264,0.00050449057,0.00019180724],"domain_scores_gemma":[0.9971908,0.001612492,0.00014690317,0.00062650745,0.00034536686,0.00007802102],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020545695,0.0009586488,0.0012101699,0.00254705,0.0017165256,0.0037689104,0.002514716,0.0024050109,0.016761206],"category_scores_gemma":[0.0075611225,0.0010328533,0.0013776729,0.0036246825,0.004320688,0.014177123,0.0027306306,0.005553284,0.0051116897],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000008322911,0.0000048995757,0.00003699291,0.000022038706,0.0000029803991,0.000027580822,0.00006317131,0.00046735923,0.00013226553,0.98201376,0.003147638,0.014072975],"study_design_scores_gemma":[0.0000012693464,0.0000011406194,0.00002088656,0.00000858233,0.0000028741997,0.00003486571,0.000009359703,0.0017208137,0.00013584168,0.99390453,0.0041552493,0.0000045865895],"about_ca_topic_score_codex":0.0008903981,"about_ca_topic_score_gemma":0.0005156745,"teacher_disagreement_score":0.016761206,"about_ca_system_score_codex":0.0016404494,"about_ca_system_score_gemma":0.0012504697,"threshold_uncertainty_score":0.056071818},"labels":[],"label_agreement":null},{"id":"W2417845429","doi":"10.5007/2175-7968.2016v36nesp1p177","title":"Exploring theoretical functions of corpus data in teaching translation","year":2016,"lang":"en","type":"article","venue":"Cadernos de Tradução","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Trois-Rivières","funders":"","keywords":"Computer science; Interpretation (philosophy); Natural language processing; Corpus linguistics; Adaptation (eye); Artificial intelligence; Machine translation; Translation (biology); Noun; Source text; Text corpus; Linguistics; Psychology; Programming language","score_opus":0.13640600440870163,"score_gpt":0.3143859330559562,"score_spread":0.17797992864725456,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2417845429","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07787599,0.004889667,0.7492526,0.029589217,0.00038639246,0.0008202461,0.0007713065,0.0005139641,0.13590066],"genre_scores_gemma":[0.60249776,0.002848145,0.3843374,0.001340179,0.00022599415,0.0024719888,0.0007791417,0.00036828293,0.0051311124],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.95859015,0.036995392,0.0007979825,0.0012416877,0.0020987117,0.0002760843],"domain_scores_gemma":[0.7406469,0.23521079,0.0032753611,0.013256538,0.0065683536,0.0010420199],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04333765,0.00080162816,0.0007765172,0.006729412,0.0039040314,0.011135966,0.0032160636,0.0026060755,0.011647871],"category_scores_gemma":[0.11037693,0.0011198518,0.0007550749,0.007926759,0.014044963,0.023712046,0.008806038,0.004868618,0.0016280302],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011742827,0.00017266221,0.0035436386,0.00052893977,0.000039521925,0.00022777115,0.03141149,0.0029280458,0.0006653393,0.83402014,0.0031121064,0.12323284],"study_design_scores_gemma":[0.00011618602,0.00013241125,0.0015971138,0.001670444,0.000057062447,0.00028701848,0.019493133,0.029668128,0.002522952,0.83628064,0.10810053,0.000074364936],"about_ca_topic_score_codex":0.0029165945,"about_ca_topic_score_gemma":0.0047048116,"teacher_disagreement_score":0.04333765,"about_ca_system_score_codex":0.0056697116,"about_ca_system_score_gemma":0.0047993106,"threshold_uncertainty_score":0.22919416},"labels":[],"label_agreement":null},{"id":"W2423192275","doi":"10.7146/hjlcb.v11i21.25479","title":"Leksikografien på egne ben. Fordelingsstrukturer og byggedele i et brugerorienteret perspektiv","year":2017,"lang":"da","type":"article","venue":"HERMES - Journal of Language and Communication in Business","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Praxis Spinal Cord Institute","funders":"","keywords":"Humanities; Sociology; Philosophy; Art","score_opus":0.019069389850175145,"score_gpt":0.3255831972039896,"score_spread":0.30651380735381445,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2423192275","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07429644,0.055099923,0.053579915,0.05100264,0.015877258,0.00043654174,0.018364001,0.0076236604,0.72371966],"genre_scores_gemma":[0.16403627,0.023612343,0.030329809,0.003226352,0.001956539,0.0001916539,0.013387787,0.0035144228,0.75974476],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99773645,0.00044387835,0.0001755881,0.00049448555,0.0008931598,0.000256528],"domain_scores_gemma":[0.9967836,0.0009877615,0.00027777534,0.00050345215,0.0009549133,0.00049241574],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026340368,0.0009705764,0.000634898,0.0019607383,0.0020082681,0.008000848,0.0013749041,0.0020456742,0.19745612],"category_scores_gemma":[0.0058403513,0.0006362949,0.0005820534,0.0018760898,0.0013216328,0.006131068,0.0042977594,0.0027088893,0.116765395],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006337349,0.00041244645,0.004644579,0.0013769433,0.000050470353,0.0010244965,0.0039837877,0.0019275355,0.00783216,0.039623152,0.26506054,0.6734302],"study_design_scores_gemma":[0.000018499122,0.00006475235,0.0025003687,0.00053280575,0.000025355232,0.00061506074,0.0013847192,0.00053507776,0.0044412524,0.007429578,0.9824086,0.0000438833],"about_ca_topic_score_codex":0.0062942877,"about_ca_topic_score_gemma":0.007261025,"teacher_disagreement_score":0.19745612,"about_ca_system_score_codex":0.0020517684,"about_ca_system_score_gemma":0.0030271916,"threshold_uncertainty_score":0.66055655},"labels":[],"label_agreement":null},{"id":"W245141691","doi":"10.12794/metadc177176","title":"A Comparative Analysis of Web-based Machine Translation Quality: English to French and French to English","year":2012,"lang":"en","type":"dissertation","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Advanced Research Projects Agency; Defense Advanced Research Projects Agency; Alberta-Pacific Forest Industries","keywords":"Machine translation; Natural language processing; Computer science; Linguistics; Translation (biology); Quality (philosophy); Artificial intelligence; World Wide Web; Chemistry; Philosophy","score_opus":0.024650067501143565,"score_gpt":0.33622579588283574,"score_spread":0.3115757283816922,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W245141691","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9629627,0.0045278366,0.0073534106,0.000892024,0.000063695545,0.000084181134,0.0009592947,0.00021947536,0.022937361],"genre_scores_gemma":[0.9953945,0.0007116531,0.0022644175,0.000066255656,0.000047124322,0.000030145071,0.00062681426,0.00008476887,0.0007743151],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9821538,0.008877859,0.0017076778,0.0010172185,0.005832368,0.00041112278],"domain_scores_gemma":[0.8325554,0.119632035,0.012841198,0.0049895495,0.02904118,0.00094054075],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.015754195,0.00029130146,0.0005607076,0.007920799,0.00076437904,0.0039300765,0.00048457808,0.00055470323,0.0034132185],"category_scores_gemma":[0.09308736,0.00024757607,0.0005933617,0.00954303,0.0018164208,0.004195985,0.0012066201,0.0005939603,0.00076166505],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0023137408,0.000348858,0.518099,0.002177704,0.0012529317,0.0009481086,0.0826852,0.0038218289,0.005637053,0.007392368,0.0045829676,0.3707402],"study_design_scores_gemma":[0.00010290359,0.0009246118,0.9425773,0.0006332486,0.0004150469,0.001080651,0.025260637,0.0065344945,0.0053334823,0.0029198593,0.0140803065,0.00013734821],"about_ca_topic_score_codex":0.009400176,"about_ca_topic_score_gemma":0.01082219,"teacher_disagreement_score":0.015754195,"about_ca_system_score_codex":0.0017798066,"about_ca_system_score_gemma":0.0011209784,"threshold_uncertainty_score":0.08331716},"labels":[],"label_agreement":null},{"id":"W2461701302","doi":"","title":"Descriptional complexity of formal systems : 17th International Workshop, DCFS 2015 Waterloo, ON, Canada, June 25-27, 2015 : proceedings","year":2015,"lang":"en","type":"book","venue":"Springer eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Nondeterministic finite automaton; Regular language; Deterministic finite automaton; Deterministic automaton; Mathematics; Discrete mathematics; Nondeterministic algorithm; Succinctness; Pumping lemma for regular languages; Decidability; Quantum finite automata; Formal language; Finite-state machine; Computer science; Automata theory; Automaton; Theoretical computer science; Algorithm; Programming language","score_opus":0.03406205820868684,"score_gpt":0.2609885873500331,"score_spread":0.22692652914134626,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2461701302","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023657303,0.120314255,0.7169267,0.042556234,0.0043462636,0.00023736013,0.0042127734,0.0025313345,0.0852178],"genre_scores_gemma":[0.2951062,0.10338988,0.4231762,0.00326397,0.0050974055,0.0007347887,0.015241626,0.002568696,0.1514213],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99754804,0.000423994,0.00018425426,0.00034111255,0.0013337182,0.00016885402],"domain_scores_gemma":[0.99559516,0.0023571935,0.00011219328,0.00063958,0.0010701409,0.00022568314],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034040161,0.0009445863,0.001369002,0.001686721,0.0013004612,0.0071425256,0.0019149265,0.0010960334,0.016358148],"category_scores_gemma":[0.006509097,0.0013171316,0.0017184084,0.0023670776,0.0029945653,0.006258768,0.0029953115,0.0047620744,0.0021179859],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009597869,0.00010959782,0.00082677364,0.00095635833,0.00008594371,0.0001001682,0.00085358217,0.012199221,0.002112615,0.5136699,0.19402835,0.2749614],"study_design_scores_gemma":[0.000047204183,0.000029207644,0.0015171206,0.00060190435,0.000054008848,0.0002287763,0.00040674818,0.03795051,0.002286363,0.62212926,0.3346809,0.00006801351],"about_ca_topic_score_codex":0.040008523,"about_ca_topic_score_gemma":0.04922455,"teacher_disagreement_score":0.040008523,"about_ca_system_score_codex":0.012275917,"about_ca_system_score_gemma":0.007954078,"threshold_uncertainty_score":0.08906847},"labels":[],"label_agreement":null},{"id":"W2467761858","doi":"10.7202/1028613ar","title":"L’apport des dictionnaires électroniques pour l’élaboration de thésaurus","year":2015,"lang":"fr","type":"article","venue":"Documentation et bibliothèques","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":true,"ca_institutions":"CBC (Canada)","funders":"","keywords":"Humanities; Art; Political science; Philosophy","score_opus":0.04701012017949653,"score_gpt":0.3584889895498974,"score_spread":0.31147886937040087,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2467761858","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023791716,0.0053326664,0.9165322,0.0013829557,0.00076504995,0.0006516557,0.007642055,0.006091061,0.03781071],"genre_scores_gemma":[0.046079047,0.0038067661,0.91798514,0.00026782855,0.0001409313,0.0003461417,0.0072798454,0.001332761,0.0227615],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99703515,0.00086466956,0.0006046686,0.00050958974,0.00088434643,0.00010173282],"domain_scores_gemma":[0.9922166,0.003571056,0.0002903288,0.0015981873,0.0022120757,0.00011172569],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003181198,0.0013278972,0.0011871402,0.0111046415,0.001549869,0.004925302,0.001109864,0.001010761,0.02177952],"category_scores_gemma":[0.0141902175,0.0010000922,0.0019167054,0.009258347,0.0012647911,0.0052479743,0.0024186922,0.0019317991,0.013170432],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00026214196,0.00009375214,0.0035664109,0.003933935,0.00035051984,0.0012872807,0.0051088813,0.004130891,0.046828188,0.0992659,0.024663134,0.81050897],"study_design_scores_gemma":[0.00008007875,0.00014382497,0.008506701,0.0014652964,0.0002745151,0.0022147829,0.0027039913,0.014186053,0.04148189,0.0347231,0.8940345,0.0001853219],"about_ca_topic_score_codex":0.010226754,"about_ca_topic_score_gemma":0.015972165,"teacher_disagreement_score":0.02177952,"about_ca_system_score_codex":0.0015665867,"about_ca_system_score_gemma":0.0031672735,"threshold_uncertainty_score":0.072859704},"labels":[],"label_agreement":null},{"id":"W2467976555","doi":"10.7202/1035940ar","title":"Perspectives de l’architecture Trame/Cadre pour les alignements multilingues","year":2016,"lang":"fr","type":"article","venue":"Nouvelles perspectives en sciences sociales","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Art","score_opus":0.04369134693901564,"score_gpt":0.3539226427018476,"score_spread":0.3102312957628319,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2467976555","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007123525,0.0015148349,0.9716859,0.0016791698,0.00017772944,0.000112243695,0.00033587188,0.0078709135,0.009499745],"genre_scores_gemma":[0.05216524,0.0012681712,0.93107814,0.00035722647,0.00011863654,0.00019455027,0.0013921044,0.0013436937,0.012082235],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9965305,0.0012231772,0.00026960077,0.0008373747,0.00094461546,0.0001948224],"domain_scores_gemma":[0.99432373,0.001842479,0.00027787974,0.0016398744,0.0016885301,0.00022757688],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0061737997,0.0010528759,0.00075375027,0.0022523466,0.0012936372,0.0066241478,0.0024975284,0.0019065602,0.008911258],"category_scores_gemma":[0.008699542,0.0010146725,0.0013563004,0.002237162,0.0022419665,0.0068458524,0.0019969398,0.0027517932,0.005303613],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00043255519,0.00013055529,0.0025444292,0.0012380558,0.00016759666,0.0005289224,0.0036367825,0.03663142,0.05111326,0.40238118,0.01718234,0.48401293],"study_design_scores_gemma":[0.00006651721,0.00024834176,0.00319328,0.0006741844,0.00018971575,0.00076552556,0.0010505093,0.18255948,0.04486084,0.16120653,0.60500026,0.00018479189],"about_ca_topic_score_codex":0.012460093,"about_ca_topic_score_gemma":0.014292651,"teacher_disagreement_score":0.012460093,"about_ca_system_score_codex":0.0021240644,"about_ca_system_score_gemma":0.0038090853,"threshold_uncertainty_score":0.03265059},"labels":[],"label_agreement":null},{"id":"W2468025210","doi":"10.3765/amp.v2i0.3750","title":"Long-Distance Phonotactics as Tier-Based Strictly 2-Local Languages","year":2016,"lang":"en","type":"article","venue":"Proceedings of the Annual Meetings on Phonology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":62,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia; University of Ottawa","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Locality; Phonotactics; Dependency (UML); Computer science; Simple (philosophy); Class (philosophy); Blocking (statistics); Transparency (behavior); Mathematics; Linguistics; Artificial intelligence; Phonology","score_opus":0.006324693909055038,"score_gpt":0.25133418997092144,"score_spread":0.2450094960618664,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2468025210","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7662307,0.00015910542,0.20292139,0.00027721486,0.000041152372,0.00007481696,0.00036002122,0.00074627594,0.029189289],"genre_scores_gemma":[0.9848952,0.00002404161,0.01226766,0.000044371784,0.0000068966538,0.000018413417,0.00008992732,0.000079938785,0.002573481],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9997621,0.00006130288,0.00001946904,0.00005180715,0.000057015426,0.000048422615],"domain_scores_gemma":[0.99925846,0.00020867365,0.00012858782,0.00021943141,0.00010534149,0.00007953648],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00035535562,0.00010711765,0.00015330607,0.00042091915,0.00032417892,0.0013350747,0.00038913763,0.00031445446,0.0037882908],"category_scores_gemma":[0.0012588929,0.0001818462,0.00030982026,0.0004196649,0.0013949223,0.002436933,0.00090015755,0.00050746725,0.00045153053],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020334304,0.000035517216,0.00936658,0.00009997157,0.000022494341,0.00043854545,0.004890555,0.006461264,0.06871676,0.8782402,0.0010125564,0.030512186],"study_design_scores_gemma":[0.0000617213,0.00017258279,0.03453377,0.000037990558,0.000041239306,0.0013723075,0.0027133196,0.059710413,0.011359249,0.8722532,0.017632434,0.000111689966],"about_ca_topic_score_codex":0.0012201942,"about_ca_topic_score_gemma":0.0024766217,"teacher_disagreement_score":0.0037882908,"about_ca_system_score_codex":0.00045874386,"about_ca_system_score_gemma":0.00034223014,"threshold_uncertainty_score":0.01267308},"labels":[],"label_agreement":null},{"id":"W2468840277","doi":"10.18653/v1/s16-1151","title":"CLaC at SemEval-2016 Task 11: Exploring linguistic and psycho-linguistic Features for Complex Word Identification","year":2016,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"SemEval; Task (project management); Computer science; Natural language processing; Identification (biology); Word (group theory); Context (archaeology); Artificial intelligence; Ranking (information retrieval); Random forest; Linguistics; History","score_opus":0.06689074162379126,"score_gpt":0.33920537470663775,"score_spread":0.2723146330828465,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2468840277","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.43076473,0.0077439607,0.1696803,0.0027116477,0.0034165632,0.0037180553,0.1314178,0.18896365,0.06158335],"genre_scores_gemma":[0.51162803,0.000735951,0.21764514,0.0014336635,0.00042258122,0.002216992,0.24023375,0.0051553547,0.02052848],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.997269,0.00076892384,0.00019754267,0.0010466083,0.00047898994,0.00023892226],"domain_scores_gemma":[0.99491185,0.00232533,0.00023738164,0.0011277036,0.0009410965,0.00045671387],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032994556,0.0034116602,0.001836872,0.0025950144,0.0013795721,0.0032469016,0.002550147,0.0033731079,0.020810349],"category_scores_gemma":[0.01218396,0.0007064026,0.0016991971,0.0012620456,0.0006390254,0.004533738,0.0037728532,0.0030695975,0.02090353],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0031235358,0.0015265538,0.016696377,0.0027187744,0.00048016733,0.0009659003,0.0011501536,0.0073749935,0.043318827,0.0028445271,0.33942148,0.5803788],"study_design_scores_gemma":[0.0017696671,0.002496602,0.091889374,0.0008797539,0.00064637687,0.006193645,0.003321089,0.46503782,0.109526716,0.026860127,0.29046404,0.00091477623],"about_ca_topic_score_codex":0.008314414,"about_ca_topic_score_gemma":0.014832388,"teacher_disagreement_score":0.020810349,"about_ca_system_score_codex":0.0012026798,"about_ca_system_score_gemma":0.0018537854,"threshold_uncertainty_score":0.06961751},"labels":[],"label_agreement":null},{"id":"W2471177583","doi":"10.18653/v1/s16-1102","title":"CNRC at SemEval-2016 Task 1: Experiments in Crosslingual Semantic Textual Similarity","year":2016,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Computer science; SemEval; Semantic similarity; Natural language processing; Similarity (geometry); Artificial intelligence; Semantics (computer science); Machine translation; Task (project management); Information retrieval; Programming language","score_opus":0.02861404306301647,"score_gpt":0.3041952073667663,"score_spread":0.27558116430374985,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2471177583","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8111943,0.005373276,0.05351874,0.0021834536,0.0031416165,0.005804078,0.034073256,0.03491868,0.04979259],"genre_scores_gemma":[0.6685597,0.0008214624,0.14803241,0.0026322496,0.00075827364,0.0048376215,0.1443532,0.005752175,0.024252947],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.97948575,0.01089095,0.0019999149,0.0040636314,0.002743981,0.00081581227],"domain_scores_gemma":[0.96615535,0.017011428,0.00090618484,0.009226268,0.004862634,0.0018380684],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01421357,0.0032793044,0.002975051,0.002043738,0.0030728155,0.0028683285,0.0062680575,0.005317229,0.014305293],"category_scores_gemma":[0.049189005,0.0011818508,0.0018513747,0.0027827332,0.0021175193,0.009000408,0.0061616986,0.004429095,0.012596355],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.01647766,0.018579941,0.013981768,0.008563082,0.0016606593,0.0034076434,0.007099921,0.030245075,0.053796608,0.0071632504,0.31022042,0.52880394],"study_design_scores_gemma":[0.018230533,0.021719823,0.06332067,0.001293035,0.0012662212,0.011006423,0.011039088,0.32103708,0.18557225,0.037397567,0.3264605,0.0016568147],"about_ca_topic_score_codex":0.013692117,"about_ca_topic_score_gemma":0.011191885,"teacher_disagreement_score":0.014305293,"about_ca_system_score_codex":0.0018741586,"about_ca_system_score_gemma":0.0020935077,"threshold_uncertainty_score":0.0751695},"labels":[],"label_agreement":null},{"id":"W2473554196","doi":"10.29173/cais6","title":"Apports de la generation automatique de textes en langue naturelle a la recherche d'information","year":2013,"lang":"fr","type":"article","venue":"Proceedings of the Annual Conference of CAIS / Actes du congrès annuel de l ACSI","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Philosophy","score_opus":0.04668985207382938,"score_gpt":0.3015774937197818,"score_spread":0.25488764164595246,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2473554196","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0652844,0.001086946,0.8913446,0.0019329269,0.0006339815,0.000997294,0.0005818496,0.013284952,0.024853133],"genre_scores_gemma":[0.20189956,0.000863926,0.74031204,0.000651181,0.00021846595,0.00083237846,0.0018081721,0.0041943747,0.049219925],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9845035,0.0074555175,0.0009450452,0.0019839683,0.0046652528,0.0004467099],"domain_scores_gemma":[0.9604617,0.025582323,0.0010112075,0.005394768,0.0069909124,0.0005591561],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008661647,0.0019262497,0.001145713,0.0019246982,0.0018572456,0.005317572,0.001973796,0.0025439651,0.020458166],"category_scores_gemma":[0.03746372,0.00096165395,0.0018399898,0.0014725744,0.0019315097,0.0043673906,0.0030095219,0.0023203958,0.010808222],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015781707,0.00052292447,0.003789223,0.002452898,0.00020113369,0.002128659,0.027977725,0.008479245,0.13837302,0.039457094,0.020084657,0.7549553],"study_design_scores_gemma":[0.0003725647,0.001424374,0.0075398725,0.0008130138,0.0005408866,0.0033626237,0.0090826005,0.108539775,0.30760902,0.027078724,0.5331502,0.00048634343],"about_ca_topic_score_codex":0.0035177479,"about_ca_topic_score_gemma":0.0033512325,"teacher_disagreement_score":0.020458166,"about_ca_system_score_codex":0.0010896589,"about_ca_system_score_gemma":0.002131798,"threshold_uncertainty_score":0.068439364},"labels":[],"label_agreement":null},{"id":"W2473631628","doi":"10.4000/anglophonia.775","title":"A natural-language semantics approach to infinitival and gerund-participial complementation in English","year":2007,"lang":"fr","type":"article","venue":"Anglophonia Caliban/Sigma","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Philosophy; Humanities; Linguistics","score_opus":0.022156190420336813,"score_gpt":0.3058076419473211,"score_spread":0.28365145152698423,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2473631628","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.28609172,0.0038469199,0.42605332,0.00546805,0.00022986227,0.00009575389,0.00017627665,0.0002691358,0.27776897],"genre_scores_gemma":[0.970768,0.00048280292,0.019016683,0.00016356674,0.000066184075,0.00004085399,0.00004988271,0.000057047706,0.0093550375],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99939,0.00031771144,0.000044095457,0.00008395377,0.0000957881,0.00006839011],"domain_scores_gemma":[0.9991247,0.00051866955,0.00009927883,0.00006891938,0.00016564147,0.000022723905],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00090648496,0.0004169953,0.00030424877,0.0013128489,0.001421575,0.0027618415,0.00070283824,0.0006351404,0.0034399766],"category_scores_gemma":[0.0014038138,0.00035181287,0.00045975597,0.0007233304,0.004823661,0.0048050736,0.0011159058,0.00091780967,0.00030636485],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000016384794,0.000009591169,0.0007200635,0.000046295703,0.0000043745968,0.00027161688,0.008488809,0.00050630776,0.00089954655,0.9824019,0.00020728417,0.0064278655],"study_design_scores_gemma":[0.000030013809,0.0000588262,0.004084152,0.00016389259,0.000044767727,0.0013273952,0.017422467,0.010574216,0.0046988963,0.88748306,0.07405534,0.000056905603],"about_ca_topic_score_codex":0.005794085,"about_ca_topic_score_gemma":0.006595821,"teacher_disagreement_score":0.005794085,"about_ca_system_score_codex":0.0023409284,"about_ca_system_score_gemma":0.0011891287,"threshold_uncertainty_score":0.016984701},"labels":[],"label_agreement":null},{"id":"W2475426007","doi":"10.1075/nlp.9.05hab","title":"Arabic preprocessing for Statistical Machine Translation","year":2012,"lang":"en","type":"book-chapter","venue":"Natural language processing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Arabic; Preprocessor; Computer science; Natural language processing; Translation (biology); Machine translation; Artificial intelligence; Linguistics; Philosophy; Chemistry","score_opus":0.01989698859571343,"score_gpt":0.29778559573148655,"score_spread":0.2778886071357731,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2475426007","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0032104745,0.041367218,0.68170106,0.00276356,0.004012957,0.00031321688,0.001595624,0.01123346,0.25380233],"genre_scores_gemma":[0.030514793,0.03346782,0.71410185,0.001585932,0.0022318328,0.00053095986,0.0044124536,0.004429098,0.2087253],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992717,0.00015896565,0.00005679534,0.0001300971,0.00035835183,0.000024169782],"domain_scores_gemma":[0.99902785,0.00039515,0.000052389725,0.00021642969,0.00027973382,0.000028476385],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00070880493,0.0019391692,0.00069329754,0.0020852885,0.0008792074,0.0024801015,0.00090902305,0.0009274623,0.056097288],"category_scores_gemma":[0.0025095053,0.0005148371,0.00063174183,0.0040768175,0.0005829395,0.0020397108,0.0010717175,0.0020889402,0.058444213],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005313639,0.000041191663,0.00016354368,0.00083541684,0.000029262543,0.00023148324,0.00020551469,0.002020884,0.012822002,0.055457477,0.13941687,0.78872323],"study_design_scores_gemma":[0.000008660308,0.000044878787,0.0004376755,0.0002818006,0.000016965228,0.0008856605,0.00006760712,0.008333332,0.011935171,0.040272687,0.9376802,0.000035262434],"about_ca_topic_score_codex":0.000580272,"about_ca_topic_score_gemma":0.0009912975,"teacher_disagreement_score":0.056097288,"about_ca_system_score_codex":0.00074615376,"about_ca_system_score_gemma":0.0008219837,"threshold_uncertainty_score":0.18766409},"labels":[],"label_agreement":null},{"id":"W2475431137","doi":"10.4018/978-1-4666-3970-6.ch004","title":"Machine Learning Approaches for Bangla Statistical Machine Translation","year":2013,"lang":"en","type":"book-chapter","venue":"IGI Global eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Bengali; Machine translation; Computer science; Artificial intelligence; Natural language processing; Selection (genetic algorithm); Machine learning; Sentence; Evaluation of machine translation; Phrase; Example-based machine translation; Machine translation software usability","score_opus":0.03521928799437656,"score_gpt":0.26464482612106704,"score_spread":0.22942553812669048,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2475431137","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0021078791,0.008696623,0.9711779,0.0007861773,0.00031342884,0.000066873516,0.00020340078,0.0010125232,0.015635168],"genre_scores_gemma":[0.06349309,0.0141525855,0.8909906,0.00047185717,0.0005132411,0.00033078907,0.0012243491,0.00037627344,0.028447201],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991806,0.00028015947,0.00007887375,0.00017275718,0.00025725947,0.000030303918],"domain_scores_gemma":[0.99892265,0.00058783486,0.000071248025,0.00017032966,0.00022662773,0.000021318609],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001080938,0.0010273786,0.0006572789,0.0014914235,0.0006477075,0.00223693,0.001403582,0.0009841612,0.008038384],"category_scores_gemma":[0.0028561756,0.0004968769,0.0009821778,0.002759022,0.0006978056,0.0018005524,0.0010896288,0.0020852338,0.008257497],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000036061032,0.000081950595,0.0006322324,0.00068549585,0.00009596588,0.00020166412,0.00026132245,0.033034403,0.0064699,0.12084407,0.020074263,0.8175827],"study_design_scores_gemma":[0.00001774208,0.000072924435,0.0012854962,0.00025354535,0.0000632218,0.00076216314,0.0001564113,0.5447463,0.007816744,0.27268696,0.17206143,0.000077095785],"about_ca_topic_score_codex":0.0012708488,"about_ca_topic_score_gemma":0.0019822621,"teacher_disagreement_score":0.008038384,"about_ca_system_score_codex":0.0010394809,"about_ca_system_score_gemma":0.0007720352,"threshold_uncertainty_score":0.026891053},"labels":[],"label_agreement":null},{"id":"W2478414704","doi":"10.4018/978-1-4666-2169-5.ch011","title":"NLP and Digital Library Management","year":2012,"lang":"en","type":"book-chapter","venue":"IGI Global eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Parsing; Natural language processing; Named-entity recognition; Artificial intelligence; Task (project management); Digital library; Field (mathematics); Anaphora (linguistics); Word (group theory); Resolution (logic); Information retrieval; Linguistics; Engineering","score_opus":0.009291320080401802,"score_gpt":0.2202000755509445,"score_spread":0.21090875547054272,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2478414704","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.002004835,0.13083363,0.02998105,0.017602649,0.0007913069,0.00007224771,0.0002262264,0.0006203,0.8178679],"genre_scores_gemma":[0.093935825,0.23297486,0.07247063,0.007579572,0.0025535598,0.00036254243,0.0013327571,0.00057452574,0.5882157],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99875355,0.00042127355,0.00006591618,0.00018425293,0.0004903764,0.00008450441],"domain_scores_gemma":[0.9992889,0.00040065203,0.0000604087,0.00011492938,0.000087861714,0.000047118636],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009904291,0.00044913858,0.00046178102,0.002818706,0.0018543369,0.00852118,0.0017195753,0.0019615458,0.026431423],"category_scores_gemma":[0.0021262576,0.00026304036,0.00028391124,0.008073932,0.004090169,0.009383494,0.0029194579,0.0017375577,0.009435724],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000070794813,0.000032885007,0.00010147588,0.00055084395,0.0000064832648,0.00015103116,0.0010984085,0.0010901678,0.0002320447,0.61209565,0.08126393,0.30337],"study_design_scores_gemma":[0.0000031085165,0.0000036687932,0.00014798433,0.0003372912,0.0000030129606,0.000157078,0.0004801071,0.000541105,0.00012835758,0.19863975,0.79955083,0.00000768442],"about_ca_topic_score_codex":0.004231893,"about_ca_topic_score_gemma":0.0040532923,"teacher_disagreement_score":0.026431423,"about_ca_system_score_codex":0.005187493,"about_ca_system_score_gemma":0.0030064934,"threshold_uncertainty_score":0.08842188},"labels":[{"model":"gemma","categories":[],"domain":null,"study_design":"not_applicable","genre":"methods","about_ca_system":false,"about_ca_topic":false,"confidence":"low"},{"model":"gpt","categories":[],"domain":null,"study_design":"theoretical_or_conceptual","genre":"methods","about_ca_system":false,"about_ca_topic":false,"confidence":"high"}],"label_agreement":"split"},{"id":"W2478838389","doi":"10.1075/slcs.167.11new","title":"Low-level patterning of pronominal subjects and verb tenses in English","year":2015,"lang":"en","type":"book-chapter","venue":"Studies in language companion series","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Linguistics; Verb; Subject (documents); Focus (optics); Personal pronoun; Argument (complex analysis); Computer science; Past tense; Semantics (computer science); Psychology; Natural language processing; Artificial intelligence; Philosophy; Medicine","score_opus":0.064386435882927,"score_gpt":0.31792672511246467,"score_spread":0.25354028922953764,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2478838389","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98306113,0.001621654,0.0037045674,0.00015692497,0.0000075661574,0.000011920776,0.0003399108,0.00003427107,0.011062063],"genre_scores_gemma":[0.99411017,0.00053053163,0.0019278642,0.000030384348,0.000015406704,0.000018938701,0.00033846748,0.000061489205,0.0029667562],"study_design_codex":"observational","study_design_gemma":"not_applicable","domain_scores_codex":[0.99932694,0.00024450946,0.000058150956,0.00019364068,0.00011800071,0.00005877104],"domain_scores_gemma":[0.99666303,0.0022946922,0.000543808,0.00016009957,0.00022157124,0.00011677733],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009190995,0.00012944975,0.00024775366,0.0021674258,0.0008416938,0.0015343577,0.0002645305,0.00022370135,0.002416683],"category_scores_gemma":[0.0028328502,0.0002858227,0.0001367501,0.0024773427,0.0015562093,0.0017441688,0.00092349685,0.00042965633,0.0004595643],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007024212,0.00015556239,0.3515414,0.00111507,0.00013561781,0.0011746851,0.25335082,0.0003856902,0.13423939,0.055028744,0.002061037,0.20010963],"study_design_scores_gemma":[0.00001611807,0.00014008839,0.9315291,0.00017172568,0.000051725947,0.0013173077,0.025293358,0.0010018757,0.006166206,0.012775982,0.021481177,0.00005536069],"about_ca_topic_score_codex":0.0017660148,"about_ca_topic_score_gemma":0.00563557,"teacher_disagreement_score":0.002416683,"about_ca_system_score_codex":0.00038522575,"about_ca_system_score_gemma":0.00033354736,"threshold_uncertainty_score":0.008084595},"labels":[],"label_agreement":null},{"id":"W2482111388","doi":"10.1093/acprof:oso/9780199684359.003.0009","title":"Antisymmetry and Hixkaryana*","year":2013,"lang":"en","type":"book-chapter","venue":"Oxford University Press eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Antisymmetry; Antisymmetric relation; Object (grammar); Raising (metalworking); Subject (documents); Generalization; Specifier; Computer science; Complement (music); Order (exchange); Linguistics; Mathematics; Artificial intelligence; Geometry; Philosophy; Noun phrase","score_opus":0.01351248911769882,"score_gpt":0.19849025519653635,"score_spread":0.18497776607883754,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2482111388","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.22178118,0.0031053277,0.11858574,0.0034661812,0.00040505777,0.000086075706,0.0004060103,0.0005939202,0.65157044],"genre_scores_gemma":[0.927715,0.00070902164,0.014465495,0.00042104156,0.00009605645,0.000044157507,0.00027727793,0.00017522424,0.056096695],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99974996,0.0000622285,0.000017597944,0.00005104299,0.00006940515,0.000049719318],"domain_scores_gemma":[0.9998047,0.0000786259,0.000026817785,0.00003364368,0.000045478715,0.0000107494125],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00032936616,0.00030009288,0.00017621425,0.0004952087,0.00089874014,0.001390644,0.00054527767,0.0003660816,0.008883182],"category_scores_gemma":[0.0005548796,0.00022123808,0.00034174192,0.0004080871,0.002912026,0.002760191,0.0010873429,0.0011012586,0.0010473127],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000015293252,0.0000036193808,0.00028227986,0.000032526033,0.0000024451122,0.00007437654,0.0009797763,0.00010658674,0.0011068516,0.98563474,0.0010981709,0.010663316],"study_design_scores_gemma":[0.000012827168,0.00003741003,0.0020268671,0.00007358598,0.000013612828,0.00063302607,0.0018852963,0.002914105,0.00433291,0.85046136,0.13758342,0.000025639185],"about_ca_topic_score_codex":0.0015013532,"about_ca_topic_score_gemma":0.0014253863,"teacher_disagreement_score":0.008883182,"about_ca_system_score_codex":0.0008774507,"about_ca_system_score_gemma":0.0005319843,"threshold_uncertainty_score":0.029717207},"labels":[],"label_agreement":null},{"id":"W2482638226","doi":"10.1075/slcs.158.05bar","title":"Chapter 5. A data-driven analysis of the structure type ‘man–nature relationship’ in Romanian","year":2014,"lang":"en","type":"book-chapter","venue":"Studies in language companion series","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Romanian; Type (biology); Geography; Linguistics; Philosophy; Geology; Paleontology","score_opus":0.041767781892636906,"score_gpt":0.3310492307393064,"score_spread":0.2892814488466695,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2482638226","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07410443,0.017980307,0.38501552,0.008043918,0.000989791,0.0005048011,0.009336786,0.0010034661,0.50302094],"genre_scores_gemma":[0.5062547,0.014157833,0.23887773,0.0019508428,0.0007570777,0.0010042663,0.013139203,0.0018206256,0.22203763],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.9996216,0.00014647635,0.000017527405,0.00008687463,0.00010564603,0.000021787093],"domain_scores_gemma":[0.9992925,0.0004939801,0.0000337824,0.0000893061,0.00008055579,0.000009839455],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00063963933,0.0003336135,0.00034251637,0.0012836868,0.00085089763,0.0017223704,0.00064676034,0.00042043853,0.010592744],"category_scores_gemma":[0.0018836044,0.0003602068,0.00067501026,0.002411677,0.0011598046,0.0019559425,0.00067037635,0.0014044424,0.0021414214],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000061808554,0.00008712971,0.0035917978,0.00071215036,0.00003984937,0.00044183305,0.0064680367,0.004095907,0.0038665622,0.7945563,0.05242906,0.13364948],"study_design_scores_gemma":[0.000011325457,0.000036204394,0.012221449,0.00050154026,0.000026881458,0.00047217793,0.0015447024,0.011845638,0.0050556166,0.21358815,0.7546574,0.000038819762],"about_ca_topic_score_codex":0.002523225,"about_ca_topic_score_gemma":0.004157882,"teacher_disagreement_score":0.010592744,"about_ca_system_score_codex":0.0019226822,"about_ca_system_score_gemma":0.0006928196,"threshold_uncertainty_score":0.035436213},"labels":[],"label_agreement":null},{"id":"W2484137238","doi":"10.4018/978-1-4666-8690-8.ch009","title":"Translational Mismatches Involving Clitics (Illustrated from Serbian ~ Catalan Language Pair)","year":2015,"lang":"en","type":"book-chapter","venue":"Advances in linguistics and communication studies","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Clitic; Serbian; Catalan; Rotation formalisms in three dimensions; Linguistics; Phrase; Computer science; Natural language processing; Artificial intelligence; Mathematics; Philosophy","score_opus":0.056768424272511296,"score_gpt":0.3457635642439991,"score_spread":0.28899513997148785,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2484137238","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2756356,0.014839433,0.060355,0.0039801844,0.0025258537,0.0003207678,0.0036598807,0.0019675777,0.6367157],"genre_scores_gemma":[0.8565368,0.0033444492,0.053114574,0.0010771106,0.00021026625,0.00021093286,0.0038841418,0.001026058,0.0805956],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99936,0.0002896582,0.000039873285,0.000117024116,0.00014473624,0.000048673148],"domain_scores_gemma":[0.99930775,0.00029067518,0.00005136058,0.00013364192,0.00019345774,0.000023097124],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00085081335,0.0008095241,0.0004441051,0.0013417202,0.0017544532,0.002080826,0.0006985518,0.000585749,0.008322844],"category_scores_gemma":[0.0017838147,0.00030854007,0.0002969384,0.0017069725,0.0016759232,0.0011315924,0.0013613501,0.0015294286,0.0026014415],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009243375,0.00013926374,0.006572227,0.0017913644,0.000098999175,0.0058587147,0.034856535,0.004631682,0.03344594,0.4211709,0.12132223,0.3691879],"study_design_scores_gemma":[0.000046314202,0.000107602245,0.012607776,0.00043593752,0.00005183676,0.0032570395,0.006979821,0.005095287,0.01465521,0.02437962,0.93230677,0.00007689245],"about_ca_topic_score_codex":0.021538308,"about_ca_topic_score_gemma":0.033215437,"teacher_disagreement_score":0.021538308,"about_ca_system_score_codex":0.0024622316,"about_ca_system_score_gemma":0.0015803379,"threshold_uncertainty_score":0.042825878},"labels":[],"label_agreement":null},{"id":"W2484793855","doi":"10.1075/btl.79.02bow","title":"1. A comparative evaluation of bilingual concordancers and translation memory systems","year":2008,"lang":"en","type":"book-chapter","venue":"Benjamins translation library","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":36,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Translation (biology); Computer science; Natural language processing; Psychology; Linguistics; Biology; Philosophy","score_opus":0.08412356182501192,"score_gpt":0.3005762978231263,"score_spread":0.21645273599811438,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2484793855","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7110936,0.033064716,0.049903944,0.0014995896,0.00073519326,0.0014997269,0.0054843635,0.008436677,0.18828213],"genre_scores_gemma":[0.85322076,0.0077037564,0.086178824,0.00050295796,0.00021029527,0.0004100653,0.013147493,0.001988619,0.03663724],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.97374356,0.01536748,0.0017707042,0.0017853488,0.0070035313,0.00032941028],"domain_scores_gemma":[0.94263214,0.04035654,0.0016446967,0.004257693,0.010408175,0.00070077484],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02308329,0.00054969895,0.0009743126,0.0038091522,0.0011819038,0.004271892,0.0013609936,0.001025713,0.010385183],"category_scores_gemma":[0.056098476,0.00032849342,0.00051074114,0.005561313,0.00091554574,0.0031899225,0.00193952,0.0004728977,0.004531752],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.005738016,0.0005348276,0.016295856,0.0057475697,0.00047951806,0.000364795,0.011965149,0.004290542,0.018000757,0.007989057,0.027620493,0.9009734],"study_design_scores_gemma":[0.0022603092,0.020311348,0.24955243,0.0064038453,0.0034697244,0.0049022203,0.03573851,0.083934635,0.19221939,0.020642834,0.37964773,0.00091698056],"about_ca_topic_score_codex":0.0032445383,"about_ca_topic_score_gemma":0.0062414245,"teacher_disagreement_score":0.02308329,"about_ca_system_score_codex":0.0014113628,"about_ca_system_score_gemma":0.0016137072,"threshold_uncertainty_score":0.122077525},"labels":[],"label_agreement":null},{"id":"W2487056937","doi":"10.1017/cbo9780511843815.014","title":"Imperfect Active Indicative and Imperfect of the Verb “to be”","year":2011,"lang":"en","type":"book-chapter","venue":"Cambridge University Press eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Imperfect; Verb; Action (physics); Linguistics; Present tense; Computer science; Context (archaeology); Past tense; Artificial intelligence; Natural language processing; Present perfect; History; Philosophy; Physics","score_opus":0.017904432495727,"score_gpt":0.21245701338847958,"score_spread":0.19455258089275257,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2487056937","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0070021176,0.003860415,0.079687916,0.0031138037,0.0020163541,0.000115781346,0.0010796353,0.0010141921,0.9021097],"genre_scores_gemma":[0.38137496,0.0057460656,0.045347426,0.002239878,0.002579962,0.00022077054,0.0029329367,0.0018903462,0.5576677],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.999587,0.00013430018,0.00003542253,0.00009812529,0.000112814596,0.00003230861],"domain_scores_gemma":[0.99942756,0.00031666434,0.00004969987,0.00009997371,0.00008256367,0.000023545841],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006678117,0.00085017434,0.00030850008,0.0011724522,0.0010679755,0.0027329298,0.0007239649,0.0010799284,0.019480087],"category_scores_gemma":[0.0013481992,0.00065356976,0.00035385918,0.00089047407,0.0032269612,0.005907321,0.0010687521,0.0028193186,0.005961902],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000047187325,0.00001196422,0.00014226134,0.00021345328,0.000005820966,0.00015326167,0.0026476167,0.000110366454,0.0015079193,0.9409881,0.03531846,0.018853497],"study_design_scores_gemma":[0.000008452403,0.000034668632,0.0006602209,0.00020808344,0.000020250036,0.000854149,0.0005049331,0.00085690914,0.0022875234,0.11056931,0.88397527,0.000020194762],"about_ca_topic_score_codex":0.0014663704,"about_ca_topic_score_gemma":0.003188801,"teacher_disagreement_score":0.019480087,"about_ca_system_score_codex":0.0010487649,"about_ca_system_score_gemma":0.00088126183,"threshold_uncertainty_score":0.06516737},"labels":[],"label_agreement":null},{"id":"W2487318846","doi":"10.1075/lllt.21.07ham","title":"3. Natural language processing tools and CALL","year":2008,"lang":"fr","type":"book-chapter","venue":"Language learning and language teaching","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Computer science; Natural (archaeology); History; Archaeology","score_opus":0.014983967531578376,"score_gpt":0.2776514782608649,"score_spread":0.2626675107292865,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2487318846","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01705654,0.0073263757,0.6871699,0.009576108,0.000684063,0.00036079055,0.0024433613,0.016410397,0.25897253],"genre_scores_gemma":[0.2916561,0.008144151,0.48443717,0.0052860053,0.0009765951,0.0007333931,0.0052929036,0.005191661,0.19828205],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99693954,0.0007816068,0.00024494986,0.00066880777,0.001012036,0.00035302024],"domain_scores_gemma":[0.9964735,0.0020119364,0.00018860625,0.00073259167,0.00049401436,0.00009933649],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021357993,0.00085370435,0.0005050465,0.0027027335,0.0017284341,0.011101871,0.001988166,0.0024779153,0.03052527],"category_scores_gemma":[0.0061921533,0.0006691649,0.0012420062,0.0023646455,0.0032973988,0.010486543,0.0031007165,0.0023090413,0.009455162],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013481891,0.000064612344,0.0015487793,0.00082715554,0.00004489931,0.0006701052,0.004782787,0.0043552737,0.0068102702,0.66848636,0.030267261,0.28200766],"study_design_scores_gemma":[0.000016291904,0.000034356875,0.0015369375,0.00041967412,0.000040990475,0.00068245985,0.0015923688,0.010483883,0.007597238,0.16154134,0.8159825,0.0000720367],"about_ca_topic_score_codex":0.011582745,"about_ca_topic_score_gemma":0.011200823,"teacher_disagreement_score":0.03052527,"about_ca_system_score_codex":0.0023345582,"about_ca_system_score_gemma":0.0025312498,"threshold_uncertainty_score":0.10211718},"labels":[],"label_agreement":null},{"id":"W2490167115","doi":"10.1017/cbo9780511808876.005","title":"Context-free grammars and languages","year":2008,"lang":"en","type":"book-chapter","venue":"Cambridge University Press eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Rule-based machine translation; Computer science; Content (measure theory); Context-free grammar; Context (archaeology); Natural language processing; Programming language; Artificial intelligence; Mathematics; Geography","score_opus":0.014700870138824481,"score_gpt":0.20343569166564562,"score_spread":0.18873482152682114,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2490167115","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004252501,0.1931495,0.15717806,0.008735236,0.0023116588,0.00010576315,0.0010528659,0.0018790446,0.63133544],"genre_scores_gemma":[0.12258504,0.18738417,0.10110484,0.004567356,0.00239426,0.0006003227,0.0035459392,0.0015489369,0.57626915],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9995702,0.00013579987,0.000027509233,0.00006605834,0.00016529452,0.000035136323],"domain_scores_gemma":[0.9996123,0.00024997845,0.000011854344,0.00006641031,0.0000419577,0.000017582906],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005923273,0.0009562406,0.00058386836,0.0014369839,0.00089889514,0.0026177445,0.0009778426,0.0010758323,0.021241156],"category_scores_gemma":[0.0013638189,0.00059235585,0.0006210906,0.0027265872,0.0030352338,0.0043249098,0.0012579041,0.0022135589,0.008863699],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000010164004,0.000014076861,0.0000799499,0.0004799685,0.000011783865,0.00013679688,0.00078514987,0.0016941276,0.0005392143,0.80417603,0.07406969,0.11800299],"study_design_scores_gemma":[0.000004184614,0.000006430464,0.00008735299,0.00019415237,0.0000044290996,0.000168047,0.000102679034,0.00040881927,0.00019990685,0.63789785,0.36091805,0.000008034072],"about_ca_topic_score_codex":0.0018156541,"about_ca_topic_score_gemma":0.0025711553,"teacher_disagreement_score":0.021241156,"about_ca_system_score_codex":0.0015262332,"about_ca_system_score_gemma":0.0015836376,"threshold_uncertainty_score":0.07105875},"labels":[],"label_agreement":null},{"id":"W2490467509","doi":"10.1075/cilt.322.06sci","title":"Perspectives on morphological complexity","year":2012,"lang":"en","type":"book-chapter","venue":"Amsterdam studies in the theory and history of linguistic science. Series 4, Current issues in linguistic theory","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Biology; Evolutionary biology","score_opus":0.0767060282888059,"score_gpt":0.35491976217726356,"score_spread":0.27821373388845766,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2490467509","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.028458,0.045420825,0.12204808,0.03684655,0.001509075,0.00002768455,0.0002480994,0.0002093265,0.7652324],"genre_scores_gemma":[0.8272537,0.045031223,0.0546727,0.006703464,0.0064864103,0.00014505908,0.00049261254,0.00043284593,0.058782022],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9987293,0.00043278388,0.000069511996,0.0002723909,0.00038031666,0.0001157552],"domain_scores_gemma":[0.9963779,0.002597291,0.00019998688,0.0003806161,0.0002923583,0.00015187268],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012878868,0.0006123634,0.0005642368,0.003144465,0.001902082,0.006708781,0.0011851799,0.0017688265,0.010676116],"category_scores_gemma":[0.0037476118,0.0003007574,0.0008048111,0.0021858746,0.01590428,0.014989577,0.0035608455,0.0041941246,0.0014795419],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000042062984,0.0000033613792,0.00008236731,0.00003024693,0.0000022201114,0.000027472674,0.00035090954,0.00021878502,0.000091904374,0.992922,0.00077192625,0.005494674],"study_design_scores_gemma":[0.0000016581762,0.000005436541,0.00020521315,0.0000346866,0.0000017634151,0.00008411142,0.00021679685,0.0004343484,0.00007687131,0.9662222,0.03271122,0.00000578239],"about_ca_topic_score_codex":0.00080066593,"about_ca_topic_score_gemma":0.000717738,"teacher_disagreement_score":0.010676116,"about_ca_system_score_codex":0.002343449,"about_ca_system_score_gemma":0.0007658412,"threshold_uncertainty_score":0.035715163},"labels":[],"label_agreement":null},{"id":"W2492159292","doi":"10.4018/978-1-60960-625-1.ch002","title":"An Overview of Shallow and Deep Natural Language Processing for Ontology Learning","year":2011,"lang":"en","type":"book-chapter","venue":"IGI Global eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":26,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Athabasca University; Simon Fraser University","funders":"","keywords":"Computer science; Ontology; Natural language processing; Artificial intelligence; Dependency (UML); Deep learning; Task (project management); Semantic Web; Process (computing); Programming language; Engineering","score_opus":0.03124175699176025,"score_gpt":0.30990639736975645,"score_spread":0.2786646403779962,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2492159292","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0013093203,0.052791156,0.87508035,0.0020786244,0.000609711,0.00016903748,0.0006976827,0.0018867758,0.06537735],"genre_scores_gemma":[0.026784608,0.088013984,0.8168718,0.0016157588,0.00091284973,0.0005630185,0.0031307817,0.0010190428,0.061088104],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99938536,0.00013043094,0.00007218856,0.00010578003,0.00026507437,0.00004119006],"domain_scores_gemma":[0.99927324,0.0004555229,0.00003002686,0.00010745087,0.00011051813,0.000023213539],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008535067,0.0011232579,0.00068140257,0.002036206,0.0007420604,0.0029230856,0.0015673588,0.0014095415,0.020474983],"category_scores_gemma":[0.0018951189,0.00073319586,0.0012568748,0.00354648,0.0010840358,0.00528187,0.0016899976,0.0026146104,0.010270784],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000019558012,0.00007160945,0.0001989216,0.002587534,0.000048915528,0.0002151877,0.00048387365,0.0058970936,0.00488111,0.24199651,0.04827685,0.6953228],"study_design_scores_gemma":[0.000006756708,0.000023543094,0.00035280173,0.0007613163,0.000025060874,0.0004095865,0.0001290623,0.021845002,0.002934432,0.39261293,0.5808597,0.00003972259],"about_ca_topic_score_codex":0.001930894,"about_ca_topic_score_gemma":0.0037050066,"teacher_disagreement_score":0.020474983,"about_ca_system_score_codex":0.0014156165,"about_ca_system_score_gemma":0.0013554436,"threshold_uncertainty_score":0.06849557},"labels":[],"label_agreement":null},{"id":"W2492447297","doi":"10.4018/978-1-60566-766-9.ch014","title":"Machine Learning in Natural Language Processing","year":2010,"lang":"en","type":"book-chapter","venue":"IGI Global eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa; Agricultural Research Institute of Ontario","funders":"","keywords":"Computer science; Artificial intelligence; Natural language processing; Information extraction; Software deployment; Text processing; Natural (archaeology); Software engineering","score_opus":0.008130415791064697,"score_gpt":0.25641158434281264,"score_spread":0.24828116855174795,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2492447297","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0025849831,0.34790802,0.4061779,0.015868383,0.004150365,0.00021369242,0.0008483338,0.0016114668,0.22063686],"genre_scores_gemma":[0.08399552,0.34022117,0.41311347,0.008244864,0.0076042246,0.0009921021,0.0031305186,0.0007937301,0.14190432],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9987207,0.00046187945,0.00008755387,0.00019949024,0.000480008,0.000050292532],"domain_scores_gemma":[0.9981116,0.0014475057,0.00005770101,0.00020034004,0.00014249969,0.00004029728],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014436864,0.0010530434,0.0010367824,0.001786448,0.0006899284,0.0036372792,0.0013395177,0.001929246,0.013125718],"category_scores_gemma":[0.0032990559,0.0004303355,0.00053524994,0.003806018,0.0026647905,0.0051625366,0.0014982768,0.002941418,0.008067988],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000015442098,0.00005670003,0.00028994845,0.0015853645,0.000054611988,0.0002290138,0.00047611105,0.0038652755,0.0006797687,0.50426793,0.08972342,0.3987565],"study_design_scores_gemma":[0.000007419801,0.000018159964,0.00026073016,0.00060526613,0.0000111551935,0.00025721625,0.000115147384,0.004098278,0.0004100856,0.5524659,0.44172892,0.000021710022],"about_ca_topic_score_codex":0.0007403913,"about_ca_topic_score_gemma":0.0008112686,"teacher_disagreement_score":0.013125718,"about_ca_system_score_codex":0.0012388497,"about_ca_system_score_gemma":0.0012056356,"threshold_uncertainty_score":0.043909907},"labels":[],"label_agreement":null},{"id":"W2492585692","doi":"10.1007/978-3-642-30353-1","title":"Advances in Artificial Intelligence","year":2012,"lang":"en","type":"book","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa; Concordia University","funders":"","keywords":"Computer science; Artificial intelligence","score_opus":0.017557529096290415,"score_gpt":0.29480140081870504,"score_spread":0.2772438717224146,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2492585692","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.001533342,0.2173931,0.041197725,0.0075949784,0.010137701,0.000077902216,0.00030541338,0.00080017286,0.72095966],"genre_scores_gemma":[0.02052927,0.16774555,0.026503325,0.002705941,0.0054945988,0.00014132434,0.000695698,0.00046172558,0.7757226],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99933165,0.00010101088,0.000032614596,0.00008627083,0.0004025505,0.000045861503],"domain_scores_gemma":[0.9993418,0.00026360346,0.00003428802,0.00015597984,0.00013550807,0.00006884671],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007850481,0.0012173761,0.0013189832,0.0017886508,0.00064956694,0.003531601,0.0010737149,0.0010928181,0.052979957],"category_scores_gemma":[0.0017064697,0.00047608046,0.00047257802,0.0026894903,0.0013166048,0.004356419,0.0021091027,0.0027204677,0.03476092],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000025537242,0.000058450867,0.00013477259,0.0007395431,0.00002926547,0.00005188475,0.0001828356,0.0006460826,0.000939887,0.14949001,0.29683787,0.5508638],"study_design_scores_gemma":[0.0000052869923,0.000017508472,0.00022336356,0.0002579246,0.000010702613,0.000097924676,0.000048025973,0.0006413593,0.0002532016,0.054489046,0.94395,0.0000056404074],"about_ca_topic_score_codex":0.00046135666,"about_ca_topic_score_gemma":0.0010176644,"teacher_disagreement_score":0.052979957,"about_ca_system_score_codex":0.0008984926,"about_ca_system_score_gemma":0.001095237,"threshold_uncertainty_score":0.1772356},"labels":[],"label_agreement":null},{"id":"W2497444983","doi":"10.4018/978-1-60960-741-8.ch008","title":"Practical Programming for NLP","year":2012,"lang":"en","type":"book-chapter","venue":"IGI Global eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Python (programming language); Computer science; Artificial intelligence; Programming language; C programming language; Natural language programming; Programming paradigm; Natural language processing; Very high-level programming language; Field (mathematics); First-generation programming language; Inductive programming; Programming domain; Natural language; Universal Networking Language; Software; Mathematics","score_opus":0.034936156635203934,"score_gpt":0.3233560967596598,"score_spread":0.2884199401244559,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2497444983","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0010769586,0.006348481,0.7119475,0.005224424,0.0008295408,0.00015270807,0.00041412894,0.0037180542,0.27028823],"genre_scores_gemma":[0.033619437,0.016158534,0.6985023,0.003102684,0.0010211777,0.0008930792,0.0018847629,0.0031715417,0.24164648],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9992791,0.00022213589,0.000052648655,0.00013901823,0.00025994796,0.000047166726],"domain_scores_gemma":[0.99927217,0.00045002994,0.000021680007,0.00012211711,0.00010306531,0.000030937586],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00094880344,0.0010654406,0.0004662406,0.0008943203,0.0012181737,0.002942365,0.0014874098,0.0010375733,0.04288166],"category_scores_gemma":[0.0027309097,0.0006385135,0.00096456916,0.0017021114,0.0021761276,0.0056375735,0.0022741433,0.003952864,0.018117042],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000076275664,0.000022063035,0.00004970981,0.00034988835,0.00000786231,0.00008433229,0.00046280964,0.0012192415,0.00062361703,0.8359771,0.064864874,0.096330926],"study_design_scores_gemma":[0.0000051497645,0.0000054024986,0.00005039413,0.00016678228,0.0000032935175,0.00021457956,0.00006781634,0.0022282484,0.00037451313,0.380926,0.6159493,0.000008585547],"about_ca_topic_score_codex":0.00082118943,"about_ca_topic_score_gemma":0.0011018709,"teacher_disagreement_score":0.04288166,"about_ca_system_score_codex":0.001526726,"about_ca_system_score_gemma":0.0013459539,"threshold_uncertainty_score":0.14345348},"labels":[],"label_agreement":null},{"id":"W2498103754","doi":"10.1007/1-4020-4308-2_4","title":"Further Consequences of VP-Remnant Movement: Some Common Negation Structures in SLQZ","year":2006,"lang":"en","type":"book-chapter","venue":"Studies in natural language and linguistic theory","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Negation; Movement (music); Geology; Linguistics; Philosophy; Aesthetics","score_opus":0.01243672952516312,"score_gpt":0.28961578033094426,"score_spread":0.27717905080578115,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2498103754","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4662183,0.0015583907,0.11487581,0.010843062,0.00046937293,0.00016911354,0.00088503934,0.0008166913,0.40416414],"genre_scores_gemma":[0.9810879,0.00032598642,0.0069512106,0.0003038487,0.00016369054,0.00005455891,0.00028666106,0.0002652468,0.010560731],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9989955,0.00036361758,0.00006390302,0.00018908594,0.00019638271,0.00019142817],"domain_scores_gemma":[0.99800676,0.0012679992,0.00009714946,0.00032613397,0.00023232005,0.000069539856],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013054879,0.00081523234,0.0008470696,0.0013244502,0.0030669519,0.0034698925,0.0014206844,0.0023362224,0.011924748],"category_scores_gemma":[0.0036564087,0.000673592,0.0012047935,0.0018370548,0.006757476,0.012074148,0.0032528315,0.0046620616,0.0011260514],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000094337,0.000031296066,0.0008346927,0.000060648876,0.000013283954,0.001475722,0.0037104902,0.00037734912,0.0022582014,0.9801681,0.0028426128,0.008133126],"study_design_scores_gemma":[0.000048270016,0.000023265116,0.0014376744,0.000033294287,0.000027660304,0.0013509532,0.0024795,0.0035363582,0.0020373298,0.9806095,0.008379081,0.000037067057],"about_ca_topic_score_codex":0.0035850655,"about_ca_topic_score_gemma":0.002708928,"teacher_disagreement_score":0.011924748,"about_ca_system_score_codex":0.0012660815,"about_ca_system_score_gemma":0.00067522126,"threshold_uncertainty_score":0.039892256},"labels":[],"label_agreement":null},{"id":"W24988732","doi":"","title":"Évaluation du potentiel terminologique de candidats termes","year":2006,"lang":"fr","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Humanities; Computer science; Philosophy","score_opus":0.02543232463868689,"score_gpt":0.2627112773840772,"score_spread":0.2372789527453903,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W24988732","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.86380625,0.0033820453,0.11804525,0.00043023087,0.00016988836,0.00022392756,0.0021775134,0.0063038673,0.005461016],"genre_scores_gemma":[0.9004396,0.00081867614,0.08884055,0.000083462764,0.00009626259,0.0001730126,0.0056006494,0.00044540418,0.003502323],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99174374,0.003138819,0.00080954714,0.0014209452,0.0026532272,0.00023366945],"domain_scores_gemma":[0.9323096,0.05419665,0.0022596538,0.0028435027,0.0076392917,0.0007511648],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012160789,0.0009176907,0.0014328811,0.006768613,0.00078277645,0.0034016408,0.0014165909,0.0019373981,0.0025826013],"category_scores_gemma":[0.055689197,0.0004435919,0.0010722844,0.0036250113,0.0008683996,0.003359001,0.0011183774,0.0011897846,0.0016622221],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.006370709,0.0007034198,0.120007895,0.0013299787,0.0010956943,0.00040969686,0.0014891844,0.06358163,0.060609646,0.003929837,0.0039730733,0.7364993],"study_design_scores_gemma":[0.00031952423,0.0021352654,0.11390501,0.00015385528,0.0006455995,0.0010815155,0.0011514786,0.78580105,0.08101896,0.0033539624,0.01015459,0.00027915763],"about_ca_topic_score_codex":0.008965994,"about_ca_topic_score_gemma":0.0072869416,"teacher_disagreement_score":0.012160789,"about_ca_system_score_codex":0.0014490845,"about_ca_system_score_gemma":0.0013579167,"threshold_uncertainty_score":0.06431317},"labels":[],"label_agreement":null},{"id":"W2499142558","doi":"10.1002/9781119105664.part3","title":"Inflectional Morphology and Phrase Structure Variation","year":2015,"lang":"en","type":"other","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Morphology (biology); Variation (astronomy); Phrase; Linguistics; Geography; Geology; Paleontology; Astrophysics; Philosophy; Physics","score_opus":0.00804055697718365,"score_gpt":0.26114361714792766,"score_spread":0.253103060170744,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2499142558","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14171688,0.0061309636,0.40704846,0.003733524,0.00077706645,0.00012425038,0.0078682,0.009653501,0.42294717],"genre_scores_gemma":[0.72543025,0.003411761,0.092342325,0.00038717306,0.00053159543,0.000079071135,0.011628981,0.0040855757,0.16210319],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99942005,0.00014349475,0.000044213346,0.00016575048,0.00018473563,0.0000416826],"domain_scores_gemma":[0.998874,0.00039225974,0.00013517468,0.0003003541,0.00025156292,0.00004654613],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00050877983,0.0005230284,0.0003560966,0.0015260588,0.0005705968,0.0030147228,0.00067267905,0.00045459857,0.031033205],"category_scores_gemma":[0.0032988207,0.00028695038,0.00041218215,0.0027687792,0.0007248206,0.002454355,0.00146686,0.0010995127,0.012208369],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016563767,0.000055615274,0.0039694738,0.000331795,0.000043704582,0.00046801657,0.00088626787,0.0018534174,0.01606474,0.20292899,0.036093924,0.73713845],"study_design_scores_gemma":[0.00004104508,0.00008523902,0.03101673,0.00023970229,0.000102859165,0.0029059867,0.0007638951,0.03678565,0.023844784,0.5783987,0.32570788,0.000107610904],"about_ca_topic_score_codex":0.001294108,"about_ca_topic_score_gemma":0.0018698669,"teacher_disagreement_score":0.031033205,"about_ca_system_score_codex":0.000440369,"about_ca_system_score_gemma":0.00053559674,"threshold_uncertainty_score":0.10381645},"labels":[],"label_agreement":null},{"id":"W2499647683","doi":"10.1075/slcs.121.18ter","title":"Clause dependency relations in East Greenlandic Inuit","year":2010,"lang":"en","type":"book-chapter","venue":"Studies in language companion series","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Dependency (UML); Geography; History; Computer science; Artificial intelligence","score_opus":0.03421529722981083,"score_gpt":0.3131635633379276,"score_spread":0.2789482661081168,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2499647683","genre_codex":"empirical","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9911675,0.00021453788,0.00026146372,0.00010380639,0.0000030943786,0.0000076063984,0.000123401,0.0000074796862,0.00811121],"genre_scores_gemma":[0.9986866,0.00007880214,0.00011970529,0.000016054912,0.0000010779041,0.0000034822924,0.00008587034,0.000006142003,0.0010021841],"study_design_codex":"qualitative","study_design_gemma":"not_applicable","domain_scores_codex":[0.9997917,0.00005185168,0.00001687705,0.000067468485,0.00002740215,0.000044667497],"domain_scores_gemma":[0.9995301,0.0001978418,0.00013326107,0.00002690229,0.00007795668,0.000034104596],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00024098728,0.00021139056,0.00022076564,0.0011220387,0.0015398426,0.0010044479,0.00031250468,0.00018573066,0.005322217],"category_scores_gemma":[0.0010950388,0.00023086391,0.00007333221,0.0012604034,0.0010176842,0.0007893471,0.00085795915,0.00039945485,0.00023177921],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00059020263,0.00017221528,0.2541471,0.0005534337,0.00012315915,0.008709022,0.5809466,0.00045909715,0.057763074,0.030279616,0.0026030245,0.06365344],"study_design_scores_gemma":[0.000041145093,0.00009449046,0.7904211,0.00013244638,0.00011610383,0.0031958867,0.16433948,0.0014976764,0.007125158,0.0039190506,0.029049445,0.00006795347],"about_ca_topic_score_codex":0.24241315,"about_ca_topic_score_gemma":0.4156686,"teacher_disagreement_score":0.24241315,"about_ca_system_score_codex":0.0030447736,"about_ca_system_score_gemma":0.00077151373,"threshold_uncertainty_score":0.48200434},"labels":[],"label_agreement":null},{"id":"W2500777574","doi":"10.1007/978-1-4020-4519-6_41","title":"Analytic and Synthetic Sentences","year":2007,"lang":"en","type":"book-chapter","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"","keywords":"Natural language processing; Linguistics; Computer science; Artificial intelligence; Philosophy","score_opus":0.02339805899188916,"score_gpt":0.2714439752811004,"score_spread":0.24804591628921127,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2500777574","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029641392,0.0024400267,0.7051536,0.0032953508,0.0011771307,0.000502325,0.0065790843,0.005722386,0.24548872],"genre_scores_gemma":[0.3821325,0.002611035,0.47052637,0.00085208006,0.000799921,0.0007620589,0.015246573,0.0021688498,0.124900624],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99887115,0.0004938052,0.00007675338,0.00020883958,0.00031753257,0.000031793603],"domain_scores_gemma":[0.9966049,0.0022137268,0.00009114858,0.00042121764,0.00058218685,0.00008683559],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001024731,0.00081650267,0.00044233084,0.0012265525,0.0007835735,0.0024468375,0.00075197424,0.0006206908,0.03389971],"category_scores_gemma":[0.008920692,0.00036418086,0.00037928074,0.00093474035,0.0009164861,0.0030414448,0.0011138391,0.0011276363,0.008606377],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025203617,0.000120844576,0.0004054426,0.000995257,0.00003663894,0.0005053792,0.0019685137,0.005886884,0.018549709,0.5133413,0.11829885,0.3396392],"study_design_scores_gemma":[0.000054259104,0.00012292885,0.0006584179,0.0002729398,0.000046119643,0.0016070671,0.0014257011,0.06335827,0.026821082,0.42708328,0.47849038,0.0000595501],"about_ca_topic_score_codex":0.0004257194,"about_ca_topic_score_gemma":0.00061215454,"teacher_disagreement_score":0.03389971,"about_ca_system_score_codex":0.0005849378,"about_ca_system_score_gemma":0.0005247784,"threshold_uncertainty_score":0.113405764},"labels":[],"label_agreement":null},{"id":"W2501397611","doi":"","title":"Natural language processing techniques for the purpose of sentinel event information extraction","year":2012,"lang":"en","type":"dissertation","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"","keywords":"Computer science; Security token; Biomedical text mining; Artificial intelligence; Natural language processing; Parsing; Named-entity recognition; Information extraction; Text mining; F1 score; Event (particle physics); Domain (mathematical analysis); Scope (computer science); Task (project management); Programming language","score_opus":0.007830810632446958,"score_gpt":0.3138503835524668,"score_spread":0.30601957292001986,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2501397611","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0026797892,0.00024767683,0.97904724,0.0006547997,0.0001866228,0.00036284924,0.002420498,0.009880573,0.004519868],"genre_scores_gemma":[0.02354889,0.0004390071,0.966042,0.00038397455,0.00013604079,0.00034915018,0.0036538371,0.0009426614,0.0045044958],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.996784,0.00069110777,0.00054931256,0.0008161405,0.0010421064,0.00011734851],"domain_scores_gemma":[0.99000144,0.005424286,0.0012308838,0.0013034057,0.0019546424,0.000085335305],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004280805,0.0013013397,0.00073023536,0.0035711639,0.0009134476,0.0024968418,0.0014816024,0.00109633,0.014293914],"category_scores_gemma":[0.011600259,0.00090331724,0.0016519686,0.0031341324,0.0011872804,0.0040996997,0.0017887922,0.0031235325,0.014349624],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00027057732,0.00015331068,0.0024471576,0.0023810794,0.00016304872,0.0011233123,0.0019265144,0.007113982,0.11360381,0.09818414,0.0613243,0.7113087],"study_design_scores_gemma":[0.00011835514,0.00022353935,0.0029294828,0.00061676296,0.00025831646,0.0024515665,0.001183078,0.13681889,0.21133327,0.15105645,0.49278384,0.00022637914],"about_ca_topic_score_codex":0.0010460747,"about_ca_topic_score_gemma":0.00180512,"teacher_disagreement_score":0.014293914,"about_ca_system_score_codex":0.0009732117,"about_ca_system_score_gemma":0.0029255424,"threshold_uncertainty_score":0.047817886},"labels":[],"label_agreement":null},{"id":"W2501826610","doi":"10.1075/ubli.6","title":"Corpus-Based Perspectives in Linguistics","year":2007,"lang":"en","type":"book","venue":"Usage-based linguistic informatics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Corpus linguistics; Linguistics; Applied linguistics; Sociology; Computer science; Psychology; Philosophy","score_opus":0.016293178293732554,"score_gpt":0.27984819407704575,"score_spread":0.2635550157833132,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2501826610","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0029047949,0.25122944,0.24560606,0.04944511,0.0056478963,0.00010851212,0.0009084214,0.0008126334,0.44333708],"genre_scores_gemma":[0.20040508,0.3096171,0.29629624,0.014880506,0.017048407,0.0016017626,0.0038667435,0.0026627139,0.15362146],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9918759,0.005183136,0.00035994122,0.00075836637,0.0016285168,0.00019417205],"domain_scores_gemma":[0.9820613,0.015110628,0.0002837553,0.0012473307,0.0010259504,0.00027088283],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007629286,0.001332298,0.0014009859,0.013185978,0.0047712154,0.0182951,0.003002978,0.0033355625,0.009900685],"category_scores_gemma":[0.011421779,0.0010067712,0.0007601228,0.019794501,0.019712418,0.026369208,0.0037984394,0.007890913,0.0027151133],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000040951563,0.000006810303,0.00005332716,0.00018936957,0.000009072583,0.000040746712,0.0011807285,0.00021987672,0.000056005843,0.9684284,0.013651571,0.01615993],"study_design_scores_gemma":[0.000005339757,0.0000066434513,0.00014616738,0.00054569397,0.000007528196,0.00017705002,0.0011815397,0.0009949311,0.00011660671,0.6593397,0.3374673,0.000011368526],"about_ca_topic_score_codex":0.0047490974,"about_ca_topic_score_gemma":0.007522516,"teacher_disagreement_score":0.0182951,"about_ca_system_score_codex":0.007168773,"about_ca_system_score_gemma":0.0029249291,"threshold_uncertainty_score":0.052013338},"labels":[],"label_agreement":null},{"id":"W2502902171","doi":"10.7939/r3nx5z","title":"Grapheme-to-phoneme conversion and its application to transliteration","year":2011,"lang":"en","type":"article","venue":"University of Alberta Library","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Grapheme; Computer science; Pronunciation; Artificial intelligence; Natural language processing; Task (project management); Word (group theory); Discriminative model; Transliteration; Speech recognition; Sequence (biology); Machine translation","score_opus":0.007316081359181788,"score_gpt":0.18062282282129177,"score_spread":0.17330674146210998,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2502902171","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0048404955,0.0014727509,0.98759216,0.0004971693,0.00016007513,0.00003538685,0.00010074519,0.0009926152,0.004308621],"genre_scores_gemma":[0.2230802,0.005531245,0.751918,0.00042553525,0.00043757603,0.00019939264,0.0005100257,0.0010101771,0.016887756],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99937433,0.0001658963,0.00003318059,0.0002497034,0.00013186444,0.000044966087],"domain_scores_gemma":[0.9991616,0.0005010107,0.00007576273,0.00014839707,0.00009112351,0.00002217154],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00071064295,0.00074570446,0.00049609767,0.00078805746,0.00046381055,0.0013445735,0.0010450348,0.0011119028,0.004202988],"category_scores_gemma":[0.0030463443,0.00061059784,0.0013004612,0.0011332094,0.0010300296,0.0018449366,0.0012536221,0.0019889912,0.0025664603],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000083356776,0.000062122504,0.0009339563,0.0003007059,0.000074371965,0.00033403086,0.0005857945,0.26075286,0.012504453,0.34676933,0.008736947,0.3688621],"study_design_scores_gemma":[0.000007095307,0.000046250057,0.00044021747,0.00004408584,0.000020417177,0.00021574723,0.000041104555,0.8081115,0.007139826,0.15248571,0.031413056,0.00003488382],"about_ca_topic_score_codex":0.0033206851,"about_ca_topic_score_gemma":0.0026485918,"teacher_disagreement_score":0.004202988,"about_ca_system_score_codex":0.0008517262,"about_ca_system_score_gemma":0.0007477593,"threshold_uncertainty_score":0.014060438},"labels":[],"label_agreement":null},{"id":"W2505696146","doi":"10.1609/aaai.v30i1.9876","title":"What’s Hot in Human Language Technology: Highlights from NAACL HLT 2015","year":2016,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Data science; History","score_opus":0.039601484243117456,"score_gpt":0.3154644048457776,"score_spread":0.27586292060266016,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2505696146","genre_codex":"commentary","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017482206,0.27743796,0.010457463,0.6193146,0.023346404,0.00009634137,0.0041929223,0.0012708363,0.04640124],"genre_scores_gemma":[0.34671703,0.33668312,0.026380794,0.18055558,0.04818173,0.00047884646,0.018523194,0.002457485,0.040022165],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9897133,0.0029309066,0.0010670966,0.0012224643,0.004064208,0.0010021286],"domain_scores_gemma":[0.968739,0.0172014,0.0015763994,0.001072582,0.0073907864,0.0040198225],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.019804107,0.00063235266,0.0009094488,0.005197439,0.0026815708,0.015688187,0.0013616774,0.0040020263,0.005038441],"category_scores_gemma":[0.029734427,0.00054139795,0.0005544389,0.009596406,0.004902476,0.015664723,0.00649238,0.005899709,0.0023291525],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00050427933,0.00017871032,0.0050432784,0.002971135,0.00004685668,0.00039580165,0.0056391596,0.0006749187,0.0024907219,0.03294095,0.6320133,0.31710085],"study_design_scores_gemma":[0.00003543971,0.00011314164,0.010267614,0.0017030587,0.000029301053,0.00042628372,0.006116621,0.00066725624,0.0019604187,0.016468639,0.96210724,0.0001050166],"about_ca_topic_score_codex":0.013048469,"about_ca_topic_score_gemma":0.0256537,"teacher_disagreement_score":0.019804107,"about_ca_system_score_codex":0.0064003724,"about_ca_system_score_gemma":0.005661436,"threshold_uncertainty_score":0.104735374},"labels":[],"label_agreement":null},{"id":"W2505721561","doi":"10.1075/hsm.12.13koc","title":"Revisiting a translation effect in an oral language","year":2011,"lang":"en","type":"book-chapter","venue":"Hamburg studies in multilingualism","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Translation (biology); Linguistics; Psychology; Computer science; Natural language processing; History; Philosophy; Chemistry","score_opus":0.08427736040130414,"score_gpt":0.38767678260119676,"score_spread":0.30339942219989263,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2505721561","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.71550286,0.0017107581,0.016911998,0.0028765397,0.00048730246,0.000045764406,0.00017417163,0.00025181912,0.26203874],"genre_scores_gemma":[0.985171,0.00049577554,0.002929882,0.00035385083,0.00012264801,0.000014005993,0.0000830406,0.000114800896,0.010715009],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.99963534,0.000102138394,0.000020763442,0.00008618643,0.0001286076,0.000027030474],"domain_scores_gemma":[0.9985802,0.0008908339,0.00009501947,0.00019814866,0.00020673288,0.0000290261],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006225959,0.00031004392,0.00022979267,0.00043374556,0.0006362773,0.0015490169,0.00036473296,0.00028097632,0.011099616],"category_scores_gemma":[0.0028460585,0.00014663082,0.00019803876,0.0004047131,0.003276575,0.0026506707,0.0013858131,0.0010564644,0.0011233686],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00071154785,0.00020800882,0.019054696,0.0015717726,0.000055897264,0.0016869254,0.10215612,0.0005026289,0.1033487,0.42793235,0.0070793205,0.33569205],"study_design_scores_gemma":[0.0003633084,0.0018735612,0.15934727,0.0010108608,0.00045446656,0.0034950664,0.08318108,0.007971207,0.141739,0.2541309,0.34619895,0.00023436945],"about_ca_topic_score_codex":0.002563235,"about_ca_topic_score_gemma":0.0032391516,"teacher_disagreement_score":0.011099616,"about_ca_system_score_codex":0.000695895,"about_ca_system_score_gemma":0.0006823299,"threshold_uncertainty_score":0.037131906},"labels":[],"label_agreement":null},{"id":"W2507532992","doi":"10.18653/v1/w16-1817","title":"A Word Embedding Approach to Identifying Verb-Noun Idiomatic Combinations","year":2016,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of New Brunswick","funders":"Natural Sciences and Engineering Research Council of Canada; New Brunswick Innovation Foundation","keywords":"Noun; Verb; Linguistics; Computer science; Natural language processing; Artificial intelligence; Embedding; Word (group theory); Word embedding; Philosophy","score_opus":0.025562388564480558,"score_gpt":0.30304292863560683,"score_spread":0.27748054007112627,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2507532992","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.047144096,0.0005245176,0.9450674,0.0003004102,0.00013627084,0.0003117974,0.0012032497,0.0021371886,0.0031751117],"genre_scores_gemma":[0.24852625,0.00035138405,0.7428104,0.00015304443,0.000099216966,0.0003455428,0.003319749,0.0002661286,0.0041282927],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987587,0.0003815201,0.0001485109,0.00042218994,0.00021942711,0.00006977653],"domain_scores_gemma":[0.99788386,0.0008695151,0.00024291774,0.00036832353,0.0005592031,0.000076131764],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007973771,0.0011156545,0.00058571633,0.002957751,0.00066290086,0.0011733449,0.0009824749,0.0009535617,0.0024033],"category_scores_gemma":[0.00301738,0.000362565,0.0007532789,0.0023871856,0.00067599973,0.0029059018,0.0013532154,0.001436659,0.0013439821],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025303176,0.0006369422,0.0068745525,0.0005048293,0.00024303606,0.00031838394,0.0008117915,0.01205916,0.046371512,0.024475474,0.009930241,0.8975211],"study_design_scores_gemma":[0.00007460175,0.00040706378,0.009870641,0.0001343718,0.00020990649,0.0014064703,0.001116478,0.84365743,0.040522903,0.07553824,0.02690362,0.00015833849],"about_ca_topic_score_codex":0.0015718126,"about_ca_topic_score_gemma":0.004171571,"teacher_disagreement_score":0.002957751,"about_ca_system_score_codex":0.00038893402,"about_ca_system_score_gemma":0.00089755486,"threshold_uncertainty_score":0.008039832},"labels":[],"label_agreement":null},{"id":"W2508129077","doi":"10.20381/ruor-4638","title":"An Unsupervised Approach to Detecting and Correcting Errors in Text","year":2011,"lang":"en","type":"dissertation","venue":"Library and Archives Canada (Government of Canada)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Artificial intelligence; Natural language processing; Error detection and correction; Word (group theory); Spelling; Verb; Punctuation; Algorithm; Mathematics; Linguistics","score_opus":0.0052972131944226996,"score_gpt":0.17575992314797487,"score_spread":0.17046270995355217,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2508129077","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0072376803,0.00012894331,0.98861545,0.00027972684,0.000047450292,0.0002236238,0.00015963378,0.0019583518,0.001349077],"genre_scores_gemma":[0.083633535,0.00027328436,0.9094364,0.00026740166,0.00013675842,0.00038899537,0.0007566409,0.0005039721,0.004602963],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9890036,0.0029844109,0.0009855777,0.003193162,0.0035474552,0.0002857947],"domain_scores_gemma":[0.9672687,0.013116075,0.0042503783,0.007108479,0.007947802,0.00030863044],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0040814783,0.0013300134,0.0012824117,0.005293638,0.001517333,0.0031816063,0.0037812414,0.0019569192,0.0019207397],"category_scores_gemma":[0.02150302,0.00082079327,0.0018268046,0.0030113026,0.0031188072,0.004251657,0.0023909982,0.001897944,0.0023745392],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021365912,0.0003865247,0.010255105,0.0010109453,0.00034186925,0.00050082995,0.0023637991,0.049283817,0.074568145,0.03867341,0.008611588,0.8137904],"study_design_scores_gemma":[0.00005086121,0.00035238377,0.009306198,0.0002718641,0.00023722158,0.0017609714,0.0013460926,0.70195174,0.13893414,0.098817974,0.04674817,0.00022252262],"about_ca_topic_score_codex":0.0029685928,"about_ca_topic_score_gemma":0.0044685686,"teacher_disagreement_score":0.005293638,"about_ca_system_score_codex":0.0014190632,"about_ca_system_score_gemma":0.0030626843,"threshold_uncertainty_score":0.021585166},"labels":[],"label_agreement":null},{"id":"W2508414753","doi":"10.1515/ling-2016-0019","title":"The lexicon in Functional Discourse Grammar: Theory, typology, description","year":2016,"lang":"en","type":"article","venue":"Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Lethbridge","funders":"","keywords":"Lexeme; Lexicon; Linguistics; Computer science; Morpheme; Grammar; Natural language processing; Artificial intelligence; Lexical item; Frame (networking); Philosophy","score_opus":0.01899950790440063,"score_gpt":0.28098725840908806,"score_spread":0.2619877505046874,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2508414753","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.043378536,0.0082105035,0.5239718,0.017762853,0.00063666695,0.00017204143,0.0006155652,0.00063854863,0.40461347],"genre_scores_gemma":[0.91952425,0.0032041774,0.06030967,0.0011792907,0.00048431003,0.0003984776,0.00054994726,0.0002755597,0.014074402],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9977138,0.0015486062,0.000117012874,0.00019761364,0.0002682383,0.00015479558],"domain_scores_gemma":[0.9973494,0.0017703482,0.00017142283,0.0003660904,0.00025399608,0.0000887477],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024541565,0.00057706836,0.0006303738,0.004965008,0.0018303118,0.0077474355,0.0014374931,0.0021175132,0.0055377977],"category_scores_gemma":[0.0044628894,0.0003404355,0.00067608996,0.004372642,0.015264009,0.011368363,0.002888184,0.0020531397,0.000713696],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000013170867,0.0000011679741,0.00006457016,0.000009119753,6.735732e-7,0.000019132374,0.00033972439,0.0001534474,0.000019717518,0.9975068,0.0003778418,0.001506425],"study_design_scores_gemma":[0.0000038112391,0.0000042967336,0.00008290543,0.000045937893,0.0000023432178,0.00012429568,0.00070644607,0.0028981,0.00009846944,0.97569555,0.020331336,0.0000064227042],"about_ca_topic_score_codex":0.0030403736,"about_ca_topic_score_gemma":0.0017012252,"teacher_disagreement_score":0.0077474355,"about_ca_system_score_codex":0.004887666,"about_ca_system_score_gemma":0.0016851199,"threshold_uncertainty_score":0.035462677},"labels":[],"label_agreement":null},{"id":"W2509251152","doi":"10.18653/v1/w16-2370","title":"BAD LUC$@$WMT 2016: a Bilingual Document Alignment Platform Based on Lucene","year":2016,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"Fonds de recherche du Québec – Nature et technologies","keywords":"Computer science; Task (project management); Information retrieval; Natural language processing; Heuristic; Artificial intelligence; World Wide Web","score_opus":0.010974859221806086,"score_gpt":0.26708258817819597,"score_spread":0.2561077289563899,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2509251152","genre_codex":"software","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":"software","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11291391,0.0019632298,0.2637239,0.0010955202,0.0012336591,0.0019018787,0.084796146,0.5005349,0.031836927],"genre_scores_gemma":[0.2332027,0.000430289,0.40123686,0.0006316502,0.00027352083,0.0014760375,0.31514937,0.02676586,0.020833727],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.995133,0.0013502407,0.0004319145,0.0016443284,0.00094856037,0.00049199734],"domain_scores_gemma":[0.99549305,0.00086665666,0.0002157423,0.0018445983,0.0011101719,0.00046975503],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005151579,0.0023006322,0.0016483479,0.002869365,0.0019069132,0.0023617754,0.0027355175,0.0012815625,0.02558825],"category_scores_gemma":[0.010106488,0.0013594099,0.001458482,0.0029762364,0.00090595044,0.006791395,0.005246431,0.002503587,0.030929044],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.006243238,0.0014693062,0.005236063,0.0025567617,0.00072226353,0.0013943294,0.0018963385,0.009424642,0.1177966,0.01323431,0.46742848,0.3725977],"study_design_scores_gemma":[0.0019053934,0.0026835832,0.015235838,0.00032585132,0.00044022265,0.0018839522,0.0014556283,0.13958259,0.25686082,0.017859396,0.5608222,0.00094453484],"about_ca_topic_score_codex":0.010877561,"about_ca_topic_score_gemma":0.012492893,"teacher_disagreement_score":0.02558825,"about_ca_system_score_codex":0.001161129,"about_ca_system_score_gemma":0.0031890129,"threshold_uncertainty_score":0.08560121},"labels":[],"label_agreement":null},{"id":"W2509440722","doi":"10.18653/v1/w16-2005","title":"Morphological Reinflection via Discriminative String Transduction","year":2016,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Innovates","keywords":"Discriminative model; Heuristics; Computer science; Context (archaeology); String (physics); Task (project management); Transduction (biophysics); Natural language processing; Artificial intelligence; Mathematics; Biology; Engineering","score_opus":0.018863406226743675,"score_gpt":0.27354343792731006,"score_spread":0.2546800317005664,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2509440722","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0697414,0.00058010913,0.8413986,0.00075880997,0.0008989658,0.0004415092,0.0021126596,0.0635392,0.020528762],"genre_scores_gemma":[0.44087932,0.00056128873,0.5154118,0.0012294587,0.00025685152,0.00020097916,0.011344201,0.007888567,0.02222753],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99774194,0.00029057587,0.00019476721,0.0008291081,0.00060230866,0.00034129174],"domain_scores_gemma":[0.995647,0.0008463652,0.00020105905,0.002525793,0.0005994792,0.0001802763],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015967648,0.0021659755,0.0019579607,0.0021191966,0.0013238796,0.0025841873,0.003424959,0.002402117,0.028800106],"category_scores_gemma":[0.004832615,0.0008370712,0.0017946783,0.0019799618,0.0018188467,0.005141162,0.0073971087,0.0037276398,0.019631492],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00062310364,0.00046065237,0.0018232791,0.0007166935,0.00015178735,0.001068401,0.00042080999,0.014731903,0.1014748,0.028046465,0.03697617,0.81350595],"study_design_scores_gemma":[0.00020102653,0.00082219427,0.0037974578,0.00022179617,0.00030584616,0.004884586,0.0008735876,0.40424657,0.3115257,0.14865589,0.124131806,0.0003334897],"about_ca_topic_score_codex":0.0011651985,"about_ca_topic_score_gemma":0.0029195787,"teacher_disagreement_score":0.028800106,"about_ca_system_score_codex":0.0005431719,"about_ca_system_score_gemma":0.0012661219,"threshold_uncertainty_score":0.09634596},"labels":[],"label_agreement":null},{"id":"W2510506392","doi":"","title":"On referring expressions in query answering over first order knowledge bases","year":2016,"lang":"en","type":"article","venue":"Principles of Knowledge Representation and Reasoning","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Conjunctive query; Context (archaeology); Noun phrase; Knowledge base; Description logic; Expression (computer science); Class (philosophy); Natural language processing; Artificial intelligence; Information retrieval; Relational database; Programming language; Noun","score_opus":0.03158008085999161,"score_gpt":0.3247805165416627,"score_spread":0.2932004356816711,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2510506392","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009615632,0.0020173928,0.97463596,0.0029338803,0.00008012961,0.00016594742,0.00024092144,0.00067586906,0.0096341185],"genre_scores_gemma":[0.20613343,0.004863232,0.775047,0.0023388849,0.00070044596,0.0006482262,0.001306939,0.0007831168,0.008178764],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9836865,0.0075639742,0.0013371967,0.0022042198,0.004093744,0.0011144293],"domain_scores_gemma":[0.96817833,0.025656391,0.00092509796,0.0031366118,0.0017795985,0.00032396044],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.016115434,0.0012395268,0.001865307,0.004518604,0.0035481162,0.009089249,0.0040885513,0.002971215,0.006070493],"category_scores_gemma":[0.03761599,0.0017831697,0.0035221782,0.009999847,0.01129345,0.029029356,0.008340033,0.0059994366,0.0016285123],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000674635,0.00004525032,0.0004057155,0.00020218668,0.00003258503,0.00029532227,0.0019257778,0.011029443,0.000922454,0.9588526,0.0019763776,0.02424487],"study_design_scores_gemma":[0.000023452309,0.000041142306,0.00022439545,0.00010411292,0.0000665918,0.00023806083,0.0004075039,0.057257786,0.0021437875,0.9257613,0.013680813,0.000051125186],"about_ca_topic_score_codex":0.015041712,"about_ca_topic_score_gemma":0.008222971,"teacher_disagreement_score":0.016115434,"about_ca_system_score_codex":0.0060941647,"about_ca_system_score_gemma":0.0023418674,"threshold_uncertainty_score":0.08522761},"labels":[],"label_agreement":null},{"id":"W2512726229","doi":"10.3765/bls.v28i1.3833","title":"Anaphoric R-Expressions as Bound Variables","year":2002,"lang":"en","type":"article","venue":"Proceedings of the Annual Meeting of the Berkeley Linguistics Society","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Upper and lower bounds; Combinatorics; Mathematics; Linguistics; Philosophy; Mathematical analysis","score_opus":0.011954709330198111,"score_gpt":0.24685501855118963,"score_spread":0.23490030922099153,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2512726229","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13427743,0.0021385867,0.5470935,0.0024779802,0.00032492128,0.00014582704,0.0005052667,0.0021205358,0.310916],"genre_scores_gemma":[0.89846736,0.0011500554,0.06205225,0.00053549994,0.00026385265,0.00007898078,0.00041760434,0.00097300217,0.03606153],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9968919,0.0013257,0.00015623713,0.00079179584,0.0006162191,0.00021813149],"domain_scores_gemma":[0.9969759,0.0016247738,0.00027093297,0.0007698526,0.00029493164,0.0000635392],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017396292,0.00062876,0.000681986,0.0014140017,0.002589824,0.0065117925,0.0015262738,0.0015160367,0.015615346],"category_scores_gemma":[0.0048561497,0.001054254,0.00061487005,0.0019487947,0.005286023,0.012392372,0.0044147843,0.0030345875,0.0029820958],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000061979226,0.0000137679535,0.00042290497,0.00013668872,0.000012533259,0.0005198968,0.0057258722,0.00036625558,0.003965789,0.9737916,0.0014895031,0.013493048],"study_design_scores_gemma":[0.00008770553,0.00006038257,0.0017632461,0.00024336684,0.00020419566,0.0028737576,0.005839802,0.01064239,0.032945137,0.7725835,0.17263204,0.0001245013],"about_ca_topic_score_codex":0.0014158093,"about_ca_topic_score_gemma":0.0011680824,"teacher_disagreement_score":0.015615346,"about_ca_system_score_codex":0.0013156177,"about_ca_system_score_gemma":0.0007203831,"threshold_uncertainty_score":0.052238524},"labels":[],"label_agreement":null},{"id":"W2513202451","doi":"10.18653/v1/w16-2317","title":"NRC Russian-English Machine Translation System for WMT 2016","year":2016,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Machine translation; Lemmatisation; Computer science; Natural language processing; Artificial intelligence; Phrase; Task (project management); Example-based machine translation; Machine translation software usability; Word (group theory); Linguistics; Engineering","score_opus":0.012114956006793484,"score_gpt":0.24861856763183726,"score_spread":0.23650361162504377,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2513202451","genre_codex":"software","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16316643,0.0025090985,0.23015839,0.0046037775,0.007366117,0.005207255,0.15194933,0.30467287,0.13036665],"genre_scores_gemma":[0.15732768,0.0006242999,0.38852218,0.0006648053,0.00048819341,0.0024650684,0.3526025,0.013417169,0.08388807],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99676394,0.0006470043,0.0003046916,0.00080429733,0.0011528181,0.00032729225],"domain_scores_gemma":[0.9956709,0.00023899208,0.00017572645,0.0011232457,0.0021987457,0.0005922514],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005088549,0.001944631,0.002036019,0.0018601613,0.0020843241,0.0019728022,0.0023581502,0.0015109414,0.020593328],"category_scores_gemma":[0.006745418,0.0007086448,0.0010078067,0.0018271741,0.00053297327,0.002700791,0.0029832388,0.002348918,0.04114412],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010080958,0.0008102904,0.0021986132,0.0010191675,0.00014459294,0.0012357463,0.0012015715,0.004322099,0.062240243,0.0076902336,0.6076336,0.31049576],"study_design_scores_gemma":[0.000679701,0.00109122,0.008246542,0.0002040342,0.00012669073,0.0024510666,0.0006443035,0.050242554,0.11269889,0.0070525045,0.81627196,0.00029054732],"about_ca_topic_score_codex":0.00855725,"about_ca_topic_score_gemma":0.011390223,"teacher_disagreement_score":0.020593328,"about_ca_system_score_codex":0.002001075,"about_ca_system_score_gemma":0.0049797054,"threshold_uncertainty_score":0.068891525},"labels":[],"label_agreement":null},{"id":"W2513442633","doi":"10.7202/1036952ar","title":"Multilinguïsation des systèmes traitant des sous-langages","year":2016,"lang":"fr","type":"article","venue":"TTR traduction terminologie rédaction","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Philosophy; Humanities","score_opus":0.09094646430951832,"score_gpt":0.3220661154329144,"score_spread":0.2311196511233961,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2513442633","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15582557,0.0007257672,0.8226222,0.00034860204,0.00014729802,0.00019227744,0.0010171088,0.015220904,0.0039003452],"genre_scores_gemma":[0.4527126,0.0004074675,0.5318595,0.00015112359,0.00006879451,0.00029459945,0.002757475,0.0024120018,0.009336378],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9965461,0.0012211851,0.0003000868,0.00092929846,0.0008558042,0.00014755549],"domain_scores_gemma":[0.9910287,0.0051276432,0.00054026826,0.0013353226,0.001833988,0.00013403143],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030173613,0.0012384664,0.00096148724,0.0013625771,0.000854005,0.0023831371,0.001089484,0.0010777074,0.0041576116],"category_scores_gemma":[0.013370777,0.0005996741,0.0011909804,0.0009128699,0.0008556475,0.002582918,0.0013716912,0.0013860196,0.0025895047],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012447325,0.00025795848,0.0125143975,0.0015343755,0.0003762299,0.0010454977,0.0063046594,0.033074945,0.29186967,0.007895381,0.005399171,0.638483],"study_design_scores_gemma":[0.000105782994,0.0008334113,0.017249849,0.00020640674,0.0005031933,0.0013055514,0.002346658,0.54028267,0.37214062,0.012508668,0.052275345,0.00024189202],"about_ca_topic_score_codex":0.008208564,"about_ca_topic_score_gemma":0.0074132173,"teacher_disagreement_score":0.008208564,"about_ca_system_score_codex":0.0010038262,"about_ca_system_score_gemma":0.0015176717,"threshold_uncertainty_score":0.0163216},"labels":[],"label_agreement":null},{"id":"W2514277269","doi":"","title":"Automatic supervised thesauri construction with roget's thesaurus","year":2012,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Thesaurus; Computer science; Natural language processing; Artificial intelligence; Information retrieval; Measure (data warehouse); Word (group theory); Data mining; Linguistics","score_opus":0.009785524150740084,"score_gpt":0.2342934763195165,"score_spread":0.22450795216877642,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2514277269","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07353848,0.0015044272,0.847865,0.0005238822,0.00062055106,0.0022141156,0.003307479,0.056691207,0.013734881],"genre_scores_gemma":[0.09718618,0.00036851864,0.8813304,0.0002119745,0.00009492408,0.0007855199,0.010933049,0.0027273737,0.0063621714],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99439853,0.0015382577,0.000625953,0.002193664,0.0010666735,0.00017703937],"domain_scores_gemma":[0.9900383,0.0035039748,0.00079109735,0.0023889998,0.0030572736,0.00022042323],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004181372,0.0015859613,0.0019874424,0.006508684,0.0020509984,0.003640889,0.00351114,0.0018007245,0.008484702],"category_scores_gemma":[0.019005448,0.0014919249,0.002711957,0.0037525017,0.0017671846,0.007222646,0.0046523847,0.003324614,0.0081227645],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025739544,0.0002886662,0.0031947396,0.0012445614,0.00020761848,0.00036132932,0.002220849,0.0064873416,0.03814633,0.007932588,0.033380568,0.906278],"study_design_scores_gemma":[0.00041472176,0.0006884153,0.015210556,0.001037856,0.0005668334,0.0025504208,0.0042285463,0.500889,0.12003125,0.039084632,0.31474587,0.0005517416],"about_ca_topic_score_codex":0.0067810817,"about_ca_topic_score_gemma":0.011920727,"teacher_disagreement_score":0.008484702,"about_ca_system_score_codex":0.0018613207,"about_ca_system_score_gemma":0.003229054,"threshold_uncertainty_score":0.02838415},"labels":[],"label_agreement":null},{"id":"W2515353635","doi":"10.18653/v1/k16-2022","title":"Discourse Relation Sense Classification Systems for CoNLL-2016 Shared Task","year":2016,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"National Key Research and Development Program of China; Atomic Energy of Canada Limited; National Natural Science Foundation of China; University of Pennsylvania","keywords":"Computer science; Task (project management); Natural language processing; Discriminative model; Artificial intelligence; Relation (database); Beijing; Linguistics; China; Data mining","score_opus":0.025975083351737108,"score_gpt":0.29967719604666676,"score_spread":0.27370211269492967,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2515353635","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.40373555,0.010152057,0.4239488,0.0040500932,0.004284467,0.0038797867,0.025903804,0.08345774,0.0405877],"genre_scores_gemma":[0.59495664,0.000956102,0.32316333,0.00077332,0.000628272,0.002097944,0.060131796,0.0012097447,0.016082957],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9947923,0.001434545,0.00058254757,0.0018573655,0.0009110787,0.00042211765],"domain_scores_gemma":[0.99425936,0.0017207895,0.00036228306,0.0016138819,0.0016183662,0.00042534704],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0050953757,0.00266458,0.0016200428,0.0039377986,0.0022966047,0.002282588,0.0026069095,0.0022661444,0.0074226474],"category_scores_gemma":[0.010815882,0.00043539665,0.0012732726,0.002228974,0.0006578756,0.005550954,0.0046563144,0.0034192887,0.0059280535],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013256969,0.0010271035,0.0036304789,0.0009131328,0.00020037853,0.000496663,0.0011666444,0.0047608833,0.037084717,0.00478129,0.07924585,0.8653672],"study_design_scores_gemma":[0.000923137,0.002019565,0.017047472,0.00042635296,0.0006033158,0.000999732,0.004658828,0.72398293,0.10924475,0.02892797,0.11076026,0.00040563953],"about_ca_topic_score_codex":0.006933751,"about_ca_topic_score_gemma":0.011874027,"teacher_disagreement_score":0.0074226474,"about_ca_system_score_codex":0.0017891754,"about_ca_system_score_gemma":0.0026683258,"threshold_uncertainty_score":0.02694726},"labels":[],"label_agreement":null},{"id":"W2517393733","doi":"10.1075/btl.126.10gia","title":"Computer science and translation","year":2016,"lang":"en","type":"book-chapter","venue":"Benjamins translation library","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Translation (biology); Computer science; Biology; Genetics","score_opus":0.021349440335713243,"score_gpt":0.2341395288439328,"score_spread":0.21279008850821957,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2517393733","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0016924145,0.17853323,0.05392002,0.03312921,0.00830426,0.00008903292,0.00022390454,0.0006721202,0.7234359],"genre_scores_gemma":[0.11426689,0.277499,0.059613205,0.018512286,0.012377395,0.0006116575,0.0009965779,0.0014352727,0.5146878],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99789715,0.0008340871,0.00014998618,0.00037574643,0.00061454764,0.000128444],"domain_scores_gemma":[0.9975599,0.0017073561,0.00007060615,0.00037606386,0.00023498065,0.000051046176],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013990053,0.0010032448,0.0007593214,0.0030188921,0.002286679,0.006776372,0.0011141166,0.0027255113,0.021749288],"category_scores_gemma":[0.004933704,0.00041321764,0.000585548,0.004148337,0.009902503,0.008796861,0.0024703697,0.0044021104,0.011298174],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000007470929,0.000008273777,0.00003069991,0.00031979822,0.000007433957,0.000057669255,0.0007842084,0.0002820544,0.00016788114,0.8760501,0.056497645,0.065786846],"study_design_scores_gemma":[0.000004197499,0.000006702673,0.00006394665,0.00025066696,0.0000025365282,0.00013163904,0.00016629312,0.00025670644,0.00019223838,0.38659,0.6123274,0.000007740656],"about_ca_topic_score_codex":0.0018504135,"about_ca_topic_score_gemma":0.0013156878,"teacher_disagreement_score":0.021749288,"about_ca_system_score_codex":0.004127934,"about_ca_system_score_gemma":0.0022945493,"threshold_uncertainty_score":0.072758615},"labels":[],"label_agreement":null},{"id":"W2517472350","doi":"","title":"Proceedings of Human Language Technologies: The 2009 Annual Conference of the North American Chapter of the Association for Computational Linguistics, Companion Volume: Student Research Workshop and Doctoral Consortium","year":2009,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Presentation (obstetrics); Graduate students; Library science; Work (physics); Computational linguistics; Computer science; Field (mathematics); Applied linguistics; Medical education; Engineering ethics; Engineering; Artificial intelligence; Medicine; Linguistics","score_opus":0.04633044714834728,"score_gpt":0.3621686348826654,"score_spread":0.3158381877343181,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2517472350","genre_codex":"editorial","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023662096,0.12280737,0.16168103,0.22364835,0.2356003,0.0024052158,0.011675924,0.0076870294,0.21083269],"genre_scores_gemma":[0.047288552,0.05710444,0.08730626,0.016544957,0.025479458,0.0018296855,0.02275754,0.0047496166,0.7369395],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99415094,0.0023908203,0.00042421362,0.00084264204,0.0017819942,0.0004094891],"domain_scores_gemma":[0.9859038,0.0040098014,0.00041622133,0.0010600201,0.0059136422,0.0026966606],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013561812,0.001176956,0.0018907252,0.0022937593,0.002925004,0.010166562,0.0026869718,0.0032943569,0.06797499],"category_scores_gemma":[0.013931361,0.00080490194,0.0010238469,0.001572171,0.0024306388,0.008530983,0.0043360842,0.004801896,0.0329698],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009901389,0.00009038551,0.0004125606,0.00025212782,0.000021060834,0.00011763281,0.00051159924,0.000089071065,0.0012349458,0.0020312504,0.92926466,0.06587567],"study_design_scores_gemma":[0.000031956315,0.00006155349,0.001498009,0.0004781736,0.000029357838,0.00028921754,0.001175827,0.0009861941,0.0008910788,0.0038598767,0.99064684,0.000051903273],"about_ca_topic_score_codex":0.0073133525,"about_ca_topic_score_gemma":0.019302413,"teacher_disagreement_score":0.06797499,"about_ca_system_score_codex":0.0033176234,"about_ca_system_score_gemma":0.007795411,"threshold_uncertainty_score":0.22739899},"labels":[],"label_agreement":null},{"id":"W2521376967","doi":"10.5539/elt.v9n10p133","title":"Effect of Alignment on Text Cohesion in the Continuation Task","year":2016,"lang":"en","type":"article","venue":"English Language Teaching","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"National Social Science Fund of China","keywords":"Cohesion (chemistry); Continuation; Psychology; Linguistics; Task (project management); Mathematics education; Computer science; Chemistry","score_opus":0.004272409237107298,"score_gpt":0.26204869625028293,"score_spread":0.2577762870131756,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2521376967","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9967763,0.00005714671,0.0015759027,0.000038716105,0.000016252614,0.000041467803,0.000025867164,0.00006436352,0.0014040865],"genre_scores_gemma":[0.9966516,0.00003852149,0.002622412,0.00001466479,0.00001258427,0.00010804897,0.000049503487,0.000030347546,0.00047221515],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99554235,0.002512429,0.00040527707,0.0006813428,0.0006345101,0.00022412861],"domain_scores_gemma":[0.91164273,0.07337408,0.0067766747,0.002720114,0.0020172102,0.0034691591],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0038697019,0.0005273035,0.00059072225,0.0004859116,0.0006943855,0.0015779706,0.00055239943,0.00056681794,0.003917982],"category_scores_gemma":[0.058280118,0.00027572564,0.00025559985,0.00042742092,0.0006250047,0.0013795905,0.0018592762,0.0007237497,0.00057891966],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.023240507,0.005195753,0.20709793,0.0014301961,0.00027217515,0.0009131,0.030554222,0.0023281497,0.36027163,0.0012447998,0.0016283295,0.36582327],"study_design_scores_gemma":[0.0010660802,0.021921357,0.8849355,0.0002214218,0.0005018695,0.0006902523,0.010664544,0.008849468,0.06195391,0.0033810851,0.0056406073,0.00017384652],"about_ca_topic_score_codex":0.00043616284,"about_ca_topic_score_gemma":0.0005156885,"teacher_disagreement_score":0.003917982,"about_ca_system_score_codex":0.00028234563,"about_ca_system_score_gemma":0.0005797363,"threshold_uncertainty_score":0.020465136},"labels":[],"label_agreement":null},{"id":"W2522406268","doi":"","title":"TIME in a semantically-annotated corpus of Canadian English","year":2014,"lang":"en","type":"article","venue":"Lirias (KU Leuven)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Linguistics; Natural language processing; Computer science; Artificial intelligence; History","score_opus":0.0072820594318619425,"score_gpt":0.21611229943049218,"score_spread":0.20883023999863023,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2522406268","genre_codex":"dataset","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1900672,0.008203928,0.005564484,0.0029157063,0.00093281106,0.00019764596,0.6920832,0.0017746638,0.09826039],"genre_scores_gemma":[0.48601335,0.0050630826,0.012082682,0.0003649002,0.00021961944,0.00027411804,0.45660833,0.0013073424,0.038066443],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9984469,0.00020078094,0.00012156096,0.0003637121,0.0006379637,0.00022912111],"domain_scores_gemma":[0.99044913,0.003524973,0.00045390535,0.0004523896,0.004805514,0.00031414386],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00079777226,0.00081505603,0.0006204652,0.007530933,0.0040425486,0.0027245341,0.0014524749,0.00078951713,0.014303936],"category_scores_gemma":[0.008784135,0.0003634052,0.00045025916,0.016767541,0.0014433387,0.0016735739,0.0009222202,0.0009604622,0.00311702],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0025711479,0.00018198145,0.027171073,0.0075124046,0.00017983216,0.0035831442,0.018436277,0.008937956,0.014682738,0.07273496,0.69859135,0.14541715],"study_design_scores_gemma":[0.00008345995,0.000029960942,0.07952331,0.0005570142,0.00013722962,0.0005583387,0.0046848254,0.003581645,0.005157898,0.0023383484,0.90317935,0.00016857717],"about_ca_topic_score_codex":0.95180774,"about_ca_topic_score_gemma":0.9679919,"teacher_disagreement_score":0.048192263,"about_ca_system_score_codex":0.020605456,"about_ca_system_score_gemma":0.029369853,"threshold_uncertainty_score":0.14950377},"labels":[],"label_agreement":null},{"id":"W2523933421","doi":"","title":"Removing the distinction between a translation memory, a bilingual dictionary and a parallel corpus","year":2007,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Natural language processing; Linguistics; Computer science; Artificial intelligence; Translation (biology); Machine translation; Speech recognition; Philosophy","score_opus":0.020836613650377075,"score_gpt":0.27813738250301157,"score_spread":0.2573007688526345,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2523933421","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.033480544,0.0009524483,0.9264171,0.0009145884,0.0005259909,0.00021017485,0.00048195836,0.016935999,0.020081153],"genre_scores_gemma":[0.2397086,0.00044287837,0.71993166,0.00087949313,0.000388733,0.00024588197,0.0015488116,0.0026100092,0.034243856],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99905723,0.00021243875,0.00008837591,0.00035411053,0.00024612068,0.00004173675],"domain_scores_gemma":[0.9943838,0.0022063248,0.00037568345,0.0020871898,0.00071232376,0.000234739],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013493482,0.0007649259,0.0012887457,0.0006860603,0.0010250508,0.0030396394,0.0017842123,0.0016393211,0.0181219],"category_scores_gemma":[0.006315218,0.0007477137,0.00036827932,0.0009328319,0.0008950717,0.0071592224,0.0023283893,0.0020564545,0.009740055],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016245956,0.00033154237,0.0020414935,0.0015292602,0.00013907664,0.0008404971,0.0015074807,0.0024851884,0.14844345,0.055541016,0.01908721,0.7664292],"study_design_scores_gemma":[0.000401395,0.0014152244,0.0046563945,0.00031632997,0.00042137303,0.013387718,0.0007569756,0.087607354,0.22628805,0.047467247,0.61692566,0.0003563628],"about_ca_topic_score_codex":0.00094575237,"about_ca_topic_score_gemma":0.0020814855,"teacher_disagreement_score":0.0181219,"about_ca_system_score_codex":0.0004787069,"about_ca_system_score_gemma":0.0009551625,"threshold_uncertainty_score":0.060623765},"labels":[],"label_agreement":null},{"id":"W2525005111","doi":"10.21700/ijcis.2016.108","title":"Arabic Word Sense Disambiguation Using Wikipedia","year":2016,"lang":"en","type":"article","venue":"International Journal of Computing and Information Sciences","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Word-sense disambiguation; Word (group theory); Arabic; Natural language processing; Computer science; SemEval; Artificial intelligence; Information retrieval; Linguistics; WordNet; Engineering","score_opus":0.019442508386292905,"score_gpt":0.3215169904704452,"score_spread":0.3020744820841523,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2525005111","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.33074272,0.010020965,0.52674675,0.0025350682,0.004363742,0.0011249882,0.031595293,0.034171764,0.058698665],"genre_scores_gemma":[0.47450092,0.002480244,0.4788015,0.0003514076,0.0003762123,0.00033071815,0.03028791,0.0014819737,0.011389131],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99883646,0.00025169048,0.00030797967,0.000323618,0.00020886799,0.00007141616],"domain_scores_gemma":[0.9984097,0.0004323503,0.00012633625,0.00015992057,0.00077228,0.00009948889],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00075614854,0.0014625193,0.0008908127,0.0096135875,0.002046606,0.0022010491,0.00055176136,0.00056721835,0.007837773],"category_scores_gemma":[0.0035039997,0.00041148803,0.0007352463,0.005584322,0.00042673066,0.0033708182,0.0018749242,0.0006380983,0.005064845],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011478805,0.0003186628,0.008829759,0.0022504851,0.00030766908,0.0028850583,0.002285852,0.0044808155,0.062400736,0.02059517,0.07418958,0.82030827],"study_design_scores_gemma":[0.0003714504,0.00047354843,0.020869035,0.001033472,0.0009816162,0.005548693,0.010700216,0.17213865,0.17411658,0.06823115,0.54501593,0.00051959493],"about_ca_topic_score_codex":0.00309766,"about_ca_topic_score_gemma":0.0038244238,"teacher_disagreement_score":0.0096135875,"about_ca_system_score_codex":0.0004286585,"about_ca_system_score_gemma":0.001547304,"threshold_uncertainty_score":0.026219964},"labels":[],"label_agreement":null},{"id":"W2526822183","doi":"10.7202/1028612ar","title":"L’analyse du contenu textuel en vue de la construction de thésaurus et de l’indexation assistées par ordinateur; applications possibles avec SATO","year":2015,"lang":"fr","type":"article","venue":"Documentation et bibliothèques","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Philosophy","score_opus":0.017761797599125405,"score_gpt":0.3402805174293606,"score_spread":0.3225187198302352,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2526822183","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09266303,0.003257475,0.86410093,0.0016314074,0.00023635536,0.00026912434,0.0006709777,0.0032755164,0.033895124],"genre_scores_gemma":[0.22755979,0.0027881665,0.7310831,0.00017970416,0.00016256951,0.00039455012,0.0014393037,0.0017419129,0.034650996],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9979704,0.0007135774,0.000208415,0.00029730023,0.0007315284,0.000078760335],"domain_scores_gemma":[0.9954542,0.0018932736,0.0003114577,0.0007097308,0.0014990174,0.00013233077],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026852589,0.00090568466,0.0009684814,0.005731602,0.0012699432,0.0044916524,0.0010304556,0.0009353203,0.010279269],"category_scores_gemma":[0.010824951,0.0007832383,0.0010899298,0.005652734,0.001991502,0.0047130696,0.0027086162,0.0010657625,0.004115658],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00052365114,0.000074110394,0.008198706,0.0021781335,0.0001908297,0.001501271,0.024578292,0.0069943443,0.16796045,0.16188447,0.008189934,0.6177258],"study_design_scores_gemma":[0.000077616125,0.00041906815,0.018475756,0.0012580039,0.00048681741,0.0035651207,0.016667154,0.088811874,0.11608849,0.11477987,0.63901216,0.00035804446],"about_ca_topic_score_codex":0.0077383937,"about_ca_topic_score_gemma":0.009729611,"teacher_disagreement_score":0.010279269,"about_ca_system_score_codex":0.0014943086,"about_ca_system_score_gemma":0.0023858936,"threshold_uncertainty_score":0.03438759},"labels":[],"label_agreement":null},{"id":"W2527786628","doi":"10.1007/978-3-319-46565-4_5","title":"Entity Typing and Linking Using SPARQL Patterns and DBpedia","year":2016,"lang":"en","type":"book-chapter","venue":"Communications in computer and information science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa; Polytechnique Montréal","funders":"","keywords":"SPARQL; Computer science; Entity linking; Named graph; Information retrieval; Linked data; Pipeline (software); Ontology; Exploit; Context (archaeology); RDF; Semantic Web; Dependency (UML); Natural language processing; Knowledge base; World Wide Web; Artificial intelligence; Programming language; Biology","score_opus":0.04272373627817886,"score_gpt":0.30970495499498546,"score_spread":0.2669812187168066,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2527786628","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0029850036,0.00034879314,0.97863597,0.0003907622,0.00020897243,0.000112509726,0.0012389753,0.0088960985,0.007182976],"genre_scores_gemma":[0.033026773,0.00097453553,0.94239974,0.00028393185,0.000076129705,0.0001313547,0.005865871,0.0033362044,0.013905499],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99682534,0.00060194323,0.00054195535,0.00082152087,0.0010605599,0.00014869077],"domain_scores_gemma":[0.9950054,0.0020236834,0.0001759431,0.0020215143,0.0006360273,0.00013746809],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035384526,0.00080404646,0.0012551341,0.002896759,0.001344151,0.0071578803,0.0027829104,0.001082407,0.014122998],"category_scores_gemma":[0.010588437,0.0015057256,0.0024415604,0.005999781,0.0009621406,0.011198498,0.0051998897,0.0026953213,0.008761396],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016445591,0.00023050747,0.0024159004,0.00080198154,0.00014482392,0.00063295924,0.0010400215,0.0069238166,0.0067502935,0.1577711,0.053222,0.76990205],"study_design_scores_gemma":[0.00007187912,0.0000570156,0.0012635702,0.00055641885,0.00020496266,0.0019743992,0.0006775257,0.14098454,0.04541528,0.45622543,0.35241583,0.00015308619],"about_ca_topic_score_codex":0.002494458,"about_ca_topic_score_gemma":0.0041483804,"teacher_disagreement_score":0.014122998,"about_ca_system_score_codex":0.0007015828,"about_ca_system_score_gemma":0.0018103997,"threshold_uncertainty_score":0.0472461},"labels":[],"label_agreement":null},{"id":"W2528039253","doi":"10.1007/978-3-319-46565-4_3","title":"Collective Disambiguation and Semantic Annotation for Entity Linking and Typing","year":2016,"lang":"en","type":"book-chapter","venue":"Communications in computer and information science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa; Polytechnique Montréal","funders":"","keywords":"Computer science; Heuristics; Task (project management); Natural language processing; Annotation; Information retrieval; Entity linking; Semantic annotation; Artificial intelligence; Knowledge base","score_opus":0.033698256612100214,"score_gpt":0.31277809580133026,"score_spread":0.2790798391892301,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2528039253","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0035020579,0.0011333759,0.9773375,0.00047095708,0.0005464139,0.000064112464,0.0003068545,0.0047559016,0.011882777],"genre_scores_gemma":[0.070586495,0.00097590673,0.8914286,0.00023159434,0.00040528245,0.0001974963,0.0023944902,0.0021497763,0.031630356],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9963264,0.0008719544,0.0003772113,0.0010719905,0.0011338133,0.00021864196],"domain_scores_gemma":[0.99401206,0.001999737,0.00022534878,0.0028244818,0.0008146315,0.00012373355],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035958316,0.001496947,0.0019441916,0.0038434751,0.0030797936,0.0052846265,0.0044770744,0.002149155,0.016688764],"category_scores_gemma":[0.009478502,0.0012565074,0.0021320994,0.00630469,0.0028765036,0.012155224,0.006715151,0.003911898,0.0103237815],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012284823,0.00009232066,0.0005491628,0.00037752205,0.000060300932,0.00019306496,0.0010928318,0.0058809174,0.0055616735,0.31078607,0.048076767,0.62720644],"study_design_scores_gemma":[0.00001448343,0.000030524472,0.0005209199,0.00017054011,0.00007837961,0.00036461902,0.00038179217,0.11287737,0.018071825,0.70359087,0.16382112,0.00007759423],"about_ca_topic_score_codex":0.0022291713,"about_ca_topic_score_gemma":0.005198048,"teacher_disagreement_score":0.016688764,"about_ca_system_score_codex":0.001389139,"about_ca_system_score_gemma":0.0019787108,"threshold_uncertainty_score":0.055829465},"labels":[],"label_agreement":null},{"id":"W2529201282","doi":"10.13053/cys-20-3-2465","title":"A Comparison of Methods for Identifying the Translation of Words in a Comparable Corpus: Recipes and Limits","year":2016,"lang":"en","type":"article","venue":"Computación y Sistemas","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Translation (biology); Natural language processing; Computer science; Artificial intelligence; Linguistics; Philosophy; Biology","score_opus":0.13028510169271237,"score_gpt":0.4400326972395826,"score_spread":0.30974759554687026,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2529201282","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.21941288,0.049667697,0.68711245,0.0017561981,0.0013655084,0.0014032292,0.0037382257,0.012767607,0.022776218],"genre_scores_gemma":[0.2560461,0.008016388,0.7187874,0.00032648505,0.00032974186,0.001314967,0.0099091185,0.0023228119,0.0029471046],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9675146,0.016077269,0.0033901904,0.0049592135,0.0074136998,0.000644925],"domain_scores_gemma":[0.93154174,0.044740595,0.0019000666,0.013122606,0.0076228883,0.0010721144],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02429789,0.0027436765,0.002105161,0.008618273,0.0013593601,0.0050849724,0.0029375483,0.003488737,0.0026716057],"category_scores_gemma":[0.089989245,0.001241772,0.0018197855,0.0064874794,0.001911108,0.008091491,0.0046488903,0.0023680832,0.0030333817],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019174446,0.0005330344,0.010187785,0.004288473,0.0018989498,0.00033816486,0.0020746016,0.021745611,0.016707191,0.012595205,0.012845258,0.91486824],"study_design_scores_gemma":[0.0018517714,0.004961319,0.06556346,0.0032417846,0.0028004814,0.007014851,0.008812598,0.59387434,0.10497282,0.10910949,0.09662766,0.0011694835],"about_ca_topic_score_codex":0.00225374,"about_ca_topic_score_gemma":0.0035740226,"teacher_disagreement_score":0.02429789,"about_ca_system_score_codex":0.0011237641,"about_ca_system_score_gemma":0.0020160943,"threshold_uncertainty_score":0.12850106},"labels":[],"label_agreement":null},{"id":"W2531207078","doi":"10.1162/tacl_a_00067","title":"Fully Character-Level Neural Machine Translation without Explicit Segmentation","year":2017,"lang":"en","type":"article","venue":"Transactions of the Association for Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":415,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"York University; Samsung Advanced Institute of Technology; Samsung; Nvidia","keywords":"Computer science; Machine translation; Character (mathematics); Pooling; Encoder; Convolutional neural network; Artificial intelligence; Translation (biology); Natural language processing; Segmentation; Speech recognition; Task (project management); Language model; Representation (politics)","score_opus":0.038881259439473397,"score_gpt":0.3152326302466266,"score_spread":0.2763513708071532,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2531207078","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06779244,0.0007854792,0.9006131,0.0003423279,0.00023692379,0.00011273133,0.0013587138,0.014669129,0.014088998],"genre_scores_gemma":[0.6053543,0.00052268646,0.36572236,0.00039105985,0.000104567,0.00020415416,0.007608874,0.00093440735,0.019157542],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99957794,0.00008596748,0.00003321502,0.00016696187,0.00008662008,0.000049267353],"domain_scores_gemma":[0.9991353,0.00023459185,0.0000637335,0.0003126864,0.00022577246,0.000027890066],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00040190128,0.0009828565,0.00066124415,0.00045984826,0.0003852232,0.0008666583,0.001083708,0.00087407074,0.0061237444],"category_scores_gemma":[0.0021532942,0.00034396126,0.00059000624,0.0010211072,0.00037856968,0.0018106056,0.00090742606,0.001024423,0.0045761485],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005324047,0.0003176903,0.001353153,0.000491166,0.00016107854,0.00048594724,0.00018088486,0.18456735,0.09449791,0.018318903,0.020804856,0.6782887],"study_design_scores_gemma":[0.000024890289,0.00011860493,0.00058795844,0.000021266314,0.00003288747,0.000222429,0.000027940447,0.93966883,0.03822535,0.011165153,0.009879828,0.000024935502],"about_ca_topic_score_codex":0.0036810718,"about_ca_topic_score_gemma":0.007645729,"teacher_disagreement_score":0.0061237444,"about_ca_system_score_codex":0.00050638994,"about_ca_system_score_gemma":0.00116801,"threshold_uncertainty_score":0.020485997},"labels":[],"label_agreement":null},{"id":"W2531882892","doi":"10.1111/cogs.12414","title":"Grammaticality, Acceptability, and Probability: A Probabilistic View of Linguistic Knowledge","year":2016,"lang":"en","type":"article","venue":"Cognitive Science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":261,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Menzies Centre for Australian Studies, King's College London, University of London; Economic and Social Research Council; University of Toronto; Chalmers Tekniska Högskola; University of Reading; Universiteit van Amsterdam; Weizmann Institute of Science; Göteborgs Universitet; University of Edinburgh; King's College London","keywords":"Grammaticality; Probabilistic logic; Natural language processing; Artificial intelligence; Computer science; Sentence; Natural language; Set (abstract data type); Linguistics; Grammar","score_opus":0.030819051176993476,"score_gpt":0.329199145905938,"score_spread":0.29838009472894456,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2531882892","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16859914,0.0016103148,0.78240716,0.015746005,0.00014471854,0.00007291646,0.00037698227,0.00056155596,0.030481223],"genre_scores_gemma":[0.9425486,0.00048381722,0.054245736,0.0008245838,0.00024853548,0.00010224438,0.00019248297,0.00013869595,0.0012152693],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9942795,0.0025209289,0.00023301065,0.0011240843,0.0015997088,0.00024274261],"domain_scores_gemma":[0.9661623,0.02468512,0.0029006433,0.003241659,0.0020643906,0.0009457276],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0059877355,0.000674979,0.0011220301,0.0026978117,0.0014006837,0.0063103833,0.0021737206,0.0025270442,0.0026062068],"category_scores_gemma":[0.04243263,0.00091420114,0.0014043989,0.0016372022,0.014189078,0.016702753,0.003812434,0.005449856,0.00039164248],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00026631614,0.00012958223,0.01839811,0.0003064181,0.00019835377,0.00025329145,0.006592702,0.028656702,0.0054632174,0.87350076,0.0018848137,0.06434987],"study_design_scores_gemma":[0.000014282287,0.00005454635,0.0075795716,0.00004506372,0.000033667842,0.00021219678,0.0002892359,0.03810593,0.00068599515,0.950958,0.001955048,0.000066397115],"about_ca_topic_score_codex":0.0042213667,"about_ca_topic_score_gemma":0.0023896873,"teacher_disagreement_score":0.0063103833,"about_ca_system_score_codex":0.0017835159,"about_ca_system_score_gemma":0.0010763633,"threshold_uncertainty_score":0.031666517},"labels":[],"label_agreement":null},{"id":"W2532566882","doi":"10.1075/lfab.5.06rez","title":"Building and interpreting nonthematic A-positions","year":2011,"lang":"en","type":"book-chapter","venue":"Language faculty and beyond","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Nautical Research Society","funders":"","keywords":"Merge (version control); Spec#; Locality; Mathematics; Variable (mathematics); Computer science; Mathematical analysis; Linguistics; Parallel computing; Philosophy","score_opus":0.014964764629513438,"score_gpt":0.270820391177278,"score_spread":0.25585562654776456,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2532566882","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14953756,0.0006005326,0.50853497,0.0016639865,0.00022840408,0.00016087118,0.0003950076,0.0017980748,0.33708057],"genre_scores_gemma":[0.92142975,0.00023441103,0.04959972,0.0002536723,0.00008102141,0.00009970234,0.00033344608,0.0011236878,0.026844518],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99728215,0.00095878815,0.00019121527,0.00069419947,0.0006005209,0.00027314833],"domain_scores_gemma":[0.9964522,0.0011479056,0.0002851282,0.0013858062,0.00057530776,0.00015370133],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00244881,0.0007642381,0.0005343187,0.0015308288,0.0025848867,0.006477115,0.0024692006,0.0017408984,0.013433354],"category_scores_gemma":[0.0058735604,0.0009471831,0.0009892606,0.0012835384,0.0074035362,0.01413027,0.004936896,0.0035180969,0.002639006],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000024615416,0.000013763692,0.0007054709,0.000056536923,0.000010315462,0.00016431331,0.006472015,0.00030422414,0.0029219296,0.9755096,0.0007731895,0.013044089],"study_design_scores_gemma":[0.000071951974,0.00008803978,0.0039076097,0.00016021429,0.000042656935,0.0005137022,0.007832923,0.011369598,0.014161,0.86339754,0.0983668,0.00008805422],"about_ca_topic_score_codex":0.006434568,"about_ca_topic_score_gemma":0.0054426165,"teacher_disagreement_score":0.013433354,"about_ca_system_score_codex":0.0032613147,"about_ca_system_score_gemma":0.0016959494,"threshold_uncertainty_score":0.0449391},"labels":[],"label_agreement":null},{"id":"W2533603510","doi":"","title":"TAL et réseaux sociaux","year":2013,"lang":"fr","type":"book","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Humanities; Political science; Sociology; Art","score_opus":0.02307827046548959,"score_gpt":0.2665178984126601,"score_spread":0.2434396279471705,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2533603510","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0014597041,0.044520665,0.00082942523,0.010281582,0.005699578,0.000018613171,0.00017412362,0.00013457579,0.93688166],"genre_scores_gemma":[0.020098867,0.019218303,0.00043647646,0.0011406911,0.0039515016,0.000041880347,0.0002040508,0.000114979506,0.9547932],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99883145,0.00032065428,0.000045385816,0.0001931753,0.00044507443,0.00016425955],"domain_scores_gemma":[0.9992908,0.00014974599,0.000050994495,0.00007254349,0.00019338251,0.00024258101],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00090214185,0.001104515,0.00079085183,0.004597395,0.0046019168,0.009205486,0.0010095567,0.0019702935,0.10924414],"category_scores_gemma":[0.0015993753,0.00025527255,0.00045443134,0.0058801123,0.0043437285,0.0065366016,0.0029054817,0.0030345598,0.033357967],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000026935119,0.000033279153,0.0001295144,0.00018269476,0.000005616258,0.00006945429,0.0020990712,0.0001473024,0.00017001502,0.63332963,0.29200068,0.07180584],"study_design_scores_gemma":[0.0000014946647,0.0000049851487,0.00010252422,0.000037896905,0.0000011648647,0.00002979218,0.00027571333,0.000024212926,0.000018549226,0.009681985,0.98981935,0.000002253141],"about_ca_topic_score_codex":0.009348967,"about_ca_topic_score_gemma":0.014103883,"teacher_disagreement_score":0.10924414,"about_ca_system_score_codex":0.0059047397,"about_ca_system_score_gemma":0.0030325702,"threshold_uncertainty_score":0.36545807},"labels":[],"label_agreement":null},{"id":"W2537796433","doi":"","title":"5.7 Rule ordering relationships","year":2004,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science","score_opus":0.022884844094756517,"score_gpt":0.2644703792999414,"score_spread":0.24158553520518486,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2537796433","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01893571,0.0014520319,0.71222323,0.0033953984,0.0012895237,0.00092687644,0.006684547,0.008957328,0.24613535],"genre_scores_gemma":[0.17601533,0.0014223466,0.6940386,0.0015104364,0.0005787205,0.00055392814,0.011679136,0.002610161,0.111591265],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99387616,0.0015817398,0.000616898,0.0009178277,0.0024389846,0.0005683404],"domain_scores_gemma":[0.99242353,0.0027994227,0.00026900074,0.0021635545,0.0020976,0.00024689862],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0039540436,0.00093808514,0.00079894345,0.003248419,0.0018988892,0.0042803218,0.0024717143,0.0015575554,0.06693456],"category_scores_gemma":[0.011766249,0.00079913833,0.0017650778,0.0025540069,0.0011019291,0.005924893,0.002647952,0.0027080367,0.02265257],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004724882,0.0005087932,0.0050369957,0.00071319845,0.00012903665,0.0014951342,0.000921781,0.0046969065,0.009687647,0.42035532,0.06578953,0.49019316],"study_design_scores_gemma":[0.00010009116,0.0001789949,0.002279716,0.0003348188,0.0002301268,0.0016801967,0.00064745144,0.027324209,0.028378183,0.38983378,0.5489196,0.00009275999],"about_ca_topic_score_codex":0.003339971,"about_ca_topic_score_gemma":0.006020682,"teacher_disagreement_score":0.06693456,"about_ca_system_score_codex":0.0009035045,"about_ca_system_score_gemma":0.002047854,"threshold_uncertainty_score":0.22391838},"labels":[],"label_agreement":null},{"id":"W2538309467","doi":"10.1121/1.3654646","title":"Finding schwa: Comparing the results of an automatic aligner with human judgments when identifying schwa in a corpus of spoken French","year":2011,"lang":"en","type":"article","venue":"The Journal of the Acoustical Society of America","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Wilfrid Laurier University; University of Ottawa","funders":"","keywords":"Schwa; Computer science; Phone; Context (archaeology); Natural language processing; Word (group theory); Speech recognition; Phonetics; Coding (social sciences); Artificial intelligence; Linguistics; Vowel; Mathematics; Statistics; History","score_opus":0.039035844611682295,"score_gpt":0.287982443924876,"score_spread":0.24894659931319368,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2538309467","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9385588,0.00038206135,0.056214392,0.000108384585,0.000076141594,0.00017516817,0.00041980925,0.0012121222,0.0028529065],"genre_scores_gemma":[0.88587636,0.00020533496,0.11072052,0.000118749034,0.000056493245,0.00019890553,0.0007968872,0.00041362352,0.0016130705],"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9891227,0.0059307343,0.00059149135,0.0022940363,0.0016398311,0.00042125775],"domain_scores_gemma":[0.9623904,0.028229024,0.0018102293,0.0024776263,0.0045274594,0.000565266],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011363224,0.00079235143,0.0007131545,0.0020104072,0.0008274987,0.0018455891,0.00059426605,0.0011959643,0.0021166932],"category_scores_gemma":[0.04485023,0.00035502628,0.0004619272,0.00080367184,0.0011574428,0.0016110437,0.0011899485,0.00045339722,0.0016207658],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0048994683,0.00041512446,0.08591506,0.0012610849,0.0004894866,0.0005237578,0.0201525,0.0025464206,0.48641747,0.00082955556,0.0020011899,0.39454877],"study_design_scores_gemma":[0.0003180871,0.004290527,0.6704896,0.00010924151,0.0004800227,0.0027520547,0.014283206,0.043433674,0.2505592,0.001962676,0.0106537705,0.00066785864],"about_ca_topic_score_codex":0.0025890553,"about_ca_topic_score_gemma":0.006148302,"teacher_disagreement_score":0.011363224,"about_ca_system_score_codex":0.0004051941,"about_ca_system_score_gemma":0.0004838426,"threshold_uncertainty_score":0.06009519},"labels":[],"label_agreement":null},{"id":"W2540255286","doi":"10.1109/edcomp.1984.680689","title":"A Micro-based Dictionary System For The Blind","year":2005,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Regina","funders":"","keywords":"Computer science; Artificial intelligence; Speech recognition","score_opus":0.014640855042254913,"score_gpt":0.26850219385907914,"score_spread":0.2538613388168242,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2540255286","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024054982,0.0007621801,0.88182163,0.0004325865,0.0007353574,0.00034588692,0.002972738,0.073419355,0.015455299],"genre_scores_gemma":[0.2069765,0.0007134972,0.7386888,0.0008229998,0.00033711002,0.0005408551,0.005327478,0.004183769,0.042408943],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99925894,0.000096865515,0.0001004079,0.0002278967,0.00024537204,0.00007053921],"domain_scores_gemma":[0.9972345,0.000614766,0.00012861013,0.00091736024,0.0008808728,0.00022385898],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00065138895,0.00057090336,0.0010713781,0.0015285675,0.0011568563,0.0018867933,0.0018446337,0.0010007322,0.03499185],"category_scores_gemma":[0.003407078,0.00050917565,0.00031279083,0.0014647851,0.0004729245,0.003486279,0.0031549388,0.0009994907,0.017719187],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00206348,0.00018463121,0.0018185338,0.00045391533,0.00006708047,0.0003750803,0.00050378917,0.0017514029,0.081209116,0.022389295,0.07373375,0.8154501],"study_design_scores_gemma":[0.00092763273,0.0011498083,0.0025509438,0.00026003204,0.00032863073,0.0031725126,0.000867857,0.19901124,0.25147754,0.062056378,0.47785664,0.00034078397],"about_ca_topic_score_codex":0.0017018255,"about_ca_topic_score_gemma":0.0037223895,"teacher_disagreement_score":0.03499185,"about_ca_system_score_codex":0.00043471422,"about_ca_system_score_gemma":0.001504983,"threshold_uncertainty_score":0.11705941},"labels":[],"label_agreement":null},{"id":"W2541525715","doi":"10.1109/have.2004.1391898","title":"A prototype natural language interface for animation systems","year":2005,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Natural language user interface; Animation; Interface (matter); Avatar; Interpreter; Natural language; Human–computer interaction; Parsing; Natural language understanding; Domain (mathematical analysis); User interface; Computer animation; Interface description language; Programming language; Artificial intelligence; Computer graphics (images)","score_opus":0.010286403231042598,"score_gpt":0.299815075749493,"score_spread":0.28952867251845044,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2541525715","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016324177,0.00016890067,0.8529745,0.00039932548,0.00017441562,0.0009286212,0.00053126964,0.120163664,0.008335086],"genre_scores_gemma":[0.11683152,0.00019803765,0.8574824,0.00054371485,0.000076869335,0.0011757246,0.0021468645,0.004324907,0.017219996],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9991136,0.00023987195,0.00008288188,0.00020794141,0.00029163127,0.00006398181],"domain_scores_gemma":[0.9986695,0.0006080111,0.000050316437,0.00019922924,0.0003487108,0.00012426129],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016275165,0.0006190703,0.0005984232,0.00048537023,0.00049995066,0.001467048,0.003326977,0.0013238733,0.023487596],"category_scores_gemma":[0.003968082,0.00065162615,0.0005842633,0.00034696094,0.00056051335,0.0027287211,0.0010183544,0.0014112748,0.005071491],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0023969545,0.0016503888,0.002253395,0.0017458574,0.00014320725,0.0020731296,0.0034181853,0.011103808,0.31060377,0.043135386,0.097156405,0.5243195],"study_design_scores_gemma":[0.0019618738,0.0024051573,0.0022872451,0.00031046028,0.00026631716,0.0033842807,0.0005535297,0.37021357,0.17188697,0.023582794,0.42280042,0.00034751504],"about_ca_topic_score_codex":0.0011839607,"about_ca_topic_score_gemma":0.00091681146,"teacher_disagreement_score":0.023487596,"about_ca_system_score_codex":0.00046505677,"about_ca_system_score_gemma":0.0008190091,"threshold_uncertainty_score":0.07857382},"labels":[],"label_agreement":null},{"id":"W2543062344","doi":"","title":"3.5.3 Place of articulation","year":2004,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Articulation (sociology); Linguistics; Political science; Philosophy; Law; Politics","score_opus":0.008135624800965224,"score_gpt":0.2540996964067478,"score_spread":0.24596407160578257,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2543062344","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025319992,0.0019150184,0.5554473,0.0018349872,0.0030285404,0.00031470263,0.0032791425,0.006171456,0.40268883],"genre_scores_gemma":[0.46345997,0.0018460652,0.32117245,0.0005168738,0.0007689361,0.00027130885,0.004744297,0.003352517,0.20386755],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9974831,0.0004072659,0.00020235556,0.0005083818,0.0009473525,0.000451622],"domain_scores_gemma":[0.99824595,0.00028829303,0.00010266914,0.0006987229,0.0005605876,0.000103745915],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0014062022,0.0012465821,0.0009836855,0.0020649207,0.0020905014,0.009057288,0.0020219195,0.0021253768,0.053460155],"category_scores_gemma":[0.0045061293,0.0006349001,0.0020211933,0.002113543,0.0026489168,0.0062350016,0.0042311656,0.0025010933,0.023699231],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00107734,0.00014410181,0.004428634,0.0008457371,0.00007194924,0.0018130039,0.003902021,0.004302391,0.042979915,0.6666605,0.026192216,0.24758233],"study_design_scores_gemma":[0.00006355373,0.00020734548,0.003413882,0.00038101085,0.00013665768,0.0027103897,0.0035822056,0.015090585,0.059012815,0.2834854,0.6317105,0.00020558051],"about_ca_topic_score_codex":0.004976411,"about_ca_topic_score_gemma":0.004507907,"teacher_disagreement_score":0.9465398,"about_ca_system_score_codex":0.0011365338,"about_ca_system_score_gemma":0.0016843195,"threshold_uncertainty_score":0.17884201},"labels":[],"label_agreement":null},{"id":"W2543556230","doi":"10.1109/nlpke.2003.1275910","title":"Fuzzy semantic measurement for synonymy and its npplication in an automatic question-answering system","year":2004,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Semantics (computer science); Natural language processing; Closeness; Sentence; Information retrieval; Question answering; Artificial intelligence; Construct (python library); Measure (data warehouse); Fuzzy logic; Semantic similarity; Novelty; Data mining; Mathematics; Programming language","score_opus":0.022732170256688992,"score_gpt":0.2858214549272483,"score_spread":0.2630892846705593,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2543556230","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14340237,0.00027851717,0.85307455,0.00028128256,0.000023201517,0.000121933124,0.0002263167,0.00087870326,0.0017131704],"genre_scores_gemma":[0.6188812,0.00007259132,0.38014415,0.00005243247,0.000038135393,0.00019372127,0.00024892384,0.000051033323,0.0003178412],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99209315,0.0034785874,0.0006561118,0.0011166551,0.002471812,0.00018365627],"domain_scores_gemma":[0.98904455,0.0073173656,0.00096833264,0.0009918266,0.0014279659,0.00024994905],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0059193065,0.00046427565,0.000943768,0.004206993,0.0011552321,0.0027178028,0.0011439056,0.0010069726,0.0015035957],"category_scores_gemma":[0.021229224,0.00032786708,0.0005843282,0.002886534,0.0017084285,0.005458221,0.0014568967,0.0009026328,0.00033557255],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001043302,0.0004946889,0.016303344,0.0007541149,0.00028891957,0.00051499694,0.009226402,0.021335917,0.1474362,0.21468128,0.002409247,0.5855116],"study_design_scores_gemma":[0.00011634938,0.0007534739,0.015962964,0.00010536522,0.00021421694,0.00097460754,0.0022047535,0.6582788,0.08091089,0.23027885,0.01000562,0.00019408947],"about_ca_topic_score_codex":0.0010738432,"about_ca_topic_score_gemma":0.0009364969,"teacher_disagreement_score":0.0059193065,"about_ca_system_score_codex":0.0010664047,"about_ca_system_score_gemma":0.0007943424,"threshold_uncertainty_score":0.031304657},"labels":[],"label_agreement":null},{"id":"W2547377758","doi":"10.7771/2380-176x.6595","title":"Pelikan's Antidisambiguation: The End of the Wax Cylinder as We Know It","year":2013,"lang":"en","type":"article","venue":"Against the grain","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Wax; Chemistry","score_opus":0.012281039016839133,"score_gpt":0.25214383970524246,"score_spread":0.23986280068840332,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2547377758","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06574519,0.009208682,0.37632245,0.12316319,0.007558556,0.000086508444,0.00048020892,0.0032669173,0.41416824],"genre_scores_gemma":[0.82524306,0.0025961068,0.07525054,0.017845148,0.0016167992,0.0000772385,0.00024165236,0.0022291706,0.0749003],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9972367,0.0010738575,0.00014388774,0.0005448485,0.00080038287,0.00020022361],"domain_scores_gemma":[0.9945517,0.0025261235,0.0002820322,0.0016259423,0.0008806824,0.00013354952],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029437125,0.00054812175,0.0008024693,0.0010687185,0.003817287,0.005185687,0.0012148763,0.002866732,0.009311803],"category_scores_gemma":[0.017491758,0.0006612688,0.00043848966,0.0008471156,0.007892301,0.019500947,0.004730039,0.006646935,0.002688034],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018795866,0.000021866068,0.00063456385,0.00009274056,0.000019183763,0.00024673244,0.004694468,0.0002832062,0.0020960579,0.90862554,0.023203967,0.05989371],"study_design_scores_gemma":[0.000027947128,0.00001840168,0.00039493418,0.00008648165,0.000023599738,0.0003670718,0.0014618817,0.002371979,0.0036934167,0.8786835,0.112822846,0.00004795783],"about_ca_topic_score_codex":0.0029087286,"about_ca_topic_score_gemma":0.003703968,"teacher_disagreement_score":0.009311803,"about_ca_system_score_codex":0.0010505299,"about_ca_system_score_gemma":0.0015841088,"threshold_uncertainty_score":0.031151056},"labels":[],"label_agreement":null},{"id":"W2548156159","doi":"","title":"4.9 Further rule writing conventions and abbreviatory devices","year":2004,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science","score_opus":0.010478475273115564,"score_gpt":0.2665633890993016,"score_spread":0.25608491382618603,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2548156159","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006222911,0.00044143843,0.8878693,0.0027636131,0.0018305761,0.00033516492,0.0022577306,0.008487383,0.089791924],"genre_scores_gemma":[0.06327838,0.000350087,0.8969428,0.0010716274,0.00048333046,0.0003860619,0.002114158,0.0025097437,0.032863773],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99231875,0.00301979,0.0016608837,0.0010709377,0.0013937782,0.0005359908],"domain_scores_gemma":[0.98328227,0.004654624,0.0005173171,0.007662718,0.0035578324,0.00032527486],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0076271486,0.0014750262,0.0015642875,0.0021707192,0.0027186184,0.007034648,0.0039024334,0.0023034753,0.043330688],"category_scores_gemma":[0.021085609,0.0011459619,0.003276419,0.0018324648,0.0031252482,0.0103770215,0.0046582795,0.0057464773,0.021950023],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00026052154,0.0001654937,0.0009248923,0.0004613858,0.000066136294,0.0005367194,0.0018631844,0.0013777575,0.009262062,0.7988845,0.057997957,0.12819937],"study_design_scores_gemma":[0.00006538172,0.00009383889,0.00060423615,0.00021798938,0.00012302959,0.000903235,0.00057919725,0.009079189,0.022107765,0.3942054,0.57187897,0.0001417778],"about_ca_topic_score_codex":0.0016772794,"about_ca_topic_score_gemma":0.002295828,"teacher_disagreement_score":0.043330688,"about_ca_system_score_codex":0.00094131596,"about_ca_system_score_gemma":0.0014521028,"threshold_uncertainty_score":0.14495564},"labels":[],"label_agreement":null},{"id":"W2548230849","doi":"10.1075/nlp.2.15mey","title":"Extracting knowledge-rich contexts for terminography","year":2001,"lang":"en","type":"book-chapter","venue":"Natural language processing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":215,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Construct (python library); Computer science; Paralanguage; Domain knowledge; Domain (mathematical analysis); Knowledge extraction; Field (mathematics); Context (archaeology); Knowledge management; Natural language processing; Data science; Artificial intelligence; Psychology; Communication; Geography; Mathematics","score_opus":0.015630032468229135,"score_gpt":0.30286852901542444,"score_spread":0.2872384965471953,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2548230849","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006694648,0.0056135827,0.9614586,0.00064546097,0.00021878842,0.00016266714,0.00046669674,0.0012028265,0.023536794],"genre_scores_gemma":[0.039570004,0.0051958705,0.94439495,0.00018241013,0.0001314663,0.00022337977,0.0015250286,0.00049381435,0.008283064],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9990559,0.00035996974,0.00009406748,0.00015774896,0.00029602257,0.00003618616],"domain_scores_gemma":[0.9973213,0.0018525376,0.000098133016,0.00042182,0.0002590098,0.000047177462],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013881346,0.0006908762,0.00059115223,0.0030333297,0.0011315914,0.002879164,0.0011629503,0.00079222466,0.0047595976],"category_scores_gemma":[0.0059986236,0.00072562293,0.00091358204,0.0033122667,0.0014118414,0.0072092996,0.0023557968,0.0018680994,0.0036112852],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000061908984,0.00003753115,0.0010432815,0.0013162857,0.00004172568,0.0006070177,0.003521673,0.0035422232,0.010266702,0.4065999,0.020075407,0.55288637],"study_design_scores_gemma":[0.000021082735,0.000032369415,0.0012840141,0.0011981163,0.000070265465,0.0013682056,0.00097531005,0.026884936,0.015587053,0.56500834,0.3875002,0.000070042675],"about_ca_topic_score_codex":0.00058003643,"about_ca_topic_score_gemma":0.0017609375,"teacher_disagreement_score":0.0047595976,"about_ca_system_score_codex":0.00078638503,"about_ca_system_score_gemma":0.0010683622,"threshold_uncertainty_score":0.015922487},"labels":[],"label_agreement":null},{"id":"W2550742442","doi":"10.16995/dscn.315","title":"How to Do Lexical Quality Estimation of a Large OCRed Historical Finnish Newspaper Collection with Scarce Resources","year":2020,"lang":"en","type":"preprint","venue":"Digital Studies / Le champ numérique","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Newspaper; Usability; Computer science; Data collection; Quality (philosophy); Digital collections; Information retrieval; World Wide Web; Statistics; Mathematics; Sociology; Media studies","score_opus":0.033293155084365665,"score_gpt":0.3027029905237361,"score_spread":0.26940983543937047,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2550742442","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6192213,0.0047868537,0.334457,0.0014838254,0.0006482087,0.0016067496,0.015630405,0.006233176,0.015932405],"genre_scores_gemma":[0.5488381,0.001596739,0.42230275,0.00023737142,0.0002510544,0.00108445,0.019890022,0.0017617611,0.0040377793],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99312073,0.0012605959,0.0012294849,0.0014283216,0.0026355286,0.00032532378],"domain_scores_gemma":[0.96440315,0.011710682,0.003397427,0.00386891,0.016025765,0.000594031],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00719854,0.0010043316,0.0011235939,0.0110068675,0.0015902315,0.005130279,0.0011399775,0.0010665547,0.003194839],"category_scores_gemma":[0.05409365,0.000704504,0.0011148588,0.007445923,0.0012282828,0.004746159,0.0020261065,0.00076842145,0.003717261],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00056071655,0.00025369125,0.08854481,0.00343529,0.0005145395,0.0015983781,0.010943536,0.004106218,0.068590745,0.0024395683,0.020989973,0.79802245],"study_design_scores_gemma":[0.00013278265,0.00077898445,0.6377292,0.0014503605,0.0010760596,0.0038568748,0.022644252,0.057180624,0.12400181,0.009448047,0.1409652,0.00073573063],"about_ca_topic_score_codex":0.009478221,"about_ca_topic_score_gemma":0.01282093,"teacher_disagreement_score":0.0110068675,"about_ca_system_score_codex":0.0011042262,"about_ca_system_score_gemma":0.0012611751,"threshold_uncertainty_score":0.038069963},"labels":[],"label_agreement":null},{"id":"W2552717952","doi":"10.1007/978-3-319-41337-2_9","title":"Definition-Based Grounding","year":2016,"lang":"en","type":"book-chapter","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Computer Research Institute of Montréal","funders":"","keywords":"Cohesion (chemistry); Computer science; Word (group theory); Similarity (geometry); Property (philosophy); Natural language processing; Context (archaeology); Artificial intelligence; Linguistics; Epistemology; Geography; Philosophy; Archaeology","score_opus":0.030241979117040334,"score_gpt":0.25153705802036563,"score_spread":0.2212950789033253,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2552717952","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0040424997,0.0033779764,0.6778034,0.0026116136,0.0009988103,0.00013742078,0.00075138046,0.0016474797,0.30862945],"genre_scores_gemma":[0.19506785,0.0080581475,0.56017035,0.0014248252,0.00095040177,0.0003114199,0.005324079,0.002807233,0.22588564],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9988431,0.000217363,0.000102193015,0.0003747712,0.0003775629,0.00008497693],"domain_scores_gemma":[0.99890125,0.00041575634,0.000037041522,0.00040762333,0.0002067254,0.000031551102],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012030733,0.0008241628,0.0008729696,0.002045681,0.0014948245,0.004278139,0.0021024174,0.0008750824,0.028232051],"category_scores_gemma":[0.0030365242,0.0008622458,0.0013195353,0.0026536917,0.0034645607,0.012532061,0.0037097284,0.003798741,0.010789157],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000010660604,0.000014043066,0.000050973005,0.0001102509,0.000007546383,0.000025345036,0.00021852092,0.0005193394,0.0005835274,0.8945156,0.012317301,0.09162677],"study_design_scores_gemma":[0.0000054408706,0.0000069984862,0.000061376166,0.0000970705,0.000011830175,0.00006292191,0.0000819237,0.0022389586,0.0011197771,0.82647794,0.16982654,0.000009234408],"about_ca_topic_score_codex":0.0013515198,"about_ca_topic_score_gemma":0.0017291799,"teacher_disagreement_score":0.028232051,"about_ca_system_score_codex":0.0017067443,"about_ca_system_score_gemma":0.0013128433,"threshold_uncertainty_score":0.094445586},"labels":[],"label_agreement":null},{"id":"W2556879810","doi":"10.1007/978-3-662-53826-5_7","title":"Compositional Event Semantics in Pregroup Grammars","year":2016,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Unification; Rule-based machine translation; Natural language processing; Concatenation (mathematics); Artificial intelligence; Event (particle physics); Tree-adjoining grammar; Semantics (computer science); Programming language; Context-free grammar; Mathematics; Arithmetic","score_opus":0.009945073461505556,"score_gpt":0.2559054803308392,"score_spread":0.24596040686933365,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2556879810","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023919612,0.0004461425,0.943647,0.0006331814,0.00022487325,0.00011841832,0.00017266344,0.0011934201,0.029644625],"genre_scores_gemma":[0.63470584,0.0007414806,0.3422791,0.0005209888,0.00032844834,0.000254671,0.0006113774,0.000871185,0.019686904],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9988066,0.0004759355,0.00008891942,0.00018679745,0.0003477656,0.000093900606],"domain_scores_gemma":[0.99925715,0.0003067683,0.00004311581,0.00017238152,0.00017631114,0.00004425355],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015502978,0.00061334966,0.00055924646,0.0009955586,0.0009854011,0.0025235119,0.0011726399,0.0010083277,0.0043474645],"category_scores_gemma":[0.0017750849,0.0004557814,0.0014631504,0.00075291184,0.0031434514,0.004168799,0.002641071,0.0020094956,0.0010875516],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000012079539,0.0000139272715,0.00005182071,0.000029420891,0.00000620523,0.000090560345,0.00031585194,0.002164521,0.0012609463,0.98901725,0.0005567852,0.006480639],"study_design_scores_gemma":[0.000010755108,0.000011928241,0.000041505486,0.00001840929,0.000008533344,0.000054571876,0.00006340629,0.01567698,0.0017653396,0.9694834,0.012855735,0.000009510958],"about_ca_topic_score_codex":0.00091149093,"about_ca_topic_score_gemma":0.00095852325,"teacher_disagreement_score":0.0043474645,"about_ca_system_score_codex":0.0010837023,"about_ca_system_score_gemma":0.0010984187,"threshold_uncertainty_score":0.014543772},"labels":[],"label_agreement":null},{"id":"W2561326051","doi":"10.1109/eusipco.2016.7760580","title":"Evaluation of graph metrics for optimizing bin-based ontologically smoothed language models","year":2016,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Institut National de la Recherche Scientifique; Université de Moncton","funders":"","keywords":"Computer science; Smoothing; Language model; Graph; PageRank; Bin; Artificial intelligence; Theoretical computer science; Set (abstract data type); Data mining; Machine learning; Natural language processing; Algorithm; Programming language","score_opus":0.06564060383813436,"score_gpt":0.3307504274541061,"score_spread":0.26510982361597174,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2561326051","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.21189483,0.00074894587,0.7746832,0.0002663814,0.000076411634,0.00020337066,0.00049304444,0.010002553,0.0016312151],"genre_scores_gemma":[0.5415463,0.00019359936,0.45410082,0.000080913305,0.000019611407,0.00018443105,0.0017603519,0.0011956684,0.00091842347],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9977654,0.0010316751,0.00014526372,0.0003688881,0.00053960853,0.0001491554],"domain_scores_gemma":[0.98919886,0.00714606,0.0005659761,0.0011244513,0.0016789176,0.00028561294],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004341077,0.0016533863,0.0010215922,0.0026771254,0.00062763476,0.0014308521,0.0016894613,0.001157746,0.0014332257],"category_scores_gemma":[0.020150257,0.00045447747,0.0007255303,0.0024614902,0.000719267,0.0025591005,0.001399791,0.001148768,0.00049709645],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00056754105,0.00018459109,0.0036763,0.00020714232,0.00016147253,0.00007600396,0.00020579885,0.7937803,0.008726871,0.007063994,0.0018854984,0.18346436],"study_design_scores_gemma":[0.000011956168,0.00005483079,0.0003697651,0.0000041412673,0.000010861908,0.00001595247,0.000029582576,0.9935888,0.0032843978,0.0023053617,0.0003160878,0.00000829737],"about_ca_topic_score_codex":0.016461998,"about_ca_topic_score_gemma":0.019188719,"teacher_disagreement_score":0.016461998,"about_ca_system_score_codex":0.002512018,"about_ca_system_score_gemma":0.0020522117,"threshold_uncertainty_score":0.032732368},"labels":[],"label_agreement":null},{"id":"W2562276887","doi":"10.71781/11124","title":"From Word Embeddings to Large Vocabulary Neural Machine Translation","year":2015,"lang":"en","type":"dissertation","venue":"Open MIND","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Samsung; Compute Canada; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs; Canadian Institute for Advanced Research","keywords":"Word (group theory); Translation (biology); Machine translation; Computer science; Natural language processing; Vocabulary; Artificial intelligence; Speech recognition; Linguistics; Biology; Philosophy","score_opus":0.030846504206258094,"score_gpt":0.357611872142071,"score_spread":0.3267653679358129,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2562276887","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.036339957,0.0017425605,0.94470894,0.0007848817,0.00043738226,0.000133269,0.0011422265,0.0067204824,0.007990393],"genre_scores_gemma":[0.44617367,0.0017806162,0.5217845,0.00044542414,0.00038542552,0.000403826,0.0051078284,0.00091808924,0.023000708],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99910444,0.000350336,0.00007441839,0.00024554264,0.00015492148,0.00007031748],"domain_scores_gemma":[0.9979025,0.0010605007,0.000120678036,0.0005516748,0.00031986332,0.000044831322],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001030161,0.0012961685,0.0008156286,0.0013295595,0.0004617993,0.0017558493,0.0012251111,0.0012406524,0.010210124],"category_scores_gemma":[0.0066628666,0.0006123325,0.0009665066,0.0021428608,0.0006242418,0.0043945285,0.0018630347,0.0016654198,0.003993069],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037794924,0.00016382347,0.0009869066,0.00048983574,0.00016030025,0.00023452086,0.00021513703,0.08985279,0.011687349,0.046730172,0.013294617,0.83580667],"study_design_scores_gemma":[0.00007514749,0.00016010899,0.0007399461,0.00007363268,0.000055797063,0.00017926551,0.00017275906,0.8265895,0.010837914,0.1462345,0.014842089,0.000039329865],"about_ca_topic_score_codex":0.0038549514,"about_ca_topic_score_gemma":0.006758209,"teacher_disagreement_score":0.010210124,"about_ca_system_score_codex":0.000743985,"about_ca_system_score_gemma":0.00096289435,"threshold_uncertainty_score":0.034156263},"labels":[],"label_agreement":null},{"id":"W2562967975","doi":"10.18653/v1/d16-1141","title":"Poet Admits // Mute Cypher: Beam Search to find Mutually Enciphering Poetic Texts","year":2016,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"","keywords":"Poetry; Computer science; Art; Literature; Combinatorics; Artificial intelligence; Mathematics","score_opus":0.015059394736580118,"score_gpt":0.2773991250396684,"score_spread":0.26233973030308827,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2562967975","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.31026652,0.001683682,0.6524536,0.002881769,0.00023361023,0.00056425604,0.001934683,0.0070192534,0.022962645],"genre_scores_gemma":[0.4199178,0.00031541314,0.5674401,0.0010331258,0.00007611768,0.00045892692,0.0026939728,0.00064396055,0.007420566],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99933064,0.00031847158,0.000034496963,0.00012640494,0.00012906286,0.00006098447],"domain_scores_gemma":[0.99661046,0.0026042762,0.0001303437,0.00030124054,0.0002592293,0.00009446782],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00225585,0.0008002826,0.0009478267,0.0017867754,0.0010659868,0.0015184301,0.0015425311,0.00154579,0.010727468],"category_scores_gemma":[0.007763276,0.00062880764,0.00093851425,0.0014890529,0.0013075019,0.0020690784,0.0016769192,0.0011077854,0.0019675104],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013889895,0.00060467265,0.019341236,0.0006555987,0.00061101496,0.00058256433,0.0016185553,0.2956428,0.009837859,0.07917819,0.060866043,0.52967244],"study_design_scores_gemma":[0.00017642061,0.00015003455,0.00083346607,0.00005282241,0.000055799566,0.00010196614,0.0004482704,0.94378865,0.002616976,0.045004692,0.006747902,0.000022849703],"about_ca_topic_score_codex":0.0024186613,"about_ca_topic_score_gemma":0.005009129,"teacher_disagreement_score":0.010727468,"about_ca_system_score_codex":0.000594619,"about_ca_system_score_gemma":0.0011422292,"threshold_uncertainty_score":0.035886943},"labels":[],"label_agreement":null},{"id":"W2563175374","doi":"10.3968/9031","title":"A Corpus-Based Study on Collocation of Technical Words in EST","year":2016,"lang":"en","type":"article","venue":"Studies in literature and language","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Collocation (remote sensing); Plural; Linguistics; Meaning (existential); Computer science; Base (topology); Natural language processing; Lexicography; Coca; Corpus linguistics; Artificial intelligence; Mathematics; Psychology; Philosophy; Epistemology","score_opus":0.014245508023456354,"score_gpt":0.32977696525406813,"score_spread":0.31553145723061177,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2563175374","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.96679384,0.0010360767,0.008985831,0.00025289142,0.00008174875,0.00012946952,0.0016127771,0.000054432632,0.02105293],"genre_scores_gemma":[0.9782591,0.0012216191,0.012476584,0.000061471896,0.00004534746,0.0001812814,0.0022099523,0.00006316704,0.005481335],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9985688,0.00056163885,0.00022767155,0.00025620294,0.00031496913,0.00007066435],"domain_scores_gemma":[0.99025947,0.006301787,0.0008394909,0.0008605598,0.0015374288,0.00020127674],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001872617,0.00019386847,0.00026072506,0.006457803,0.0022894114,0.0011845874,0.00039231544,0.0003800727,0.0026189887],"category_scores_gemma":[0.007328465,0.00016469441,0.00018049568,0.0107024945,0.0013171353,0.0017874575,0.001398298,0.00051627547,0.0004997558],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006423476,0.000600467,0.24117483,0.0036010882,0.00008729377,0.008066139,0.31845725,0.0015419731,0.049600918,0.029359492,0.011181716,0.33568653],"study_design_scores_gemma":[0.000031303014,0.0003419161,0.56112176,0.0006412866,0.00015306009,0.006334824,0.18801898,0.004345919,0.020237127,0.0032964896,0.21537526,0.000102132166],"about_ca_topic_score_codex":0.0036002262,"about_ca_topic_score_gemma":0.011492695,"teacher_disagreement_score":0.006457803,"about_ca_system_score_codex":0.0009064444,"about_ca_system_score_gemma":0.00132197,"threshold_uncertainty_score":0.009903431},"labels":[],"label_agreement":null},{"id":"W2564253719","doi":"10.18653/v1/d16-1010","title":"Comparing Computational Cognitive Models of Generalization in a Language Acquisition Task","year":2016,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Connectionism; Generalization; Artificial intelligence; Task (project management); Testbed; Cognition; Exploit; Machine learning; Language acquisition; Natural language; Cognitive model; Bayesian probability; Replicate; Computational model; Natural language processing; Artificial neural network; Psychology; Mathematics","score_opus":0.01865517472419621,"score_gpt":0.2793974344039989,"score_spread":0.2607422596798027,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2564253719","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.90153,0.00053567655,0.07493755,0.0031225942,0.00009027567,0.00010536805,0.00026431258,0.00033508372,0.019079033],"genre_scores_gemma":[0.982016,0.00019554858,0.016135661,0.00024659216,0.000030775773,0.00010247777,0.0002391875,0.000052435847,0.000981223],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9977149,0.0012233942,0.00012732763,0.0004581524,0.00032428093,0.00015198902],"domain_scores_gemma":[0.9614834,0.032568768,0.0016164966,0.00229328,0.0011121117,0.0009259506],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0068693445,0.0006740241,0.0007711117,0.001026856,0.00052651047,0.004365183,0.0026371453,0.0019732043,0.0038960958],"category_scores_gemma":[0.03834265,0.00064884487,0.0011778701,0.0007395176,0.0015231902,0.006802137,0.0017503193,0.0027583723,0.0005454422],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0028553102,0.0019435728,0.040126648,0.000637961,0.0013262285,0.00028857062,0.0032083346,0.62219316,0.008890152,0.20181508,0.004689251,0.11202571],"study_design_scores_gemma":[0.0001243905,0.0002407503,0.007875046,0.000026408123,0.00012382108,0.000074026626,0.00021587533,0.8851189,0.0010818386,0.104438014,0.00061168725,0.000069278714],"about_ca_topic_score_codex":0.0076074856,"about_ca_topic_score_gemma":0.0059769494,"teacher_disagreement_score":0.0076074856,"about_ca_system_score_codex":0.0030021472,"about_ca_system_score_gemma":0.0013553449,"threshold_uncertainty_score":0.03632903},"labels":[],"label_agreement":null},{"id":"W2566433528","doi":"10.18653/v1/w16-6010","title":"Stylistic Transfer in Natural Language Generation Systems Using Recurrent Neural Networks","year":2016,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":34,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Natural language; Natural language processing; Artificial intelligence; Artificial neural network; Natural language generation; Recurrent neural network; Transfer (computing)","score_opus":0.022342534496859422,"score_gpt":0.2778172395894374,"score_spread":0.25547470509257797,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2566433528","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.53933346,0.00059989793,0.44762915,0.0006343919,0.00011377318,0.00034356015,0.00018079788,0.005683814,0.0054811616],"genre_scores_gemma":[0.9473136,0.00008672451,0.050935745,0.00006858197,0.000015441226,0.00010122196,0.00019860339,0.000085643376,0.001194373],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99868804,0.00063871633,0.00009438818,0.00025331377,0.00021370722,0.00011180102],"domain_scores_gemma":[0.99587977,0.002718305,0.00031800746,0.0004094367,0.00057118974,0.000103254715],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004077677,0.000626396,0.00053127995,0.0005976887,0.00037260662,0.0009785257,0.0010426313,0.0007127132,0.0013836735],"category_scores_gemma":[0.011118914,0.0003875262,0.00039714758,0.0003730447,0.0005905441,0.002108646,0.0009957834,0.0009750234,0.00035391634],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007027967,0.0004923333,0.0043794946,0.00019555047,0.00016225746,0.00022833569,0.0004957726,0.6612118,0.03536917,0.005625436,0.0014514043,0.2896857],"study_design_scores_gemma":[0.0000118687085,0.00007420257,0.0003827227,0.0000045510756,0.000012476245,0.000012155174,0.000015244143,0.99220955,0.0053516086,0.0017621807,0.00015587034,0.000007689207],"about_ca_topic_score_codex":0.0045827655,"about_ca_topic_score_gemma":0.00608155,"teacher_disagreement_score":0.0045827655,"about_ca_system_score_codex":0.00139344,"about_ca_system_score_gemma":0.0004910612,"threshold_uncertainty_score":0.02156508},"labels":[],"label_agreement":null},{"id":"W2566627449","doi":"10.18653/v1/w16-6635","title":"Ranking Automatically Generated Questions Using Common Human Queries","year":2016,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Lethbridge","funders":"","keywords":"Computer science; Ranking (information retrieval); Information retrieval; Artificial intelligence","score_opus":0.024061300579327902,"score_gpt":0.3147514465918748,"score_spread":0.2906901460125469,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2566627449","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1737839,0.0028604853,0.7866438,0.0016096394,0.0004571032,0.002034444,0.0038757238,0.016952459,0.011782417],"genre_scores_gemma":[0.55243754,0.0006300114,0.42146477,0.0006515328,0.00036842583,0.000799308,0.016301079,0.0011604321,0.006186897],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9862126,0.008127088,0.0006834497,0.0020695215,0.00244979,0.00045760052],"domain_scores_gemma":[0.9641585,0.02552521,0.0014330421,0.0027460225,0.005295399,0.0008418963],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007670849,0.0021192248,0.0016585735,0.0045787906,0.0009573386,0.0026322273,0.001736007,0.002639473,0.008035677],"category_scores_gemma":[0.036969937,0.00039933773,0.001405488,0.0015388539,0.0007097235,0.0032533323,0.0020171048,0.0015542352,0.003552109],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0023877658,0.0017790704,0.025120566,0.0025301643,0.0005559746,0.0014919273,0.0024021894,0.068886824,0.0784112,0.0198651,0.0526955,0.74387383],"study_design_scores_gemma":[0.00059010525,0.0017809782,0.013483646,0.00020774626,0.00038378162,0.0012222368,0.0016688132,0.821776,0.05869146,0.04419408,0.05579938,0.00020188547],"about_ca_topic_score_codex":0.0019099015,"about_ca_topic_score_gemma":0.0035197104,"teacher_disagreement_score":0.008035677,"about_ca_system_score_codex":0.0010981738,"about_ca_system_score_gemma":0.0014006337,"threshold_uncertainty_score":0.040567815},"labels":[],"label_agreement":null},{"id":"W2566814039","doi":"10.18653/v1/w16-3512","title":"Automatic Tweet Generation From Traffic Incident Data","year":2016,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Natural language generation; Natural language; Information retrieval; Code (set theory); Artificial intelligence; World Wide Web; Language model; Data modeling; Natural language processing; Database; Programming language","score_opus":0.041078762033670016,"score_gpt":0.294360388551446,"score_spread":0.25328162651777597,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2566814039","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.33996364,0.00049827417,0.5982228,0.0016974028,0.00080750586,0.0022307672,0.021157024,0.023312885,0.012109821],"genre_scores_gemma":[0.52249956,0.0003769617,0.43449324,0.00014513085,0.00020961618,0.0009881334,0.036082976,0.0006815812,0.004522785],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989938,0.00035621115,0.00008961037,0.00021628015,0.0002805522,0.00006361991],"domain_scores_gemma":[0.994212,0.0039492217,0.00034059305,0.0005202173,0.00088308414,0.000094845374],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012719907,0.0007153064,0.00048409632,0.0033168194,0.0007131267,0.0008455044,0.00085376855,0.0006235562,0.0034351123],"category_scores_gemma":[0.009430884,0.00035943958,0.0005843438,0.00240076,0.0002818427,0.0012862575,0.00088725734,0.00082904164,0.001877856],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012526446,0.00090839196,0.025055736,0.0018933796,0.00025210704,0.0035659263,0.0025905136,0.07407716,0.08136881,0.0141864,0.04515441,0.74969447],"study_design_scores_gemma":[0.00015041817,0.0002789544,0.007344113,0.00011459298,0.000110717054,0.0008427614,0.0010162072,0.8579313,0.07431538,0.01180213,0.045998838,0.000094510586],"about_ca_topic_score_codex":0.0022286272,"about_ca_topic_score_gemma":0.0030067218,"teacher_disagreement_score":0.0034351123,"about_ca_system_score_codex":0.000571557,"about_ca_system_score_gemma":0.00065680617,"threshold_uncertainty_score":0.011491597},"labels":[],"label_agreement":null},{"id":"W2569679643","doi":"10.71781/10791","title":"Mémoires de traduction sous-phrastiques","year":2003,"lang":"fr","type":"dissertation","venue":"Open MIND","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Philosophy","score_opus":0.03414094469381923,"score_gpt":0.34224429446641674,"score_spread":0.3081033497725975,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2569679643","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.032690868,0.0637344,0.5738381,0.023548005,0.01333192,0.00031989618,0.0038346746,0.005323964,0.28337815],"genre_scores_gemma":[0.3489222,0.050727107,0.25139108,0.0048348797,0.008295518,0.00039009753,0.009272802,0.004910938,0.3212554],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9977253,0.0008884811,0.00017773214,0.00042747098,0.00060501555,0.00017599721],"domain_scores_gemma":[0.99460346,0.002276066,0.0002258487,0.0013244922,0.0013986382,0.00017151507],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032037897,0.0010421819,0.0009849853,0.00331777,0.0023968518,0.0058258725,0.0018423463,0.0012670314,0.033850834],"category_scores_gemma":[0.0143126035,0.0007154751,0.001197539,0.0028412072,0.0044298857,0.009708018,0.0022501606,0.0039359024,0.0066253934],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023068544,0.000060580074,0.0010944572,0.001019046,0.00012271601,0.00051310094,0.0107875075,0.00071268703,0.006920851,0.64786935,0.07617844,0.25449058],"study_design_scores_gemma":[0.000045807246,0.00005213741,0.0028641426,0.0004288111,0.00007868483,0.0011226257,0.002300439,0.0029031653,0.0056376993,0.13394,0.8505378,0.00008871155],"about_ca_topic_score_codex":0.024763383,"about_ca_topic_score_gemma":0.030249711,"teacher_disagreement_score":0.033850834,"about_ca_system_score_codex":0.0034292422,"about_ca_system_score_gemma":0.0030420015,"threshold_uncertainty_score":0.11324227},"labels":[],"label_agreement":null},{"id":"W2572497595","doi":"","title":"Grammar Induction as Automated Transformation between Constraint Solving Models of Language.","year":2016,"lang":"en","type":"article","venue":"International Joint Conference on Artificial Intelligence","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Transformation (genetics); Programming language; Grammar; Natural language processing; Model transformation; Constraint (computer-aided design); Artificial intelligence; Linguistics; Mathematics; Philosophy; Chemistry","score_opus":0.08747857146460501,"score_gpt":0.3393896233222696,"score_spread":0.25191105185766455,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2572497595","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009015708,0.00011715292,0.9684019,0.00035361134,0.00008920484,0.0001965913,0.00088693533,0.012951272,0.007987593],"genre_scores_gemma":[0.20555142,0.00017131703,0.780556,0.00028783784,0.000053899792,0.00026613424,0.004186119,0.0021448755,0.0067823757],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9983494,0.00055973936,0.00008817462,0.00031912402,0.0005509098,0.00013263834],"domain_scores_gemma":[0.99645424,0.0022347723,0.00015763704,0.00062312064,0.0004609152,0.00006930598],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011210621,0.0005722389,0.00049583573,0.0009526028,0.00056146644,0.001905508,0.0022448355,0.0008053738,0.008968469],"category_scores_gemma":[0.007705161,0.0005829225,0.0015365887,0.0010810395,0.0009981617,0.0020244797,0.0025113355,0.0023399754,0.0029518213],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035864874,0.00044914504,0.0020502813,0.0005710957,0.00019301685,0.0008372724,0.0008769085,0.15039921,0.022020537,0.36300227,0.034425333,0.42481622],"study_design_scores_gemma":[0.00007274206,0.000047488167,0.00025796975,0.000049881633,0.00005102429,0.00012932724,0.00016616337,0.73525596,0.013382759,0.23251078,0.01805364,0.000022273605],"about_ca_topic_score_codex":0.0042187,"about_ca_topic_score_gemma":0.0079962285,"teacher_disagreement_score":0.008968469,"about_ca_system_score_codex":0.0008996706,"about_ca_system_score_gemma":0.0020881472,"threshold_uncertainty_score":0.030002534},"labels":[],"label_agreement":null},{"id":"W2572785564","doi":"","title":"WaterlooClarke: TREC 2015 Temporal Summarization Track.","year":2015,"lang":"en","type":"article","venue":"Text REtrieval Conference","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Automatic summarization; Computer science; Track (disk drive); Multi-document summarization; Information retrieval","score_opus":0.04169239849904795,"score_gpt":0.2977835804427729,"score_spread":0.256091181943725,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2572785564","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009614935,0.018729996,0.06440945,0.012817989,0.007839403,0.0025856982,0.7714569,0.041998766,0.070546925],"genre_scores_gemma":[0.009979734,0.003177332,0.050174832,0.0014094426,0.00078336836,0.00086255866,0.84698814,0.002317729,0.08430685],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9978922,0.00054559537,0.00022190271,0.0003373482,0.00079961005,0.0002033503],"domain_scores_gemma":[0.99140036,0.0010731226,0.00038889045,0.0008902984,0.0056531397,0.00059424504],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004229312,0.0023390015,0.0019035372,0.0059315814,0.0024660842,0.0039560874,0.0032895803,0.0015338631,0.05565622],"category_scores_gemma":[0.008653516,0.00068179757,0.0009297352,0.0054818555,0.00085570576,0.005237808,0.0018525064,0.002205516,0.038103264],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000897623,0.00004223484,0.00009397145,0.00033925514,0.00003131018,0.000021955238,0.000030912724,0.00025531228,0.0021225663,0.00045038894,0.97238535,0.024136864],"study_design_scores_gemma":[0.00028360524,0.00016300156,0.0043053,0.0002951408,0.00019463334,0.00013565016,0.0003338899,0.0083240755,0.0143377045,0.0035921119,0.9679184,0.000116412775],"about_ca_topic_score_codex":0.2007812,"about_ca_topic_score_gemma":0.39428198,"teacher_disagreement_score":0.2007812,"about_ca_system_score_codex":0.004214403,"about_ca_system_score_gemma":0.010208433,"threshold_uncertainty_score":0.39922506},"labels":[],"label_agreement":null},{"id":"W2573569263","doi":"10.63317/2vji4nx2bvuq","title":"The Alaskan Athabascan Grammar Database","year":2016,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"First Nations University of Canada","funders":"","keywords":"Computer science; Grammar; Bridge (graph theory); Sentence; World Wide Web; Database; Natural language processing; Programming language; Information retrieval; Linguistics","score_opus":0.009390038554562377,"score_gpt":0.2491389669054062,"score_spread":0.23974892835084383,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2573569263","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.041041944,0.00069286994,0.0062433477,0.000410196,0.000089929454,0.00012692444,0.9102529,0.01075038,0.030391498],"genre_scores_gemma":[0.05313596,0.00052247546,0.012839443,0.00014368727,0.000018935876,0.00014867116,0.9259577,0.0010008985,0.0062323753],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99972576,0.000037949787,0.000041430972,0.00008574968,0.00008469362,0.000024326273],"domain_scores_gemma":[0.9989065,0.000290556,0.00008394485,0.00028431372,0.0003295003,0.00010520591],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00038492272,0.00065819634,0.0003915454,0.00421725,0.000805791,0.0010019686,0.0012303407,0.0006675056,0.022500876],"category_scores_gemma":[0.0023960744,0.00029098697,0.0004808978,0.003549086,0.00045849994,0.0010477968,0.0010773983,0.0006699798,0.012316107],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004415193,0.00020566244,0.017318586,0.0012650266,0.00015050775,0.0015528599,0.0011150175,0.009306369,0.0043076724,0.013082885,0.80986077,0.14139323],"study_design_scores_gemma":[0.00016495354,0.000042721087,0.022680119,0.00029780343,0.000140827,0.00065440097,0.00092212844,0.011424597,0.0028422768,0.01139686,0.9493486,0.00008466798],"about_ca_topic_score_codex":0.06016548,"about_ca_topic_score_gemma":0.1253275,"teacher_disagreement_score":0.06016548,"about_ca_system_score_codex":0.0007026705,"about_ca_system_score_gemma":0.0030778488,"threshold_uncertainty_score":0.119630516},"labels":[],"label_agreement":null},{"id":"W2573843450","doi":"","title":"Determining the Multiword Expression Inventory of a Surprise Language","year":2016,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of New Brunswick","funders":"","keywords":"Treebank; Surprise; Computer science; Natural language processing; Identification (biology); Artificial intelligence; Language model; Natural language; Language identification; Parsing; Psychology","score_opus":0.03742044249948087,"score_gpt":0.33147871213364416,"score_spread":0.29405826963416326,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2573843450","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.80091864,0.0006407504,0.18112165,0.0004942706,0.00015683915,0.00011328266,0.0028509342,0.0028599547,0.010843648],"genre_scores_gemma":[0.91772896,0.00038561493,0.07061923,0.00013544512,0.00006157615,0.00012357272,0.006808242,0.00057130755,0.0035660733],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9993356,0.00016302496,0.00006942808,0.00028134414,0.00008590304,0.00006463298],"domain_scores_gemma":[0.99782526,0.00092499296,0.00025732664,0.00024599748,0.000640376,0.00010607924],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010755688,0.0006038567,0.0004031326,0.0016019614,0.0004222373,0.0012200215,0.00046618594,0.00055127486,0.0032493253],"category_scores_gemma":[0.004425005,0.00055781094,0.00058325933,0.0006481379,0.0004697897,0.0031078944,0.0010855312,0.0013297905,0.0026374087],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009598696,0.00031213087,0.09591027,0.00087991747,0.00012635907,0.001935286,0.0051845545,0.0074650715,0.29911244,0.020927807,0.017647715,0.54953855],"study_design_scores_gemma":[0.000118129705,0.00083939516,0.20166504,0.00035674596,0.00033176076,0.006289718,0.008326554,0.5316431,0.1380048,0.03604274,0.076112986,0.00026906858],"about_ca_topic_score_codex":0.0012790109,"about_ca_topic_score_gemma":0.0016386814,"teacher_disagreement_score":0.0032493253,"about_ca_system_score_codex":0.0006053099,"about_ca_system_score_gemma":0.00080512214,"threshold_uncertainty_score":0.010870039},"labels":[],"label_agreement":null},{"id":"W2574762171","doi":"10.63317/3yjtmzi2qxqk","title":"WikiCoref: An English Coreference-annotated Corpus of Wikipedia Articles","year":2016,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":69,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Coreference; Computer science; Annotation; Natural language processing; Information retrieval; Artificial intelligence; Resource (disambiguation); Resolution (logic); Scheme (mathematics); World Wide Web","score_opus":0.01968518263512315,"score_gpt":0.25574076841985616,"score_spread":0.236055585784733,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2574762171","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11135648,0.0068774573,0.045130428,0.0013712825,0.0020452219,0.0014343213,0.75209856,0.012526209,0.067160085],"genre_scores_gemma":[0.07050043,0.0013800901,0.06493715,0.00039189256,0.00026237592,0.0010274774,0.8494106,0.0016896371,0.010400423],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99714017,0.00071508583,0.00050995144,0.00076949096,0.00069673493,0.0001686876],"domain_scores_gemma":[0.9893498,0.0047474722,0.0007536631,0.0013109647,0.0032484308,0.0005897159],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020023696,0.0011849271,0.0008596403,0.010843727,0.0029976692,0.0023216787,0.0016301546,0.0016652237,0.013275199],"category_scores_gemma":[0.013285315,0.0006641602,0.00045105646,0.00884761,0.0009445228,0.003134791,0.0032605985,0.001436869,0.009535578],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010514025,0.00066326617,0.008109978,0.013095465,0.00026949705,0.0032696975,0.007345802,0.0019868142,0.037907586,0.015693909,0.7266907,0.18391582],"study_design_scores_gemma":[0.00015865614,0.00009009058,0.02098869,0.0009533069,0.0001942617,0.002188702,0.002426453,0.0048046797,0.0205946,0.004052045,0.9433961,0.00015237047],"about_ca_topic_score_codex":0.009190063,"about_ca_topic_score_gemma":0.022980705,"teacher_disagreement_score":0.013275199,"about_ca_system_score_codex":0.0008462044,"about_ca_system_score_gemma":0.0039520008,"threshold_uncertainty_score":0.04440999},"labels":[],"label_agreement":null},{"id":"W2574986170","doi":"","title":"plWordNet 3.0 - a Comprehensive Lexical-Semantic Resource.","year":2016,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":32,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"WordNet; Computer science; Set (abstract data type); Natural language processing; Resource (disambiguation); Word (group theory); Artificial intelligence; Lexical database; Information retrieval; Word list; Linguistics; Class (philosophy); Programming language","score_opus":0.04651737362330788,"score_gpt":0.32945800977683776,"score_spread":0.2829406361535299,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2574986170","genre_codex":"dataset","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01812159,0.0043238923,0.37292784,0.0031946073,0.0015553829,0.0013566731,0.39583597,0.09180233,0.11088179],"genre_scores_gemma":[0.040277842,0.0031449562,0.19703236,0.0012388609,0.00026345393,0.002117562,0.6900859,0.023345282,0.04249373],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9986473,0.00024207823,0.0002689118,0.00035198705,0.0003898349,0.00009986013],"domain_scores_gemma":[0.9979882,0.00045504473,0.00016281432,0.00042967775,0.0007878982,0.00017620248],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018745885,0.0019502215,0.0010680539,0.0066455184,0.0014938,0.0036403167,0.0016951722,0.0014085076,0.037200913],"category_scores_gemma":[0.007872437,0.0013375501,0.0008754915,0.006012694,0.0007313044,0.013284072,0.0048596514,0.002190015,0.05351034],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00038589875,0.00010963315,0.0018581658,0.0032268628,0.00013791633,0.0006397309,0.001483224,0.0018608996,0.011038186,0.070496276,0.6297829,0.2789803],"study_design_scores_gemma":[0.000029969175,0.000025405638,0.0007516993,0.0002950037,0.00003743861,0.00034810635,0.0002850393,0.0021875198,0.004871569,0.022608325,0.968511,0.000048882834],"about_ca_topic_score_codex":0.0050150836,"about_ca_topic_score_gemma":0.006344342,"teacher_disagreement_score":0.037200913,"about_ca_system_score_codex":0.0008579882,"about_ca_system_score_gemma":0.0031997978,"threshold_uncertainty_score":0.12444937},"labels":[],"label_agreement":null},{"id":"W2575224319","doi":"","title":"Capturing Pragmatic Knowledge in Article Usage Prediction using LSTMs.","year":2016,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Interpretability; Computer science; Coreference; Artificial intelligence; Task (project management); Machine learning; Mechanism (biology); Recurrent neural network; Natural language processing; Long short term memory; Artificial neural network; Resolution (logic)","score_opus":0.04258849303752668,"score_gpt":0.33589017408603444,"score_spread":0.29330168104850773,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2575224319","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.378333,0.002924536,0.58922976,0.002421591,0.00045761664,0.00023856133,0.0038621568,0.007980743,0.014551967],"genre_scores_gemma":[0.905984,0.0004858828,0.088192746,0.00016567377,0.00012017958,0.000094037576,0.0025892751,0.0001677663,0.0022003925],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992555,0.000290092,0.000056592744,0.00020219614,0.00013182574,0.00006377717],"domain_scores_gemma":[0.9964406,0.0024017717,0.00039957938,0.00021837941,0.00046094053,0.000078805984],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013271462,0.0008343553,0.0003919166,0.0014700742,0.000326522,0.0012509866,0.0010205524,0.0010732159,0.0021836741],"category_scores_gemma":[0.01060308,0.00043155416,0.0004821419,0.0012321074,0.00038165392,0.0034637265,0.0007403807,0.0014256951,0.0011142865],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00071927434,0.0004430556,0.025434937,0.0013959596,0.0003934377,0.0009449815,0.0013690677,0.13803978,0.06548867,0.0110616805,0.013448809,0.7412604],"study_design_scores_gemma":[0.000014545089,0.000059000096,0.004229909,0.000055818793,0.000054742446,0.000120305216,0.00017003415,0.97183496,0.008592524,0.011729555,0.0031121972,0.00002638152],"about_ca_topic_score_codex":0.0051003965,"about_ca_topic_score_gemma":0.012786793,"teacher_disagreement_score":0.0051003965,"about_ca_system_score_codex":0.0007990486,"about_ca_system_score_gemma":0.00077859557,"threshold_uncertainty_score":0.010141432},"labels":[],"label_agreement":null},{"id":"W2575954684","doi":"","title":"Mixed Script Ad hoc Retrieval using back transliteration and phrase matching through bigram indexing: Shared Task report by BIT, Mesra.","year":2015,"lang":"en","type":"article","venue":"FIRE Workshops","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Bigram; Computer science; Transliteration; Search engine indexing; Natural language processing; Artificial intelligence; Phrase; Matching (statistics); Speech recognition; Mathematics","score_opus":0.04587822017000514,"score_gpt":0.2929745044033585,"score_spread":0.24709628423335334,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2575954684","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.118403316,0.0060590734,0.69519556,0.002347172,0.0020073978,0.0042281104,0.039554108,0.110384956,0.021820322],"genre_scores_gemma":[0.17830077,0.0012707537,0.66524225,0.00087046967,0.0005734598,0.0015558794,0.102227084,0.008636092,0.04132318],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99551845,0.0016625971,0.00053165975,0.000977927,0.000856203,0.00045304696],"domain_scores_gemma":[0.9881434,0.0036691783,0.0003446374,0.0035134715,0.003660191,0.00066911854],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0054823156,0.0031190075,0.003683211,0.0031023,0.0021778261,0.0044653798,0.0043262285,0.0028550623,0.023375789],"category_scores_gemma":[0.01062635,0.0013518912,0.0023087205,0.003103194,0.000860906,0.00815878,0.004352946,0.0021283266,0.026835842],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0022010375,0.0009840823,0.0012595888,0.0025686128,0.000554403,0.0007045084,0.00068032066,0.0035195332,0.12324657,0.003323512,0.1976253,0.66333246],"study_design_scores_gemma":[0.0018843414,0.0034826326,0.006999849,0.0003704265,0.0016602853,0.0028318136,0.0033325767,0.34495723,0.42206705,0.023267338,0.188306,0.0008404589],"about_ca_topic_score_codex":0.008005016,"about_ca_topic_score_gemma":0.0112812,"teacher_disagreement_score":0.023375789,"about_ca_system_score_codex":0.0009783412,"about_ca_system_score_gemma":0.0043817335,"threshold_uncertainty_score":0.078199744},"labels":[],"label_agreement":null},{"id":"W2576767807","doi":"10.63317/2deq34sevsu2","title":"A sense-based lexicon of count and mass expressions: The Bochum English Countability Lexicon","year":2016,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"WordNet; Noun; Lexicon; Natural language processing; Computer science; Artificial intelligence; Countable set; Linguistics; Lexical database; Mathematics; Combinatorics","score_opus":0.01151219341312847,"score_gpt":0.2503676765683688,"score_spread":0.2388554831552403,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2576767807","genre_codex":"methods","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.044581275,0.0017853603,0.8060731,0.002799659,0.0010361416,0.00070075825,0.03483583,0.016254915,0.09193292],"genre_scores_gemma":[0.5224845,0.0014226639,0.4079026,0.0010473083,0.00049731793,0.0009783386,0.04262823,0.0053238166,0.01771522],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9989311,0.00025030292,0.00022880478,0.0002926151,0.00022133988,0.00007578159],"domain_scores_gemma":[0.9981128,0.00067354005,0.0001822938,0.0002727053,0.0006339397,0.0001247661],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00087445887,0.0009378394,0.0010921739,0.005444003,0.0017485848,0.004382992,0.0012965698,0.0009106464,0.013821208],"category_scores_gemma":[0.003869439,0.000957495,0.000831647,0.0043487716,0.0016582665,0.006358715,0.0021681767,0.0019366975,0.005127495],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00030939834,0.00011693928,0.0023295698,0.0011617878,0.00006423578,0.0008725196,0.0034118686,0.0017506758,0.0190553,0.7608123,0.086028844,0.1240866],"study_design_scores_gemma":[0.00012103958,0.00009213723,0.005272527,0.0005873426,0.00021381376,0.0022160076,0.0024212904,0.03479287,0.016892355,0.27684394,0.66034037,0.00020631157],"about_ca_topic_score_codex":0.0046675103,"about_ca_topic_score_gemma":0.0061505367,"teacher_disagreement_score":0.013821208,"about_ca_system_score_codex":0.0015725116,"about_ca_system_score_gemma":0.0022901772,"threshold_uncertainty_score":0.046236575},"labels":[],"label_agreement":null},{"id":"W2576880689","doi":"","title":"Design Rationale for Natural Language Game Conversations.","year":2015,"lang":"en","type":"article","venue":"Foundations of Digital Games","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Natural (archaeology); Natural language; Game design; Human–computer interaction; Natural language processing; History","score_opus":0.040923176900366946,"score_gpt":0.3025134378698438,"score_spread":0.26159026096947685,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2576880689","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0052581537,0.000420105,0.9436499,0.004259894,0.00027584785,0.00066800235,0.00035180297,0.00088847644,0.04422789],"genre_scores_gemma":[0.30281374,0.00048890273,0.67286396,0.0014426457,0.00013505264,0.0025105036,0.00067954115,0.00051093387,0.01855473],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9860686,0.007934391,0.0010629358,0.0014387281,0.002877418,0.0006178154],"domain_scores_gemma":[0.98083824,0.011710481,0.0006345229,0.002214808,0.003945691,0.0006563627],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010671182,0.0010879118,0.0005260108,0.0018691408,0.0024408635,0.0060384655,0.0033038417,0.0036324176,0.016938487],"category_scores_gemma":[0.039279476,0.0012602636,0.0011457944,0.0010100786,0.0048937164,0.007062338,0.002961567,0.004145143,0.0043553873],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000040254876,0.00004976159,0.00016079054,0.00018360563,0.000011303898,0.00007138597,0.00066844176,0.0013255584,0.0010062754,0.9815704,0.0033148555,0.011597401],"study_design_scores_gemma":[0.000101883845,0.000078231715,0.00021906086,0.0003628274,0.000076302815,0.00031716164,0.0008743082,0.040527824,0.004662659,0.8565763,0.09615439,0.000049094735],"about_ca_topic_score_codex":0.0030181517,"about_ca_topic_score_gemma":0.0036415844,"teacher_disagreement_score":0.016938487,"about_ca_system_score_codex":0.002551774,"about_ca_system_score_gemma":0.0042731455,"threshold_uncertainty_score":0.056664824},"labels":[],"label_agreement":null},{"id":"W2576945865","doi":"","title":"The Importance of Discourse Context for Statistical Natural Language Generation","year":2004,"lang":"en","type":"article","venue":"Annual Meeting of the Special Interest Group on Discourse and Dialogue","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Realization (probability); Natural language generation; Natural language processing; Word (group theory); Word order; Natural language; Context (archaeology); Artificial intelligence; Natural (archaeology); Selection (genetic algorithm); Word lists by frequency; Linguistics; Statistics; Mathematics; History","score_opus":0.019652964511517327,"score_gpt":0.3155800113530325,"score_spread":0.29592704684151516,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2576945865","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16399238,0.0056342576,0.8092653,0.0031548166,0.00039165746,0.00022572758,0.00044333102,0.0016584692,0.01523414],"genre_scores_gemma":[0.8260214,0.0011343563,0.17074719,0.0002443975,0.00038616103,0.00015998566,0.0002711836,0.00033496143,0.0007004293],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9927759,0.0047177556,0.00037926115,0.0010201472,0.00090132176,0.00020555906],"domain_scores_gemma":[0.9474277,0.044105165,0.0017266619,0.0038177012,0.0022275846,0.0006951974],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0056802644,0.00073813094,0.0010707068,0.0021113248,0.00208573,0.0041381847,0.0008082642,0.001406598,0.003475006],"category_scores_gemma":[0.059439432,0.00080977357,0.0005559042,0.0019068597,0.002078083,0.009077953,0.0030405615,0.0021457921,0.0009091032],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012339762,0.0002746866,0.016671697,0.0010474077,0.00015208151,0.0007489878,0.0036058682,0.037443496,0.030075205,0.36938477,0.0033724918,0.53598934],"study_design_scores_gemma":[0.00013695104,0.00036030833,0.00868271,0.00029555286,0.00019421705,0.0006681973,0.0010387577,0.26046747,0.016482363,0.6908447,0.020624923,0.00020382852],"about_ca_topic_score_codex":0.0011885144,"about_ca_topic_score_gemma":0.0025957217,"teacher_disagreement_score":0.0056802644,"about_ca_system_score_codex":0.00076851,"about_ca_system_score_gemma":0.0011099841,"threshold_uncertainty_score":0.030040443},"labels":[],"label_agreement":null},{"id":"W2577032177","doi":"","title":"Learning in second-language searching","year":2016,"lang":"en","type":"article","venue":"International ACM SIGIR Conference on Research and Development in Information Retrieval","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence","score_opus":0.04934505363614039,"score_gpt":0.3550492955381386,"score_spread":0.3057042419019982,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2577032177","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4785658,0.0033581369,0.42021325,0.0045410055,0.00024699816,0.00019998742,0.00050030363,0.002020472,0.09035409],"genre_scores_gemma":[0.93705636,0.00053867605,0.042888384,0.00034179894,0.00007854076,0.00005927414,0.00055640313,0.00013194939,0.018348709],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99858874,0.00064024323,0.00007536053,0.0003735444,0.0001950692,0.00012698612],"domain_scores_gemma":[0.9898252,0.0077003962,0.00032148874,0.0011966478,0.0006186595,0.00033753642],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017561427,0.00041086873,0.0008787969,0.001457567,0.0010339633,0.003112584,0.0014347832,0.0014895748,0.014319831],"category_scores_gemma":[0.015621842,0.0004829372,0.0006859789,0.0015967807,0.0016766374,0.011582313,0.0020824475,0.0017931935,0.002170807],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00082709634,0.0013779268,0.027448747,0.00063760096,0.00016603376,0.00042852352,0.0032405085,0.021807007,0.008974074,0.26714662,0.0117142815,0.6562316],"study_design_scores_gemma":[0.0001815877,0.000566191,0.008109472,0.00012978428,0.000101848455,0.00083225715,0.002436485,0.33006108,0.01454972,0.6212059,0.02175048,0.000075099335],"about_ca_topic_score_codex":0.0051689935,"about_ca_topic_score_gemma":0.004935798,"teacher_disagreement_score":0.014319831,"about_ca_system_score_codex":0.0012253375,"about_ca_system_score_gemma":0.0013604915,"threshold_uncertainty_score":0.04790461},"labels":[],"label_agreement":null},{"id":"W2579409202","doi":"10.21700/ijcis.2016.119","title":"Automatic Diacritics Restoration for Dialectal Arabic Text","year":2016,"lang":"en","type":"article","venue":"International Journal of Computing and Information Sciences","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Arabic; Linguistics; Natural language processing; Artificial intelligence; Computer science; Philosophy","score_opus":0.013769201739082484,"score_gpt":0.31638996346140125,"score_spread":0.3026207617223188,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2579409202","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.26887178,0.0048181624,0.64137465,0.0011872238,0.0024042167,0.00059843226,0.005780913,0.056399666,0.018565029],"genre_scores_gemma":[0.45744404,0.0014693016,0.51014704,0.00016493081,0.0004179299,0.00016974859,0.008282709,0.0024286439,0.019475635],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9994179,0.00007712914,0.000049725655,0.00024585758,0.00013600067,0.00007333538],"domain_scores_gemma":[0.9988809,0.00021536578,0.00011022008,0.00021791525,0.00050068554,0.000074830976],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00060645276,0.0013740838,0.0008987859,0.003199238,0.0013274052,0.0013857881,0.00080422533,0.0008348408,0.008391187],"category_scores_gemma":[0.0015246854,0.00035553382,0.00086657284,0.0013962467,0.0005561991,0.0012320364,0.0010314186,0.0013912845,0.009138676],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012615676,0.00015945373,0.0015420461,0.0007628769,0.000053244108,0.00054882886,0.00037888717,0.0018341473,0.17588513,0.0027497862,0.015844455,0.7989795],"study_design_scores_gemma":[0.0002114877,0.0005910737,0.0157393,0.00021008818,0.00043268583,0.0023474847,0.0024271628,0.42006996,0.4242622,0.010641041,0.12289256,0.00017499152],"about_ca_topic_score_codex":0.002303916,"about_ca_topic_score_gemma":0.0040391134,"teacher_disagreement_score":0.008391187,"about_ca_system_score_codex":0.0003443778,"about_ca_system_score_gemma":0.00097293407,"threshold_uncertainty_score":0.028071284},"labels":[],"label_agreement":null},{"id":"W2579534470","doi":"","title":"Training Data Enrichment for Infrequent Discourse Relations","year":2016,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Parsing; Computer science; Training set; Relation (database); Natural language processing; Artificial intelligence; Training (meteorology); Confidence interval; Quality (philosophy); Machine learning; Data mining; Statistics","score_opus":0.1427233374709916,"score_gpt":0.40637640218257554,"score_spread":0.2636530647115839,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2579534470","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.33163306,0.0012241778,0.6375189,0.0015459607,0.0003439669,0.0007044203,0.0025693637,0.019973714,0.004486403],"genre_scores_gemma":[0.47832695,0.00019304316,0.5049353,0.00064299227,0.00014967001,0.00072177214,0.010426529,0.0010708482,0.0035328374],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99207824,0.0034423806,0.0006091158,0.0025194378,0.00094415585,0.00040674573],"domain_scores_gemma":[0.94682914,0.038767457,0.0017767797,0.007707406,0.004270123,0.0006490824],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009819563,0.0019215707,0.0016492169,0.002508579,0.0017553864,0.0016713565,0.0026755268,0.003472533,0.0031077168],"category_scores_gemma":[0.043632623,0.0010359377,0.001568262,0.0021134927,0.0013829153,0.004070283,0.0035791784,0.0041537695,0.0032104936],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015062721,0.0023785082,0.049404297,0.0010738946,0.0002886457,0.0013622941,0.0029370964,0.053795483,0.067340426,0.006139383,0.027295647,0.78647804],"study_design_scores_gemma":[0.00015824207,0.00059235527,0.009649676,0.00023885227,0.00023224369,0.0008444557,0.001198024,0.8404607,0.10549368,0.0147949355,0.026238633,0.00009831866],"about_ca_topic_score_codex":0.002563991,"about_ca_topic_score_gemma":0.0054675345,"teacher_disagreement_score":0.009819563,"about_ca_system_score_codex":0.00094772334,"about_ca_system_score_gemma":0.0021765863,"threshold_uncertainty_score":0.05193144},"labels":[],"label_agreement":null},{"id":"W2579578061","doi":"10.63317/2sfruoxcckcu","title":"Classifying Out-of-vocabulary Terms in a Domain-Specific Social Media Corpus","year":2016,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of New Brunswick; University of Toronto","funders":"","keywords":"Vocabulary; Computer science; Natural language processing; Task (project management); Artificial intelligence; Domain (mathematical analysis); Word (group theory); Set (abstract data type); Linguistics; Mathematics","score_opus":0.03021188006224321,"score_gpt":0.26984454680507325,"score_spread":0.23963266674283004,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2579578061","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.92329717,0.005160824,0.03197735,0.0008539199,0.00042644783,0.00043575317,0.029787514,0.0011413833,0.006919691],"genre_scores_gemma":[0.8638363,0.0022537059,0.046130933,0.00025961775,0.00031586995,0.0004003185,0.08296466,0.00027646153,0.0035621827],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99874747,0.00027246913,0.00019010511,0.00028317235,0.00041204083,0.00009464736],"domain_scores_gemma":[0.9943116,0.0038749722,0.00036070775,0.00038299532,0.00086956896,0.00020017452],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00082393474,0.0006234801,0.00049951265,0.007276343,0.0012771782,0.0011762466,0.00063586264,0.0010334861,0.0015003477],"category_scores_gemma":[0.0061909272,0.00024595973,0.00064864027,0.004537428,0.000721351,0.001696676,0.0009482535,0.0009873873,0.00094555464],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018725308,0.0017609328,0.119373225,0.005987329,0.00076536206,0.006256979,0.005456656,0.009393355,0.3042964,0.010260109,0.053488176,0.48108903],"study_design_scores_gemma":[0.0003358063,0.00093772187,0.43297866,0.001231203,0.0014150316,0.012790849,0.013529304,0.2113894,0.12846436,0.010409007,0.18622896,0.00028970808],"about_ca_topic_score_codex":0.010080171,"about_ca_topic_score_gemma":0.022517994,"teacher_disagreement_score":0.010080171,"about_ca_system_score_codex":0.00067608897,"about_ca_system_score_gemma":0.0012960414,"threshold_uncertainty_score":0.020042956},"labels":[],"label_agreement":null},{"id":"W2579759882","doi":"10.63317/47iyyzyji28j","title":"SemLinker, a Modular and Open Source Framework for Named Entity Discovery and Linking","year":2016,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University; Polytechnique Montréal; Université du Québec à Montréal","funders":"","keywords":"Computer science; Entity linking; Annotation; Knowledge base; Information retrieval; Modular design; Named-entity recognition; Cluster analysis; Named entity; Open source; Population; World Wide Web; Natural language processing; Artificial intelligence; Programming language; Software; Engineering; Task (project management)","score_opus":0.016301544275535446,"score_gpt":0.29248137003187435,"score_spread":0.2761798257563389,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2579759882","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0011664664,0.00027143123,0.81411916,0.00024971768,0.00021808366,0.0002816235,0.005902968,0.1738432,0.00394729],"genre_scores_gemma":[0.02082983,0.00076026114,0.8998486,0.00053178647,0.0001654358,0.00056479167,0.0385398,0.024496228,0.01426335],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9973151,0.00045659902,0.0002891129,0.0007401191,0.0010570472,0.00014193586],"domain_scores_gemma":[0.99514526,0.0021759092,0.0003571374,0.0014340307,0.0006235999,0.0002638926],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0044062273,0.0018746858,0.0015672484,0.006833217,0.00177702,0.0054168147,0.0044446397,0.0020224391,0.026779968],"category_scores_gemma":[0.01367062,0.0018244566,0.0029197494,0.004575938,0.0011115903,0.008522335,0.007179339,0.0038946006,0.018854009],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00072309095,0.0005426017,0.0031075324,0.0026271655,0.00093425554,0.0010956168,0.0013622161,0.009522098,0.01807088,0.088563904,0.27823147,0.59521925],"study_design_scores_gemma":[0.00022648723,0.00012275674,0.002023786,0.00043045398,0.0003338338,0.0013486021,0.00039774735,0.14184068,0.038339823,0.17524631,0.6393432,0.00034630593],"about_ca_topic_score_codex":0.0038817525,"about_ca_topic_score_gemma":0.008532386,"teacher_disagreement_score":0.026779968,"about_ca_system_score_codex":0.0010072879,"about_ca_system_score_gemma":0.0034085421,"threshold_uncertainty_score":0.08958793},"labels":[],"label_agreement":null},{"id":"W2583305969","doi":"10.1007/s10579-017-9383-x","title":"RST Signalling Corpus: a corpus of signals of coherence relations","year":2017,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":48,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Treebank; Annotation; Corpus linguistics; Natural language processing; Computer science; Parsing; Coherence (philosophical gambling strategy); Artificial intelligence; Text corpus; Linguistics; British National Corpus","score_opus":0.030780822958553684,"score_gpt":0.3193694584320568,"score_spread":0.2885886354735031,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2583305969","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2560739,0.0047461,0.088699825,0.0033437535,0.0013551705,0.0016675736,0.5526342,0.0136403935,0.077839166],"genre_scores_gemma":[0.40793693,0.0016225931,0.0760173,0.00072124874,0.00053736253,0.0024977243,0.47244325,0.0026436914,0.035579912],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99725085,0.0008702228,0.00033101355,0.00046560305,0.0009038898,0.0001785056],"domain_scores_gemma":[0.9850116,0.009570358,0.00072192337,0.0021341804,0.0021164608,0.00044551288],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020753627,0.0011649855,0.00080342835,0.0042862515,0.0015305846,0.0020724798,0.0016552615,0.0025530183,0.030605486],"category_scores_gemma":[0.017156677,0.00065667316,0.00048622757,0.0035712216,0.0013082904,0.0021690559,0.0021013923,0.0017616844,0.014354473],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0039048449,0.0009018078,0.00634534,0.008438109,0.00032413943,0.0026162004,0.00806343,0.005638235,0.14779799,0.046045136,0.4653592,0.3045656],"study_design_scores_gemma":[0.0011084436,0.0005845534,0.0727355,0.00085175893,0.00048194465,0.0042673936,0.0038040902,0.018480184,0.07631247,0.014803617,0.8060129,0.0005571572],"about_ca_topic_score_codex":0.0093255155,"about_ca_topic_score_gemma":0.010638906,"teacher_disagreement_score":0.030605486,"about_ca_system_score_codex":0.0010449528,"about_ca_system_score_gemma":0.0023501462,"threshold_uncertainty_score":0.10238558},"labels":[],"label_agreement":null},{"id":"W2584335472","doi":"10.3389/fpsyg.2017.00096","title":"Searching High and Low: Prosodic Breaks Disambiguate Relative Clauses","year":2017,"lang":"en","type":"article","venue":"Frontiers in Psychology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Centre for Research on Brain Language and Music; Université de Montréal","funders":"Agència de Gestió d'Ajuts Universitaris i de Recerca; European Research Council; Ministerio de Economía y Competitividad; Generalitat de Catalunya; European Commission","keywords":"Psychology; Cognitive psychology; Linguistics; Natural language processing; Computer science","score_opus":0.015509953744191288,"score_gpt":0.3308854901970946,"score_spread":0.31537553645290334,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2584335472","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9848653,0.00035658924,0.007834336,0.00012920005,0.000025360865,0.000025103918,0.000069346075,0.00008034824,0.0066143298],"genre_scores_gemma":[0.9909532,0.00012702351,0.007899851,0.000110153625,0.0000108292525,0.000016750584,0.00012368769,0.00009813003,0.00066029304],"study_design_codex":"bench_or_experimental","study_design_gemma":"observational","domain_scores_codex":[0.99937135,0.00019793306,0.000043705328,0.00019638918,0.000119545206,0.00007115275],"domain_scores_gemma":[0.99768317,0.001342816,0.00044180325,0.00021702448,0.00019527723,0.00011984967],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001104161,0.00048366297,0.0003526332,0.0003969415,0.00042109896,0.0017704683,0.0005088533,0.00060957967,0.004058449],"category_scores_gemma":[0.005794582,0.00038105625,0.0002326551,0.00025402,0.0007481082,0.0015398762,0.0010454311,0.00079729117,0.00062156044],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018213543,0.00013325052,0.019922998,0.00044461034,0.00006585649,0.0008473117,0.0082051,0.0006930474,0.90618265,0.0060519944,0.0004433279,0.0551885],"study_design_scores_gemma":[0.0005990504,0.0027093207,0.5099145,0.00033635998,0.00064531213,0.0023094788,0.024615467,0.018673198,0.3899539,0.032343138,0.017574957,0.00032518487],"about_ca_topic_score_codex":0.0010776771,"about_ca_topic_score_gemma":0.0015909722,"teacher_disagreement_score":0.004058449,"about_ca_system_score_codex":0.000369985,"about_ca_system_score_gemma":0.00031852207,"threshold_uncertainty_score":0.013576865},"labels":[],"label_agreement":null},{"id":"W2585956289","doi":"","title":"On-line Reference Assignment for Anaphoric and Non-Anaphoric Nouns: A Unified, Memory-Based Model in ACT-R","year":2007,"lang":"en","type":"article","venue":"eScholarship (California Digital Library)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Referent; Computer science; Noun; Artificial intelligence; Natural language processing; Noun phrase; Intuition; Proper noun; Linguistics; Psychology; Cognitive science; Philosophy","score_opus":0.02324738625499922,"score_gpt":0.26000305751849273,"score_spread":0.2367556712634935,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2585956289","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.064900964,0.0003390728,0.8988282,0.00232769,0.000099364224,0.0001239511,0.00033135057,0.0009950462,0.032054443],"genre_scores_gemma":[0.78988826,0.000364887,0.18572733,0.00044817087,0.00011374356,0.0004015074,0.00031116183,0.00035222992,0.022392644],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99843925,0.00062978297,0.00008843436,0.00039655497,0.00028872502,0.00015727947],"domain_scores_gemma":[0.99696654,0.0014486338,0.0002858058,0.00077140785,0.00032832925,0.000199171],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025700647,0.0005927012,0.0013068334,0.0016520921,0.0011430548,0.004499723,0.0044082818,0.0023256694,0.010153542],"category_scores_gemma":[0.0067691584,0.0008599463,0.0021405944,0.0010906471,0.0031616194,0.009855666,0.0019949186,0.0018035192,0.0032844734],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021427193,0.00013457244,0.0017360786,0.00011730779,0.000060977403,0.00034048644,0.0012293757,0.101109885,0.0023076122,0.85770315,0.0020768314,0.032969467],"study_design_scores_gemma":[0.00004889446,0.00006838401,0.0003725401,0.000021476551,0.000042594882,0.00017651317,0.00013929824,0.7320914,0.00094833074,0.26420605,0.0018423757,0.000042166703],"about_ca_topic_score_codex":0.007447527,"about_ca_topic_score_gemma":0.004781111,"teacher_disagreement_score":0.010153542,"about_ca_system_score_codex":0.0024495148,"about_ca_system_score_gemma":0.001734853,"threshold_uncertainty_score":0.03396696},"labels":[],"label_agreement":null},{"id":"W2586740849","doi":"10.29173/cais214","title":"Application of Predictive and Descriptive Text Mining Techniques for Analysis and Organization of Unstructured Documents","year":2013,"lang":"fr","type":"article","venue":"Proceedings of the Annual Conference of CAIS / Actes du congrès annuel de l ACSI","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Presentation (obstetrics); Humanities; Descriptive statistics; Computer science; Artificial intelligence; Philosophy; Mathematics; Statistics","score_opus":0.012707501993383725,"score_gpt":0.24961921513196994,"score_spread":0.23691171313858622,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2586740849","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.032201108,0.0005777317,0.95942193,0.00066301826,0.00006127821,0.00068927434,0.0017847047,0.0021230122,0.0024780002],"genre_scores_gemma":[0.12469302,0.0004992102,0.87053984,0.000090034846,0.00006897972,0.00048550894,0.002269134,0.000117160685,0.0012371182],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9967565,0.0010667044,0.00037110352,0.00060969987,0.001099919,0.00009607621],"domain_scores_gemma":[0.97705597,0.0153361475,0.0019466119,0.002208606,0.003137955,0.0003147346],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0046526073,0.0008061531,0.0006368758,0.0075106095,0.0009601877,0.0025839661,0.0014914534,0.0005866888,0.001599938],"category_scores_gemma":[0.018682178,0.0004035945,0.0009150678,0.0067184377,0.00095465215,0.0028822292,0.00096070976,0.0011552107,0.00095468934],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004067571,0.00041416936,0.015451044,0.0017512008,0.00017054175,0.0007310994,0.006073818,0.012753922,0.036034126,0.026430469,0.005555913,0.89422685],"study_design_scores_gemma":[0.00020705792,0.0010260502,0.036981076,0.0013245458,0.00072319683,0.00510834,0.012537473,0.5565146,0.12491542,0.15933903,0.10098256,0.00034059415],"about_ca_topic_score_codex":0.001690602,"about_ca_topic_score_gemma":0.0028391827,"teacher_disagreement_score":0.0075106095,"about_ca_system_score_codex":0.0007506397,"about_ca_system_score_gemma":0.0018557116,"threshold_uncertainty_score":0.024605632},"labels":[],"label_agreement":null},{"id":"W2587591048","doi":"10.5539/mas.v11n3p105","title":"Machine Learning with Pattern Recognition in Natural Language Processing without any Grammar and Syntax Knowledge","year":2017,"lang":"en","type":"article","venue":"Modern Applied Science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Natural language processing; Parsing; Grammar; Artificial intelligence; Syntax; Interpreter; Natural language; Parse tree; Tree (set theory); Rule-based machine translation; Natural (archaeology); Question answering; Linguistics; Programming language","score_opus":0.013302127137293606,"score_gpt":0.2714186216736702,"score_spread":0.2581164945363766,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2587591048","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006655127,0.00046597898,0.9768961,0.0003590925,0.00016634112,0.00023314163,0.0002845893,0.009083793,0.005855822],"genre_scores_gemma":[0.10101196,0.00059137325,0.88484895,0.00066120824,0.00015162505,0.00066772115,0.0013987423,0.0003040756,0.010364355],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99874014,0.00037491717,0.00011550173,0.0003865921,0.0003095686,0.00007336985],"domain_scores_gemma":[0.9989698,0.00046355784,0.00006422377,0.0002793923,0.00018766172,0.00003542406],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014586763,0.00067826064,0.0007757321,0.0010747551,0.0003823136,0.0016209987,0.001370901,0.0010967081,0.0070169833],"category_scores_gemma":[0.0036253955,0.0003942998,0.0007736657,0.0011569897,0.000770055,0.0025773617,0.0011077161,0.001121676,0.005806921],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029128775,0.00031690413,0.0024021228,0.00074724003,0.00018620862,0.0003666181,0.00023566019,0.010089511,0.026283585,0.029004944,0.017782608,0.9122933],"study_design_scores_gemma":[0.00020914589,0.0006943493,0.0044392035,0.00032318415,0.0003077506,0.001356005,0.00017942478,0.61249274,0.0842109,0.17275675,0.122902825,0.00012773844],"about_ca_topic_score_codex":0.0009898872,"about_ca_topic_score_gemma":0.0008419573,"teacher_disagreement_score":0.0070169833,"about_ca_system_score_codex":0.00040745366,"about_ca_system_score_gemma":0.000797493,"threshold_uncertainty_score":0.023474157},"labels":[],"label_agreement":null},{"id":"W2587623425","doi":"10.5539/mas.v11n4p55","title":"An Investigation into Methodology and Metrics Employed to Evaluate the (Speech-to-Speech) Way in Translation Systems","year":2017,"lang":"en","type":"article","venue":"Modern Applied Science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Speech translation; Computer science; Speech recognition; Machine translation; Natural language processing; Translation (biology); Artificial intelligence; Sentence; Speech synthesis; Example-based machine translation; Speech processing; Speech corpus; Evaluation of machine translation; Machine translation software usability","score_opus":0.10180193496742422,"score_gpt":0.37424261216731114,"score_spread":0.2724406771998869,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2587623425","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19564603,0.009346415,0.7816976,0.00072431733,0.00036898986,0.0010348696,0.0016729119,0.0024746445,0.007034257],"genre_scores_gemma":[0.50173813,0.0014178205,0.49281082,0.00014422058,0.000101816564,0.0007176166,0.0017056816,0.00035186557,0.0010120141],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.94179547,0.033024393,0.008438963,0.003924422,0.012195225,0.0006216166],"domain_scores_gemma":[0.8723701,0.07767839,0.0136115495,0.009707561,0.025606217,0.0010262238],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.040697552,0.0016619039,0.0014904233,0.008279276,0.0012104288,0.004685668,0.0015887469,0.0025722072,0.0010154558],"category_scores_gemma":[0.1253213,0.00046297876,0.001085254,0.006718797,0.0020933412,0.005212277,0.002533408,0.0015599026,0.0006411145],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011394311,0.000643157,0.092805445,0.0039800694,0.0016172983,0.00029604582,0.0047118748,0.0965036,0.046401765,0.03230673,0.0031826561,0.71641195],"study_design_scores_gemma":[0.00020535341,0.010926226,0.11881884,0.0012949397,0.0008528839,0.0026615756,0.0055017117,0.63167924,0.13850029,0.052652344,0.036037646,0.00086900743],"about_ca_topic_score_codex":0.0024614655,"about_ca_topic_score_gemma":0.002434456,"teacher_disagreement_score":0.9593024,"about_ca_system_score_codex":0.0023412013,"about_ca_system_score_gemma":0.0017226542,"threshold_uncertainty_score":0.21523184},"labels":[],"label_agreement":null},{"id":"W2587910048","doi":"10.29173/cais729","title":"Unstructured English Queries: How Users Request Information","year":2013,"lang":"en","type":"article","venue":"Proceedings of the Annual Conference of CAIS / Actes du congrès annuel de l ACSI","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Vocabulary; Syntax; Documentation; Task (project management); Grammar; Query language; Natural language; Information retrieval; Natural language processing; World Wide Web; Artificial intelligence; Linguistics; Programming language","score_opus":0.00970440161819043,"score_gpt":0.21629805202785946,"score_spread":0.20659365040966904,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2587910048","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4214248,0.0024289931,0.44136003,0.00811799,0.00024103514,0.002574025,0.006577756,0.023155145,0.094120264],"genre_scores_gemma":[0.78019476,0.001255471,0.1606419,0.002712216,0.00011296546,0.000682079,0.009135677,0.0032737139,0.04199117],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99615747,0.0019447767,0.00031751848,0.00039996865,0.00090360513,0.00027668948],"domain_scores_gemma":[0.9823631,0.012809838,0.0005930704,0.0013600618,0.0022968375,0.0005770866],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036388002,0.0007576784,0.0007586241,0.001425129,0.0012120492,0.0043862225,0.0015274152,0.0014548994,0.010339257],"category_scores_gemma":[0.025985403,0.0005665508,0.00030177215,0.0016094552,0.001428671,0.0071678157,0.0030720339,0.001021743,0.004615288],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0032349615,0.0007818128,0.035860214,0.0030398404,0.00017179226,0.004786522,0.18061844,0.004174381,0.07680811,0.07085737,0.21191502,0.4077515],"study_design_scores_gemma":[0.00036162618,0.00081991067,0.014802566,0.00075678853,0.0002459744,0.00252869,0.078343794,0.07795796,0.042871103,0.057938624,0.7229325,0.0004404553],"about_ca_topic_score_codex":0.007818004,"about_ca_topic_score_gemma":0.0115165,"teacher_disagreement_score":0.010339257,"about_ca_system_score_codex":0.0010148808,"about_ca_system_score_gemma":0.0012319044,"threshold_uncertainty_score":0.034588277},"labels":[],"label_agreement":null},{"id":"W2587922133","doi":"10.1111/lang.12224","title":"Corpus Use in Language Learning: A Meta‐Analysis","year":2017,"lang":"en","type":"article","venue":"Language Learning","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":551,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Moderation; Meta-analysis; Open data; Computer science; Transfer of learning; Psychology; Sample (material); Language acquisition; Natural language processing; Open science; Applied linguistics; Artificial intelligence; Mathematics education; Linguistics; World Wide Web; Statistics; Machine learning; Mathematics","score_opus":0.03466804115790036,"score_gpt":0.31612705750871456,"score_spread":0.2814590163508142,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2587922133","genre_codex":"review","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01869604,0.96293896,0.010910132,0.0009752963,0.00067452743,0.0027732411,0.0017502378,0.0002232763,0.001058375],"genre_scores_gemma":[0.64368856,0.31366983,0.023118638,0.00199764,0.00040752644,0.014036068,0.0019342116,0.00028654098,0.00086094916],"study_design_codex":"meta_analysis","study_design_gemma":"meta_analysis","domain_scores_codex":[0.893125,0.07224134,0.019346474,0.0068525653,0.007499039,0.0009356124],"domain_scores_gemma":[0.77423793,0.1952324,0.01408155,0.0075732423,0.007912148,0.0009627622],"candidate_categories":["metaepi_broad"],"consensus_categories":[],"category_scores_codex":[0.08739081,0.0027729704,0.014671697,0.012686026,0.0011531855,0.006096204,0.003107225,0.0023671472,0.0042282576],"category_scores_gemma":[0.19709323,0.0017557767,0.041878644,0.010363324,0.0016399745,0.0048016,0.003171998,0.0031397538,0.00035345543],"study_design_candidate":"meta_analysis","study_design_consensus":"meta_analysis","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001321905,0.000038992363,0.0048251813,0.17700815,0.80373776,0.00010962118,0.00036148072,0.00044508665,0.00021370607,0.0004277757,0.00058352994,0.010926803],"study_design_scores_gemma":[0.000618671,0.00035637588,0.003641828,0.033727176,0.95701426,0.0000953704,0.00020149999,0.0003587135,0.00038099516,0.00093727617,0.0026295986,0.000038260216],"about_ca_topic_score_codex":0.004375144,"about_ca_topic_score_gemma":0.008155971,"teacher_disagreement_score":0.9853283,"about_ca_system_score_codex":0.0043620695,"about_ca_system_score_gemma":0.005245195,"threshold_uncertainty_score":0.46217233},"labels":[],"label_agreement":null},{"id":"W2588342741","doi":"10.29173/cais573","title":"Comparing Controlled Vocabularies and Tags: Research Methodologies and Research Goals","year":2013,"lang":"fr","type":"article","venue":"Proceedings of the Annual Conference of CAIS / Actes du congrès annuel de l ACSI","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Context (archaeology); Vocabulary; Humanities; Computer science; Philosophy; Linguistics; History","score_opus":0.17036653958860926,"score_gpt":0.386395233123991,"score_spread":0.21602869353538173,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2588342741","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3077363,0.18264571,0.43941772,0.0068141627,0.0018173336,0.009440167,0.009367678,0.0010707672,0.04169016],"genre_scores_gemma":[0.65316856,0.0357514,0.2827201,0.0019188079,0.0010568692,0.012675358,0.008424666,0.0006924187,0.0035918457],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.81884223,0.11463712,0.023519248,0.016810982,0.02448038,0.0017100387],"domain_scores_gemma":[0.29418886,0.5999047,0.047493756,0.022600304,0.034160826,0.0016515497],"candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.11756078,0.0012632549,0.0031187155,0.043116607,0.003056381,0.019913245,0.004304709,0.0021726256,0.0039809216],"category_scores_gemma":[0.4538583,0.0013290808,0.0032449316,0.05762117,0.0071780714,0.023768673,0.006096577,0.0025112687,0.0011742796],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0024094637,0.00052647584,0.14048828,0.039937973,0.0056117806,0.00022725629,0.035312116,0.0029865485,0.009708881,0.10724673,0.0038950932,0.6516495],"study_design_scores_gemma":[0.0017135147,0.00619768,0.2629376,0.057835996,0.011647836,0.002584318,0.122918755,0.021011807,0.046299726,0.22504878,0.23978975,0.0020143306],"about_ca_topic_score_codex":0.016302234,"about_ca_topic_score_gemma":0.010166831,"teacher_disagreement_score":0.98008674,"about_ca_system_score_codex":0.008227452,"about_ca_system_score_gemma":0.010221368,"threshold_uncertainty_score":0.6217283},"labels":[],"label_agreement":null},{"id":"W2589278453","doi":"10.63317/2ytq4eceh48c","title":"Applying Lexical Constraints on Morpho-Syntactic Patterns for the Identification of Conceptual-Relational Content in Specialized Texts","year":2006,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Computer science; Constraint (computer-aided design); Natural language processing; Artificial intelligence; Representation (politics); Morpho; Identification (biology); Mathematics","score_opus":0.053715103528791004,"score_gpt":0.29584990621064866,"score_spread":0.24213480268185766,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2589278453","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.055295117,0.00024732322,0.9381113,0.00057316275,0.000041151918,0.0003326995,0.0009660074,0.00088719063,0.0035460799],"genre_scores_gemma":[0.29394904,0.00027330784,0.702206,0.00015691934,0.00004767703,0.00052269746,0.001701461,0.0003501082,0.0007927219],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9951971,0.0021054412,0.0005463085,0.0009917567,0.0009779712,0.00018147899],"domain_scores_gemma":[0.9822672,0.012103103,0.0016133946,0.0028243812,0.0009408286,0.0002511634],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004818045,0.000570677,0.00060910056,0.0029305147,0.0013138871,0.004199128,0.001976517,0.0010529386,0.0034548086],"category_scores_gemma":[0.026983242,0.00065049186,0.00077595236,0.0037900365,0.003538056,0.0070179533,0.0027027798,0.0018908799,0.00057545],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002772303,0.00015296407,0.009681148,0.0009891387,0.00012446131,0.00057033205,0.0040492425,0.024436843,0.04102062,0.4949639,0.005542048,0.41819203],"study_design_scores_gemma":[0.00012990831,0.00013424837,0.008287459,0.0003985211,0.00015145715,0.00071522116,0.0028352404,0.36865497,0.06544682,0.47913504,0.0739026,0.00020861956],"about_ca_topic_score_codex":0.0038796654,"about_ca_topic_score_gemma":0.005660909,"teacher_disagreement_score":0.004818045,"about_ca_system_score_codex":0.0012830024,"about_ca_system_score_gemma":0.002464269,"threshold_uncertainty_score":0.025480628},"labels":[],"label_agreement":null},{"id":"W2590414451","doi":"10.29173/cais350","title":"An Epistemological Analysis of Holocaust Survivor Transcripts","year":2013,"lang":"fr","type":"article","venue":"Proceedings of the Annual Conference of CAIS / Actes du congrès annuel de l ACSI","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Indexation; Search engine indexing; Facet (psychology); Identification (biology); Humanities; Computer science; Coding (social sciences); Information retrieval; Sociology; Philosophy; Psychology; Social science; Economics","score_opus":0.034511252485365504,"score_gpt":0.27634364033213604,"score_spread":0.24183238784677052,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2590414451","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.57623875,0.00074430107,0.27924454,0.004966739,0.00031094134,0.001328951,0.008505339,0.0010243397,0.12763606],"genre_scores_gemma":[0.87570703,0.0005905246,0.092512734,0.00027417473,0.00007708161,0.0010123078,0.0038011556,0.0004727735,0.02555219],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9946015,0.0027782898,0.00030596607,0.0006108362,0.0013089463,0.00039439905],"domain_scores_gemma":[0.97147155,0.02043796,0.0017009147,0.0021149074,0.0039030777,0.000371659],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0061616935,0.00030027176,0.00024284564,0.0074316184,0.0033867916,0.00456383,0.0007884695,0.00085362507,0.00635374],"category_scores_gemma":[0.03078369,0.00022970638,0.0003351277,0.0056053433,0.003957243,0.0031858026,0.003427491,0.0012718405,0.0013347097],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029239795,0.000049039012,0.011231775,0.0006180886,0.000018309936,0.0018790436,0.6899892,0.0005719755,0.015536263,0.1266559,0.006382649,0.14677536],"study_design_scores_gemma":[0.000017806786,0.00009244055,0.03786886,0.00095499575,0.000035095003,0.0017553746,0.63214535,0.0068218997,0.013679834,0.064512916,0.24200024,0.00011526419],"about_ca_topic_score_codex":0.006871131,"about_ca_topic_score_gemma":0.009961105,"teacher_disagreement_score":0.0074316184,"about_ca_system_score_codex":0.0047170795,"about_ca_system_score_gemma":0.0028450435,"threshold_uncertainty_score":0.034224987},"labels":[],"label_agreement":null},{"id":"W2591101772","doi":"10.29173/cais286","title":"Designing a Language-Independent Search Prototype for Accessing Multilingual Resources from Metadata-Enabled Repositories Information Need","year":2013,"lang":"fr","type":"article","venue":"Proceedings of the Annual Conference of CAIS / Actes du congrès annuel de l ACSI","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Metadata; Humanities; World Wide Web; The Internet; Computer science; Library science; Art","score_opus":0.028745422249635402,"score_gpt":0.2833949247244765,"score_spread":0.25464950247484114,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2591101772","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.062484823,0.00038305204,0.8517218,0.0008746872,0.00012966612,0.0016536119,0.0011976799,0.071110696,0.01044397],"genre_scores_gemma":[0.15037118,0.00023029589,0.8308154,0.00038361992,0.000027780827,0.0009116613,0.0024364607,0.0032980612,0.011525568],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99796546,0.00042729484,0.00027187206,0.00050464016,0.00067255326,0.00015829886],"domain_scores_gemma":[0.99619246,0.0017028479,0.00014930165,0.00056176516,0.0011751236,0.00021853043],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032877536,0.0007784972,0.0013892394,0.0015888173,0.0010362224,0.0038683747,0.0040278346,0.0021817912,0.010232692],"category_scores_gemma":[0.010559532,0.0010891866,0.0012049645,0.0011964872,0.0009661558,0.007995578,0.002150935,0.001531262,0.0061557563],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0026197454,0.0013583512,0.0067787566,0.005166785,0.0003220396,0.0040011355,0.018267876,0.011964936,0.3942962,0.04251768,0.051020328,0.46168622],"study_design_scores_gemma":[0.00078847294,0.0018584235,0.0039794664,0.0005616094,0.00049723097,0.004119979,0.0077270023,0.34225592,0.3657651,0.013123374,0.25876412,0.000559287],"about_ca_topic_score_codex":0.0058736913,"about_ca_topic_score_gemma":0.0052583553,"teacher_disagreement_score":0.010232692,"about_ca_system_score_codex":0.001238057,"about_ca_system_score_gemma":0.0016489569,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W2591613536","doi":"","title":"New perspectives on the prefix array","year":2009,"lang":"en","type":"article","venue":"UWA Profiles and Research Repository (University of Western Australia)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"","keywords":"Prefix; Combinatorics; String (physics); Integer (computer science); Alphabet; Byte; Mathematics; Value (mathematics); Discrete mathematics; Matching (statistics); Space (punctuation); Computer science","score_opus":0.05693653576477857,"score_gpt":0.32103577941901396,"score_spread":0.26409924365423537,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2591613536","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010315302,0.012884367,0.7862721,0.042509638,0.003329709,0.00005860645,0.0006207813,0.00088428357,0.14312525],"genre_scores_gemma":[0.27177653,0.01663259,0.63963574,0.008739886,0.01078201,0.00042130967,0.0012166633,0.0019465804,0.048848692],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9931685,0.003491749,0.0004666249,0.0013286911,0.0011563128,0.00038809163],"domain_scores_gemma":[0.9824362,0.010116113,0.000583214,0.0043346384,0.0017594867,0.00077024085],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007639769,0.0012211878,0.002153245,0.0038753024,0.0052779554,0.017125797,0.004410978,0.0052383123,0.025615985],"category_scores_gemma":[0.022324663,0.0014623578,0.0020625915,0.0071326443,0.019586185,0.06905443,0.0075879595,0.010351857,0.0058309925],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000006955239,0.000003887867,0.00004180796,0.000016724724,0.000001805865,0.000018222841,0.00016337847,0.00014354083,0.000052765856,0.99415153,0.0014948657,0.0039044635],"study_design_scores_gemma":[0.000003808362,0.000004538918,0.000020927302,0.000022656139,0.0000034838413,0.000055654662,0.00014481362,0.0015378878,0.00006988692,0.9737268,0.024400476,0.000009073407],"about_ca_topic_score_codex":0.002913987,"about_ca_topic_score_gemma":0.002782413,"teacher_disagreement_score":0.025615985,"about_ca_system_score_codex":0.0031130235,"about_ca_system_score_gemma":0.002254363,"threshold_uncertainty_score":0.085694015},"labels":[],"label_agreement":null},{"id":"W2592760122","doi":"10.7202/1038912ar","title":"Module NooJ du français. Traitement automatique de corpus de français parlé régional","year":2017,"lang":"fr","type":"article","venue":"Revue de l’Université de Moncton","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":true,"ca_institutions":"Université de Moncton","funders":"","keywords":"Art; Humanities","score_opus":0.012009264230770032,"score_gpt":0.22539578051494277,"score_spread":0.21338651628417274,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2592760122","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.25567573,0.0024136354,0.5250761,0.0020306665,0.00079237885,0.0014283475,0.06334189,0.06234136,0.08689987],"genre_scores_gemma":[0.42780665,0.0010971737,0.39211813,0.0006338486,0.00015287043,0.0009289537,0.0831068,0.011899577,0.082255915],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99877244,0.00020342258,0.00013494611,0.0005550987,0.00021783217,0.00011620751],"domain_scores_gemma":[0.99773943,0.0005741905,0.0001452659,0.00041984962,0.0010420048,0.00007925744],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010385603,0.0012601262,0.00043744428,0.001939195,0.00091470865,0.002266907,0.00070012873,0.0006391278,0.022660766],"category_scores_gemma":[0.0039192997,0.0006280381,0.00075584376,0.0017667138,0.0006900846,0.0023238778,0.0016128116,0.00076843245,0.009199195],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00078394415,0.00009497024,0.016357323,0.0021176161,0.00027512977,0.0013317209,0.009941043,0.003521948,0.124173544,0.013689342,0.09193678,0.73577666],"study_design_scores_gemma":[0.000084117906,0.00022921815,0.055970345,0.00038124394,0.00028766907,0.0015105368,0.0055725356,0.026589328,0.1116443,0.0061418703,0.7913754,0.0002132961],"about_ca_topic_score_codex":0.07137148,"about_ca_topic_score_gemma":0.08177005,"teacher_disagreement_score":0.07137148,"about_ca_system_score_codex":0.0015098181,"about_ca_system_score_gemma":0.0024584786,"threshold_uncertainty_score":0.14191216},"labels":[],"label_agreement":null},{"id":"W2593048361","doi":"","title":"Total vs. Partial Adjectives : Evidence from Reduplication","year":2011,"lang":"en","type":"article","venue":"The LACUS forum","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"","keywords":"Reduplication; Linguistics; History; Mathematics; Philosophy","score_opus":0.031422403893197406,"score_gpt":0.27082360851507215,"score_spread":0.23940120462187475,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2593048361","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.74970156,0.0019596226,0.12483282,0.0022624594,0.0005908099,0.00017766794,0.001437831,0.0015744207,0.11746272],"genre_scores_gemma":[0.99071103,0.00018641526,0.0061977417,0.00015898478,0.00007105779,0.000020507468,0.0003227886,0.00048087514,0.0018505102],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.98783,0.004695233,0.0016182596,0.0022308948,0.0028889163,0.00073670625],"domain_scores_gemma":[0.9333936,0.036576204,0.004813414,0.015472716,0.009147876,0.0005962525],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007409271,0.0006948956,0.0010577736,0.0022234144,0.0033276551,0.004684881,0.0015723124,0.0015285492,0.00551321],"category_scores_gemma":[0.033230714,0.0012738609,0.0009516389,0.0028450275,0.007635259,0.008814791,0.0054547274,0.0040481696,0.0012600338],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002238968,0.0001425763,0.03084429,0.0025530194,0.00037159078,0.0045807115,0.050470743,0.0022387863,0.06712839,0.74298215,0.007286174,0.089162566],"study_design_scores_gemma":[0.0003969398,0.0005951407,0.06686459,0.0007569681,0.00092047534,0.015648717,0.024984721,0.02001472,0.1064806,0.57983917,0.18275012,0.0007478767],"about_ca_topic_score_codex":0.0019587434,"about_ca_topic_score_gemma":0.0019795916,"teacher_disagreement_score":0.007409271,"about_ca_system_score_codex":0.0015407064,"about_ca_system_score_gemma":0.0011462299,"threshold_uncertainty_score":0.03918445},"labels":[],"label_agreement":null},{"id":"W2593930809","doi":"10.5539/elt.v10n4p62","title":"Concept Map Technique as a New Method for Whole Text Translation","year":2017,"lang":"en","type":"article","venue":"English Language Teaching","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Vocabulary; Paragraph; Significant difference; Mathematics education; Psychology; Equivalence (formal languages); Test (biology); Arabic; Natural language processing; Teaching method; Computer science; Artificial intelligence; Linguistics; Mathematics; Statistics","score_opus":0.01628666816202498,"score_gpt":0.3455506392016345,"score_spread":0.32926397103960947,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2593930809","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015169295,0.0011728653,0.97047573,0.0002589811,0.0004135226,0.00047536462,0.00014893294,0.0019882913,0.009897038],"genre_scores_gemma":[0.073214404,0.000968036,0.9182676,0.00009981492,0.00015329618,0.0008258782,0.0002432279,0.0002943684,0.005933317],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9973417,0.0013155589,0.00016580774,0.00032605373,0.00078257115,0.000068256384],"domain_scores_gemma":[0.99732983,0.0017970981,0.00013721526,0.00031056136,0.0003644627,0.00006076412],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018141615,0.00076583464,0.00051306345,0.0022065728,0.00042422005,0.0012941259,0.000829492,0.0004937798,0.008880974],"category_scores_gemma":[0.0051097283,0.00027024254,0.00068053266,0.0018829519,0.0008140781,0.0019597963,0.0011335826,0.00094741903,0.0027326464],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024466802,0.00023635647,0.00060661515,0.0011012219,0.00007946208,0.00042297304,0.002731078,0.0013867628,0.030333856,0.01594172,0.005783369,0.9411319],"study_design_scores_gemma":[0.00061791355,0.0035945557,0.0113034835,0.0010970965,0.00045105242,0.0112875225,0.0043503293,0.076745704,0.14707641,0.13412498,0.6089453,0.00040569584],"about_ca_topic_score_codex":0.00025948585,"about_ca_topic_score_gemma":0.0002925101,"teacher_disagreement_score":0.008880974,"about_ca_system_score_codex":0.00027773425,"about_ca_system_score_gemma":0.000745007,"threshold_uncertainty_score":0.029709816},"labels":[],"label_agreement":null},{"id":"W2595199775","doi":"","title":"Supporting Discourse Phenomena in an Interoperable NLP Framework","year":2013,"lang":"en","type":"article","venue":"Research Explorer (The University of Manchester)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Open Text (Canada)","funders":"","keywords":"Computer science; Natural language processing; Interoperability; Artificial intelligence; Linguistics; World Wide Web; Philosophy","score_opus":0.05649842664267326,"score_gpt":0.35187843554203746,"score_spread":0.2953800088993642,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2595199775","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004347301,0.00006029318,0.9865096,0.0007670889,0.000043804153,0.00010107208,0.00028938547,0.004302425,0.003578952],"genre_scores_gemma":[0.1610565,0.0001878767,0.83166194,0.0003004401,0.00009917795,0.00028779547,0.0017816315,0.0011920067,0.00343249],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99085766,0.0040804558,0.0011304222,0.0015221951,0.0019137046,0.0004955641],"domain_scores_gemma":[0.98763996,0.007772601,0.00046272378,0.0028691853,0.00083249225,0.00042297546],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011181715,0.0008359386,0.001288169,0.0033132383,0.0024787523,0.010586654,0.0038878988,0.0032947422,0.008817267],"category_scores_gemma":[0.018505532,0.0012024569,0.0026278733,0.0024871517,0.00378683,0.019234885,0.012583687,0.004713213,0.0024547868],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002588647,0.00028927217,0.0009658309,0.00052967883,0.00011078332,0.0010909872,0.004562275,0.024259236,0.011171596,0.8338436,0.0049504824,0.11796742],"study_design_scores_gemma":[0.000107364445,0.00007016953,0.00024390832,0.00019787195,0.00018624868,0.00037104107,0.0012965943,0.30403796,0.019573981,0.59064287,0.08318246,0.00008943949],"about_ca_topic_score_codex":0.0027713277,"about_ca_topic_score_gemma":0.0035381215,"teacher_disagreement_score":0.011181715,"about_ca_system_score_codex":0.0014078994,"about_ca_system_score_gemma":0.0026207592,"threshold_uncertainty_score":0.05913526},"labels":[],"label_agreement":null},{"id":"W2595284343","doi":"10.1017/jpr.2018.25","title":"Effects of limiting memory capacity on the behaviour of exemplar dynamics","year":2018,"lang":"en","type":"article","venue":"Journal of Applied Probability","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Limiting; Dynamics (music); Extinction (optical mineralogy); Class (philosophy); Mathematics; Computer science; Theoretical computer science; Cognitive psychology; Cognitive science; Artificial intelligence; Psychology","score_opus":0.013775486734763568,"score_gpt":0.23808727343321925,"score_spread":0.2243117866984557,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2595284343","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98025113,0.00024748617,0.014081368,0.0007919773,0.000028757086,0.00002121179,0.00010110695,0.00014905632,0.004327946],"genre_scores_gemma":[0.99863064,0.00006630089,0.00086634525,0.0000472893,0.000006710286,0.000013423714,0.000022566885,0.000016624834,0.00033008165],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991301,0.00040720723,0.00005167781,0.00010577228,0.000107079824,0.00019815267],"domain_scores_gemma":[0.97864234,0.015492913,0.0018146152,0.0018744969,0.00072965457,0.0014460188],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020326984,0.00045208135,0.0010088389,0.0008539757,0.0008203783,0.0014744546,0.0014638982,0.0015133222,0.0044362284],"category_scores_gemma":[0.0348893,0.00043848428,0.0006250345,0.00039359304,0.0024250492,0.0031121636,0.0020104616,0.0014973903,0.00025411666],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008897657,0.00045380948,0.016708842,0.00035724507,0.00029529072,0.001216901,0.0008695975,0.85173076,0.03802592,0.07456796,0.0020432149,0.012840681],"study_design_scores_gemma":[0.00006962593,0.00029082166,0.0046150624,0.000045754652,0.00009687776,0.0003442793,0.00031877504,0.94164747,0.005581106,0.04632307,0.0006003004,0.00006678833],"about_ca_topic_score_codex":0.0040930137,"about_ca_topic_score_gemma":0.0019050274,"teacher_disagreement_score":0.0044362284,"about_ca_system_score_codex":0.0012589422,"about_ca_system_score_gemma":0.00053802715,"threshold_uncertainty_score":0.0148406625},"labels":[],"label_agreement":null},{"id":"W2596484740","doi":"10.18653/v1/e17-1006","title":"Learning Compositionality Functions on Word Embeddings for Modelling Attribute Meaning in Adjective-Noun Phrases","year":2017,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Universität Bielefeld; Bundesministerium für Bildung und Forschung; Atomic Energy of Canada Limited; Deutsche Forschungsgemeinschaft","keywords":"Principle of compositionality; Adjective; Computer science; Natural language processing; Artificial intelligence; Noun; Noun phrase; Specifier; Meaning (existential); Word (group theory); Linguistics; Psychology; Philosophy","score_opus":0.04482606863531863,"score_gpt":0.32168216198947475,"score_spread":0.2768560933541561,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2596484740","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10438017,0.0013722541,0.888638,0.00051321497,0.00015240638,0.0001011349,0.00078066194,0.0020475485,0.002014626],"genre_scores_gemma":[0.6362031,0.0010661754,0.35347202,0.000259614,0.000169827,0.00032446242,0.004125502,0.0006730581,0.0037062108],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992581,0.0003280987,0.000059159822,0.00021630975,0.000078053585,0.000060252987],"domain_scores_gemma":[0.99738973,0.0017826649,0.00014723069,0.00028988498,0.00029281966,0.00009766106],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015091549,0.0013841335,0.0010270121,0.0014662148,0.00056839304,0.0017430759,0.0013073413,0.0013713137,0.002803933],"category_scores_gemma":[0.0076266355,0.0007748787,0.001522012,0.0014972435,0.0008485649,0.006996593,0.0024484268,0.0024524832,0.0019327013],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010156293,0.0005748192,0.010138738,0.00091261044,0.00046034018,0.00046983434,0.0017397588,0.17254978,0.01873109,0.077028364,0.011999362,0.70437974],"study_design_scores_gemma":[0.000034889847,0.000083139435,0.00064038514,0.00005854635,0.00005303911,0.000059359983,0.00020438933,0.8886635,0.00228199,0.10610939,0.0017891068,0.000022261267],"about_ca_topic_score_codex":0.002844236,"about_ca_topic_score_gemma":0.006604979,"teacher_disagreement_score":0.002844236,"about_ca_system_score_codex":0.0008082873,"about_ca_system_score_gemma":0.00083673873,"threshold_uncertainty_score":0.009380043},"labels":[],"label_agreement":null},{"id":"W2597253141","doi":"10.3115/v1/w14-24","title":"Proceedings of the ACL 2014 Workshop on Semantic Parsing","year":2014,"lang":"en","type":"paratext","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"National Institute of Informatics; European Commission; Nuance Foundation; Canadian Institute for Advanced Research; Office of Naval Research; Microsoft Research; Engineering and Physical Sciences Research Council; Defense Advanced Research Projects Agency; Xerox Foundation; Air Force Research Laboratory; National Science Foundation","keywords":"Computer science; Parsing; Artificial intelligence; Information retrieval; Natural language processing; World Wide Web","score_opus":0.01344538560664449,"score_gpt":0.2727337051459326,"score_spread":0.2592883195392881,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2597253141","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0058720936,0.021763256,0.61983216,0.06332214,0.022295266,0.0010115113,0.018725038,0.028163003,0.21901543],"genre_scores_gemma":[0.059201468,0.028276358,0.4986015,0.01465317,0.008542932,0.0022269613,0.14648774,0.015539996,0.22646986],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9912992,0.004580786,0.0008372374,0.0014227071,0.0013797142,0.00048033078],"domain_scores_gemma":[0.9866917,0.006991325,0.00023848635,0.0028710694,0.0024900143,0.0007174327],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011616612,0.0028380044,0.0028431276,0.004358341,0.0031053883,0.012896212,0.0055429526,0.0050278674,0.1259111],"category_scores_gemma":[0.025793329,0.0018524941,0.0029032144,0.0053244894,0.0036966058,0.024384297,0.00840425,0.0077810218,0.05924248],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001744501,0.00019117817,0.00030198123,0.000554044,0.000105440056,0.00023886586,0.0005708852,0.0018574758,0.00080446963,0.064922795,0.78592306,0.14435543],"study_design_scores_gemma":[0.000077754026,0.000027137383,0.00033965387,0.00047677246,0.000071653114,0.00026995994,0.00043341747,0.0094022155,0.0010124218,0.11873209,0.869104,0.000052892072],"about_ca_topic_score_codex":0.012849338,"about_ca_topic_score_gemma":0.014208876,"teacher_disagreement_score":0.1259111,"about_ca_system_score_codex":0.0044984333,"about_ca_system_score_gemma":0.0064268177,"threshold_uncertainty_score":0.42121458},"labels":[],"label_agreement":null},{"id":"W2597891111","doi":"10.1016/j.csl.2017.01.014","title":"On integrating a language model into neural machine translation","year":2017,"lang":"en","type":"article","venue":"Computer Speech & Language","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":113,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Institute for Advanced Research; Université de Montréal","funders":"","keywords":"Machine translation; Computer science; Artificial intelligence; Phrase; Natural language processing; Translation (biology); BLEU; Language model; Baseline (sea); Example-based machine translation; Turkish; Machine learning; Linguistics","score_opus":0.016226769020705842,"score_gpt":0.3056925943937503,"score_spread":0.28946582537304444,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2597891111","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020721592,0.0010915786,0.96647817,0.0013703201,0.00043977302,0.00008953134,0.0002617546,0.0043876595,0.005159703],"genre_scores_gemma":[0.377238,0.0014543721,0.6033342,0.0012697992,0.0004483782,0.0002052399,0.0012449445,0.0009349189,0.0138701415],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993325,0.0002662366,0.000049176168,0.00015893349,0.00013199869,0.00006114613],"domain_scores_gemma":[0.9979431,0.0012639241,0.000075134754,0.0002777515,0.00038197133,0.000058021113],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001580626,0.00082933984,0.0010506365,0.00081567524,0.00076676905,0.0016548432,0.0013946102,0.0017792145,0.0055995784],"category_scores_gemma":[0.005474296,0.0005853667,0.0010060229,0.0013013337,0.00057609647,0.0041709435,0.0016307187,0.001774339,0.002749508],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00047349668,0.00029155987,0.0013964843,0.0002555763,0.00025598417,0.00028977345,0.00019285346,0.27546114,0.013171976,0.044536762,0.012009058,0.65166533],"study_design_scores_gemma":[0.000021138456,0.000048939266,0.00013997214,0.00001506933,0.00004655597,0.000050420924,0.000023762932,0.95952785,0.0027212726,0.035426244,0.0019645747,0.00001416443],"about_ca_topic_score_codex":0.007235065,"about_ca_topic_score_gemma":0.011840444,"teacher_disagreement_score":0.007235065,"about_ca_system_score_codex":0.00061969395,"about_ca_system_score_gemma":0.0010959979,"threshold_uncertainty_score":0.018732488},"labels":[],"label_agreement":null},{"id":"W2597998535","doi":"10.71781/9988","title":"TAARAC : test d'anglais adaptatif par raisonnement à base de cas","year":2007,"lang":"fr","type":"dissertation","venue":"Open MIND","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Chemistry; Philosophy; Humanities","score_opus":0.04618209860121079,"score_gpt":0.36084953527045716,"score_spread":0.31466743666924635,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2597998535","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.82421094,0.0011727101,0.09183621,0.00060177525,0.0012500143,0.0019268316,0.006933102,0.04072346,0.031344958],"genre_scores_gemma":[0.8412254,0.00037192123,0.116148494,0.0003758457,0.0001700677,0.002014295,0.011432744,0.0037154758,0.024545837],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.997072,0.0010481614,0.00024016543,0.0010374889,0.00046626377,0.00013587502],"domain_scores_gemma":[0.97052413,0.022804923,0.0008076088,0.0025954058,0.0026713193,0.0005966335],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034846773,0.0017442044,0.0008441849,0.0011274918,0.0007019058,0.0018856658,0.0028418393,0.0013698778,0.013818148],"category_scores_gemma":[0.04520977,0.00071855343,0.0008137965,0.00062048435,0.0006797982,0.003549079,0.0017970377,0.0015340414,0.0051529775],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.017361967,0.0049773687,0.055075556,0.0023319472,0.0018803446,0.0008280976,0.0064705373,0.046944708,0.03990576,0.0047680503,0.07139841,0.74805725],"study_design_scores_gemma":[0.0055230116,0.010450787,0.15057403,0.00066828413,0.0024447194,0.0015499606,0.0062331576,0.63353723,0.07222585,0.012040838,0.104051895,0.0007002946],"about_ca_topic_score_codex":0.023152847,"about_ca_topic_score_gemma":0.010719633,"teacher_disagreement_score":0.023152847,"about_ca_system_score_codex":0.0006583335,"about_ca_system_score_gemma":0.0010894005,"threshold_uncertainty_score":0.046226323},"labels":[],"label_agreement":null},{"id":"W2598150924","doi":"10.1142/s2425038416300056","title":"Methods and resources for computing semantic relatedness","year":2017,"lang":"en","type":"article","venue":"Encyclopedia with Semantic Computing and Robotic Intelligence","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"WordNet; Computer science; Information retrieval; Semantic similarity; Key (lock); Section (typography); Selection (genetic algorithm); Resource (disambiguation); Knowledge base; Similarity (geometry); Representation (politics); Natural language processing; Artificial intelligence","score_opus":0.018016765008738445,"score_gpt":0.33276860517557466,"score_spread":0.3147518401668362,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2598150924","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0041333158,0.008336371,0.9473475,0.0012477068,0.000262737,0.0011277931,0.010938861,0.0052365316,0.021369228],"genre_scores_gemma":[0.04046979,0.0049953545,0.92766887,0.00048291803,0.00037639582,0.0033091072,0.016671155,0.00092234183,0.0051040077],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9764104,0.0076108887,0.002546465,0.00373706,0.009183103,0.0005121564],"domain_scores_gemma":[0.9698655,0.017037299,0.0029953779,0.005897731,0.003779639,0.0004244773],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010392546,0.0027419927,0.0021311024,0.03732538,0.0019641903,0.006948071,0.004337165,0.0029251822,0.012442216],"category_scores_gemma":[0.06197668,0.0009484844,0.0025957094,0.03466979,0.0025599818,0.012741235,0.0055735284,0.00273648,0.01091433],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014205823,0.00023880976,0.0049087554,0.0042279772,0.0005444319,0.0002601706,0.0012314258,0.010871296,0.0032072526,0.21626407,0.046873335,0.71123046],"study_design_scores_gemma":[0.00007245185,0.00015970896,0.009730142,0.001920552,0.00025827845,0.0014917443,0.0015104027,0.080997504,0.009093917,0.5706592,0.3237523,0.00035387624],"about_ca_topic_score_codex":0.0038935721,"about_ca_topic_score_gemma":0.00372379,"teacher_disagreement_score":0.03732538,"about_ca_system_score_codex":0.0028434654,"about_ca_system_score_gemma":0.0036731774,"threshold_uncertainty_score":0.05496168},"labels":[],"label_agreement":null},{"id":"W2603667920","doi":"10.13141/fmb.v2012224","title":"Einladung zur Internationalen IAML-Konferenz 2012 in Montréal","year":2015,"lang":"de","type":"article","venue":"Forum Musikbibliothek","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Political science","score_opus":0.026705043437667067,"score_gpt":0.2828025382573186,"score_spread":0.25609749481965155,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2603667920","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06961332,0.040993184,0.1096919,0.16744915,0.060414962,0.003738385,0.07699433,0.03244573,0.438659],"genre_scores_gemma":[0.08673601,0.008905836,0.03333708,0.004004577,0.0034523392,0.0012040492,0.06148931,0.0064262603,0.7944445],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9936092,0.0011637928,0.0003756823,0.0008678644,0.002215417,0.0017679951],"domain_scores_gemma":[0.9884631,0.00064334954,0.00033304578,0.00057731307,0.006029315,0.003953874],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011925597,0.002333983,0.0015905859,0.0064690253,0.0044400967,0.009354485,0.0022730713,0.0021369532,0.07810173],"category_scores_gemma":[0.007736889,0.0011238422,0.0012780634,0.005099429,0.0019138604,0.004344993,0.008023713,0.0044592726,0.026030792],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00034648908,0.00012990985,0.0008867414,0.00018431069,0.000035916106,0.0002776561,0.0007919635,0.0013653907,0.004942305,0.00718516,0.90035397,0.08350019],"study_design_scores_gemma":[0.000045154204,0.00004419027,0.004670787,0.00011743829,0.000026856562,0.00006329294,0.00038923539,0.0014048268,0.003983511,0.00078538415,0.9884112,0.00005820829],"about_ca_topic_score_codex":0.6149419,"about_ca_topic_score_gemma":0.6102487,"teacher_disagreement_score":0.3850581,"about_ca_system_score_codex":0.02433996,"about_ca_system_score_gemma":0.054999564,"threshold_uncertainty_score":0.7746516},"labels":[],"label_agreement":null},{"id":"W2605118633","doi":"10.1162/tacl_a_00047","title":"A Generative Model of Phonotactics","year":2017,"lang":"en","type":"article","venue":"Transactions of the Association for Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":28,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"National Science Foundation","keywords":"Phonotactics; Computer science; Generative grammar; Artificial intelligence; Generative model; Feature (linguistics); Natural language processing; Set (abstract data type); Probabilistic logic; Phonology; Hierarchy; Linguistics; Programming language","score_opus":0.02717737201804962,"score_gpt":0.30507878804265487,"score_spread":0.2779014160246053,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2605118633","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.046919692,0.00038551426,0.93782383,0.0009921156,0.00008644919,0.00007673486,0.0014986537,0.0013503211,0.010866715],"genre_scores_gemma":[0.84742165,0.0007016422,0.13246138,0.0005059533,0.00015383458,0.00037363858,0.0020735373,0.00068216084,0.015626213],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99945754,0.00017035718,0.0000269048,0.00017623165,0.000100737874,0.00006814725],"domain_scores_gemma":[0.9986345,0.0008625222,0.00010641645,0.00021350876,0.00012243274,0.00006061276],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00084870006,0.0006577195,0.00081594265,0.0012150869,0.0006652,0.0020763632,0.0019428002,0.00151388,0.007475665],"category_scores_gemma":[0.00446942,0.0008622581,0.0016359487,0.001217455,0.001403302,0.00320331,0.0015407373,0.001966697,0.001697197],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011711003,0.00007617051,0.005626055,0.00013569981,0.000107343905,0.0004244125,0.0010275883,0.53893876,0.006493406,0.38913697,0.006781249,0.051135182],"study_design_scores_gemma":[0.000015181019,0.000016312924,0.0005902626,0.000019029283,0.000023338878,0.00018422061,0.000034320557,0.88884705,0.0004636635,0.10634659,0.0034385019,0.000021602275],"about_ca_topic_score_codex":0.007385026,"about_ca_topic_score_gemma":0.009196887,"teacher_disagreement_score":0.007475665,"about_ca_system_score_codex":0.0010350001,"about_ca_system_score_gemma":0.0010332714,"threshold_uncertainty_score":0.025008619},"labels":[],"label_agreement":null},{"id":"W2605266158","doi":"10.1145/3132847.3133110","title":"Hierarchical RNN with Static Sentence-Level Attention for Text-Based Speaker Change Detection","year":2017,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":31,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Recurrent neural network; Dialog box; Sentence; Speech recognition; Artificial intelligence; Feature (linguistics); Task (project management); Matching (statistics); Artificial neural network; Point (geometry); Natural language processing","score_opus":0.06095496126915975,"score_gpt":0.3089554900523664,"score_spread":0.24800052878320664,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2605266158","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12413907,0.0017206671,0.8630456,0.00038159452,0.00027683165,0.00014171522,0.0004745012,0.006169202,0.0036509072],"genre_scores_gemma":[0.8172358,0.00038919406,0.17560555,0.00032204422,0.00014565884,0.00014650097,0.0010001295,0.0001878951,0.00496712],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999582,0.00008937612,0.00002601931,0.00017521756,0.00007101513,0.000056445573],"domain_scores_gemma":[0.9993474,0.000304615,0.00006698386,0.00005828551,0.00018720239,0.000035615496],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00081403414,0.0010086591,0.00058606954,0.0005882257,0.0002857611,0.00034995034,0.001077091,0.0006916832,0.0017706116],"category_scores_gemma":[0.0020221965,0.00029057305,0.0005466778,0.00042657633,0.00026036907,0.00094224745,0.00064709445,0.0010350947,0.00089761853],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005445072,0.00035473966,0.0025811307,0.0002304198,0.00015329826,0.00034202385,0.00025680635,0.1922081,0.08657569,0.0028277906,0.006669205,0.70725626],"study_design_scores_gemma":[0.0000076295155,0.000054888154,0.0007037462,0.0000065027907,0.000033900564,0.00003313279,0.000013663706,0.99099064,0.0065856245,0.0010333831,0.0005266347,0.000010162443],"about_ca_topic_score_codex":0.011490934,"about_ca_topic_score_gemma":0.016411198,"teacher_disagreement_score":0.011490934,"about_ca_system_score_codex":0.0007015052,"about_ca_system_score_gemma":0.00078991754,"threshold_uncertainty_score":0.02284813},"labels":[],"label_agreement":null},{"id":"W2605502800","doi":"10.71781/13534","title":"Parsing impoverished syntax","year":2006,"lang":"en","type":"dissertation","venue":"Open MIND","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Syntax; Parsing; Linguistics; Computer science; Natural language processing; Artificial intelligence; Programming language; Philosophy","score_opus":0.017316149246549468,"score_gpt":0.32227695896553515,"score_spread":0.3049608097189857,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2605502800","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08138873,0.04073525,0.59585387,0.02861398,0.005356481,0.0003527535,0.061505146,0.037443165,0.14875062],"genre_scores_gemma":[0.3986527,0.019394448,0.22642167,0.0021677087,0.0014435302,0.0003411911,0.0737633,0.015395094,0.26242042],"study_design_codex":"not_applicable","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99843055,0.0002693493,0.00010617062,0.00036410647,0.00066460564,0.00016523649],"domain_scores_gemma":[0.99654275,0.00066514366,0.0001442667,0.0007678951,0.0017380309,0.00014202275],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00184567,0.00096613786,0.0011294138,0.004437534,0.0021848173,0.005069398,0.0014582006,0.00082998653,0.02745975],"category_scores_gemma":[0.0062754694,0.0011174051,0.000894729,0.0055573564,0.0017762335,0.0048346864,0.0023788805,0.0021449265,0.006017347],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00026910592,0.000077527744,0.0031832727,0.00071066734,0.00019878204,0.0007124622,0.0018825828,0.004240799,0.012720636,0.34183747,0.3627587,0.27140793],"study_design_scores_gemma":[0.000054343072,0.000023244138,0.006195075,0.00041437987,0.00011790711,0.00036281737,0.0004081316,0.015528846,0.014497803,0.08945603,0.8728254,0.000116043935],"about_ca_topic_score_codex":0.24452841,"about_ca_topic_score_gemma":0.31399187,"teacher_disagreement_score":0.24452841,"about_ca_system_score_codex":0.007086141,"about_ca_system_score_gemma":0.008576152,"threshold_uncertainty_score":0.48621023},"labels":[],"label_agreement":null},{"id":"W2605653724","doi":"10.71781/10144","title":"Étude de la traçabilité entre refactorisations du modèle de classes et refactorisations du code","year":2006,"lang":"fr","type":"dissertation","venue":"Open MIND","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Mod; Humanities; Physics; Mathematics; Art; Combinatorics","score_opus":0.04079736489314598,"score_gpt":0.338778309604147,"score_spread":0.297980944711001,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2605653724","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.57419974,0.017311025,0.36590213,0.002427642,0.00039214973,0.00046060618,0.0048198043,0.017110704,0.01737623],"genre_scores_gemma":[0.7519143,0.0027306555,0.22442283,0.00022460218,0.00008674103,0.000313662,0.004772311,0.0018341313,0.013700749],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98005027,0.0038117631,0.0011124518,0.0032978922,0.010799532,0.0009280941],"domain_scores_gemma":[0.8691826,0.08404966,0.007456879,0.013135677,0.024963068,0.001212229],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01567042,0.0011516304,0.0011840194,0.0065196846,0.0016267363,0.0046852967,0.0026657155,0.0014448197,0.005016388],"category_scores_gemma":[0.119647466,0.0014380703,0.0019116475,0.0044221766,0.0015080688,0.00463902,0.0017296037,0.0029667926,0.00096180884],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016941247,0.00067995867,0.080922015,0.001823257,0.0013607892,0.000615535,0.0036801454,0.040595327,0.050364006,0.020806806,0.009670085,0.787788],"study_design_scores_gemma":[0.00055921363,0.0016424146,0.2469959,0.0011881777,0.002502206,0.0010418111,0.0017268539,0.47208327,0.18519194,0.01889056,0.06767399,0.000503668],"about_ca_topic_score_codex":0.17742005,"about_ca_topic_score_gemma":0.17610876,"teacher_disagreement_score":0.17742005,"about_ca_system_score_codex":0.008421486,"about_ca_system_score_gemma":0.007820442,"threshold_uncertainty_score":0.35277468},"labels":[],"label_agreement":null},{"id":"W2605778970","doi":"10.20381/ruor-20152","title":"Machine Translation and Translation Memory Systems: An Ethnographic Study of Translators’ Satisfaction","year":2017,"lang":"en","type":"dissertation","venue":"uO Research (University of Ottawa)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Machine translation; Ethnography; Translation (biology); Machine translation system; Computer science; Linguistics; Psychology; Natural language processing; Sociology; Anthropology; Philosophy; Chemistry","score_opus":0.07446926250646985,"score_gpt":0.35455779449819647,"score_spread":0.2800885319917266,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2605778970","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9978219,0.000193552,0.0005865416,0.00032311733,0.000010644973,0.000049231625,0.00003876282,0.0000049858077,0.00097137585],"genre_scores_gemma":[0.998041,0.00036009136,0.0004381927,0.00024741518,0.000015571344,0.000093363946,0.000045618573,0.000011031233,0.00074778136],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.99305093,0.005129908,0.0002533794,0.00039028996,0.00044954714,0.0007260611],"domain_scores_gemma":[0.9886154,0.007736365,0.0009843576,0.00035466696,0.0014071336,0.0009021481],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005793121,0.00043139872,0.0007754291,0.0015780583,0.0041414835,0.0024487542,0.000900817,0.0013491645,0.0015778608],"category_scores_gemma":[0.013951165,0.0005828953,0.0004317986,0.0016824732,0.0033895303,0.0026303588,0.0023826917,0.0018601073,0.00037082948],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000036263395,0.0000964962,0.016332734,0.000085179534,0.000006334773,0.00078167446,0.97618425,0.000029685658,0.0007975638,0.00017952443,0.00032861598,0.0051417355],"study_design_scores_gemma":[0.0000052605014,0.00020349605,0.010697973,0.000063816326,0.00000618205,0.00032379196,0.98527294,0.00015748247,0.0003206955,0.00009612317,0.002835178,0.000017084481],"about_ca_topic_score_codex":0.005933448,"about_ca_topic_score_gemma":0.008240453,"teacher_disagreement_score":0.005933448,"about_ca_system_score_codex":0.001780972,"about_ca_system_score_gemma":0.0018572552,"threshold_uncertainty_score":0.030637324},"labels":[],"label_agreement":null},{"id":"W2606393360","doi":"10.71781/9906","title":"Mood : un cadre d'applications pour le développement de décodeurs en traduction statistique","year":2006,"lang":"fr","type":"dissertation","venue":"Open MIND","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Humanities; Philosophy; Political science","score_opus":0.03027973416645373,"score_gpt":0.3379846323543191,"score_spread":0.30770489818786534,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2606393360","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008187861,0.005338698,0.9068395,0.0014569913,0.00055964396,0.00025959773,0.003073542,0.058768544,0.015515724],"genre_scores_gemma":[0.06438427,0.004723872,0.8931431,0.0006450739,0.0005205996,0.0005001218,0.0051299864,0.009329535,0.021623492],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99763405,0.0008920232,0.0002193563,0.0004863329,0.0006666994,0.0001016203],"domain_scores_gemma":[0.9943163,0.0032163823,0.00016572764,0.00102532,0.0010803492,0.00019596795],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0041447454,0.0018828398,0.0014645731,0.003071663,0.0010596259,0.0056733503,0.002481305,0.0019160131,0.032443006],"category_scores_gemma":[0.017513487,0.0014315071,0.0015961478,0.0021314523,0.00084949215,0.0049005128,0.0022314356,0.0026109037,0.01456489],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00071408856,0.00015400311,0.002364917,0.0011258885,0.00025671016,0.0002886333,0.00087157474,0.007823471,0.018106317,0.062714465,0.07218931,0.8333906],"study_design_scores_gemma":[0.00046358482,0.0004015215,0.005037464,0.00097063876,0.00048374597,0.00092350255,0.0005562801,0.30704448,0.04304449,0.18028401,0.46052483,0.0002654757],"about_ca_topic_score_codex":0.00866157,"about_ca_topic_score_gemma":0.012961641,"teacher_disagreement_score":0.032443006,"about_ca_system_score_codex":0.001483834,"about_ca_system_score_gemma":0.0014824218,"threshold_uncertainty_score":0.10853273},"labels":[],"label_agreement":null},{"id":"W2606816378","doi":"10.3366/cor.2017.0108","title":"Subtopic annotation and automatic segmentation for news texts in Brazilian Portuguese","year":2017,"lang":"en","type":"article","venue":"Corpora","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Conselho Nacional de Desenvolvimento Científico e Tecnológico; Fundação de Amparo à Pesquisa do Estado de São Paulo","keywords":"Annotation; Computer science; Segmentation; Natural language processing; Portuguese; Artificial intelligence; Process (computing); Rhetorical question; Computational linguistics; Brazilian Portuguese; Linguistics","score_opus":0.02239593382287337,"score_gpt":0.309573640430716,"score_spread":0.28717770660784264,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2606816378","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7801127,0.005109657,0.16390307,0.0018367576,0.0005291325,0.0012323706,0.015193081,0.0027001319,0.02938308],"genre_scores_gemma":[0.7440562,0.0019855928,0.2173753,0.00010957827,0.00021106019,0.0013847542,0.026054641,0.00083758624,0.00798521],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99722916,0.0012566132,0.00029571838,0.000646111,0.00045992466,0.00011236027],"domain_scores_gemma":[0.98455876,0.010280893,0.0012940168,0.0012617189,0.002327778,0.00027682792],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022968363,0.0006252937,0.00053838344,0.0054756952,0.0022733158,0.0015201734,0.0006314571,0.00061918655,0.0029320228],"category_scores_gemma":[0.01973695,0.00048497829,0.00037893114,0.004679099,0.0011383047,0.0016100968,0.0016121254,0.0007802087,0.0008519304],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016707552,0.0004039287,0.014219993,0.0083947815,0.00008434681,0.0039730947,0.101888664,0.006932846,0.17297664,0.023700004,0.02706179,0.6386932],"study_design_scores_gemma":[0.000275418,0.00060384325,0.15781331,0.0024877165,0.00031453816,0.0038463797,0.046317406,0.09339608,0.14588109,0.019460965,0.5291846,0.0004186535],"about_ca_topic_score_codex":0.010285049,"about_ca_topic_score_gemma":0.01685056,"teacher_disagreement_score":0.010285049,"about_ca_system_score_codex":0.001327701,"about_ca_system_score_gemma":0.0017195885,"threshold_uncertainty_score":0.020450354},"labels":[],"label_agreement":null},{"id":"W2607147011","doi":"10.71781/10686","title":"Extraction d'information à partir de transcription de conversations téléphoniques spécialisées","year":2004,"lang":"fr","type":"dissertation","venue":"Open MIND","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Information extraction; Extraction (chemistry); Transcription (linguistics); Computer science; Computational biology; Humanities; Biology; Chemistry; Artificial intelligence; Chromatography; Philosophy; Linguistics","score_opus":0.03865174747257399,"score_gpt":0.3454457365214059,"score_spread":0.3067939890488319,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2607147011","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20238782,0.012281055,0.60526043,0.0037525385,0.0037312948,0.0013906285,0.10320265,0.025135955,0.042857666],"genre_scores_gemma":[0.37192515,0.006174697,0.39780065,0.0005392427,0.0014969212,0.00090869865,0.1460724,0.0026924303,0.07238978],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9983543,0.00040903327,0.00013934026,0.00048298112,0.0003833376,0.00023096938],"domain_scores_gemma":[0.9964204,0.0014845988,0.00019000923,0.000500951,0.001258615,0.00014546108],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008287237,0.0017775256,0.0012878389,0.004482644,0.0011217111,0.0023722472,0.00082555803,0.0016502537,0.016006],"category_scores_gemma":[0.0047568544,0.0006370367,0.0011952023,0.0034655458,0.00057519367,0.0018084422,0.0011755029,0.0013253897,0.0128908595],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001132512,0.00019084133,0.0038594413,0.0018513522,0.0001865528,0.0018768363,0.0025004328,0.0017079259,0.19903868,0.0033167503,0.054207806,0.73013073],"study_design_scores_gemma":[0.00029077457,0.00061786297,0.06631207,0.0010238575,0.0009295097,0.0050052865,0.006593955,0.079859905,0.2782884,0.008702805,0.552042,0.0003336422],"about_ca_topic_score_codex":0.01084477,"about_ca_topic_score_gemma":0.014580186,"teacher_disagreement_score":0.016006,"about_ca_system_score_codex":0.00077875395,"about_ca_system_score_gemma":0.0017128069,"threshold_uncertainty_score":0.053545356},"labels":[],"label_agreement":null},{"id":"W2608615602","doi":"10.18653/v1/d17-1263","title":"A Challenge Set Approach to Evaluating Machine Translation","year":2017,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Machine translation; Computer science; Phrase; Set (abstract data type); Translation (biology); Natural language processing; Artificial intelligence; Divergence (linguistics); Bridge (graph theory); Strengths and weaknesses; Quality (philosophy); Neural system; Machine learning; Linguistics; Programming language; Psychology","score_opus":0.13795820532254432,"score_gpt":0.3916650449870761,"score_spread":0.2537068396645318,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2608615602","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.25860083,0.0038373596,0.69598496,0.0039931233,0.00090373214,0.0013661652,0.008298992,0.0014255473,0.025589194],"genre_scores_gemma":[0.711769,0.00048728278,0.26717305,0.00096909876,0.00044895386,0.0021657972,0.012058139,0.00039790015,0.004530902],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98131186,0.012900963,0.0007929314,0.0012283699,0.0034049773,0.00036098133],"domain_scores_gemma":[0.9604372,0.02838654,0.0014587216,0.0033127556,0.0055388664,0.0008659713],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.015038145,0.0022783254,0.001847673,0.0034822458,0.0015114278,0.0025312714,0.0025602693,0.0025527694,0.0031457609],"category_scores_gemma":[0.065508485,0.00034994198,0.0014952595,0.0028845228,0.0019018747,0.0029091905,0.0042515597,0.0031410125,0.0009648523],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0035131068,0.0020368681,0.03265211,0.0025531014,0.0025524013,0.00076520426,0.0017340577,0.35997272,0.016687654,0.08654638,0.075230196,0.4157562],"study_design_scores_gemma":[0.00023502855,0.0014008675,0.012781261,0.00013632484,0.0001853095,0.00046562863,0.0007318489,0.8419599,0.01260366,0.11031743,0.019030936,0.00015184078],"about_ca_topic_score_codex":0.0027940525,"about_ca_topic_score_gemma":0.0035755306,"teacher_disagreement_score":0.015038145,"about_ca_system_score_codex":0.0016421496,"about_ca_system_score_gemma":0.0012679055,"threshold_uncertainty_score":0.0795303},"labels":[],"label_agreement":null},{"id":"W2609094877","doi":"10.1142/9789813227927_0001","title":"Open information extraction","year":2017,"lang":"en","type":"book-chapter","venue":"World Scientific encyclopedia with semantic computing and robotic intelligence","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Computer science; Natural language processing; Infinitive; Verb; Relationship extraction; Verb phrase; Artificial intelligence; Tuple; Noun; Noun phrase; Relation (database); Variety (cybernetics); Linguistics; Phrase; Simple (philosophy); Information extraction; Mathematics; Data mining; Philosophy","score_opus":0.01962821869585735,"score_gpt":0.2892205845926096,"score_spread":0.26959236589675223,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2609094877","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007651605,0.007940133,0.49581227,0.0021038584,0.0024615005,0.0016236317,0.14599946,0.06454189,0.2718657],"genre_scores_gemma":[0.04568717,0.007831333,0.46283552,0.0011702285,0.00093531795,0.001407296,0.3091052,0.009360268,0.16166764],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9980731,0.00018618381,0.00028946344,0.0004933288,0.00079685816,0.00016105983],"domain_scores_gemma":[0.9970432,0.00066850136,0.00017211493,0.0010710077,0.000924797,0.00012036766],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014444325,0.0017901583,0.001286135,0.013290574,0.0020213865,0.0055155503,0.0015275677,0.0011871035,0.08492113],"category_scores_gemma":[0.0076625887,0.00090401905,0.0017592491,0.0109096,0.0007063807,0.007573886,0.005431529,0.0016977619,0.089114025],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014054078,0.00007594415,0.00082645076,0.0011509306,0.00007448051,0.00029236067,0.00022743897,0.0005003338,0.005414603,0.032844864,0.26197985,0.6964723],"study_design_scores_gemma":[0.00001886615,0.000025395151,0.0011071119,0.0003383715,0.00008727898,0.0006151608,0.00018095889,0.0034254023,0.01442356,0.040647015,0.93908197,0.000048805745],"about_ca_topic_score_codex":0.0014411494,"about_ca_topic_score_gemma":0.0020331067,"teacher_disagreement_score":0.08492113,"about_ca_system_score_codex":0.00079992425,"about_ca_system_score_gemma":0.0025494823,"threshold_uncertainty_score":0.28408945},"labels":[],"label_agreement":null},{"id":"W2610367900","doi":"","title":"Proceedings of the Workshop on Parsing German","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"German; Computer science; Parsing; Natural language processing; Artificial intelligence; Linguistics; Syntax; Czech; Word order; Head-driven phrase structure grammar; Style (visual arts); Dependency grammar; Computational linguistics; Generative grammar; History","score_opus":0.020554018910924243,"score_gpt":0.2750889615725848,"score_spread":0.25453494266166055,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2610367900","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009848715,0.058098517,0.21789183,0.16260402,0.1178753,0.000735541,0.029337823,0.018370673,0.3852376],"genre_scores_gemma":[0.040982507,0.02930694,0.14686783,0.02022648,0.017126163,0.0006915541,0.07512163,0.015062621,0.65461427],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.997889,0.0006611692,0.00016823587,0.0006021783,0.00044530415,0.00023417392],"domain_scores_gemma":[0.9966252,0.0011454708,0.00008217574,0.00063914584,0.0009943566,0.0005136141],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0040743225,0.0016983051,0.0011288433,0.0020433103,0.0013810653,0.0069480445,0.0021352507,0.0028401131,0.14035824],"category_scores_gemma":[0.006107227,0.00078623364,0.0019992446,0.0018829071,0.0014150142,0.008513621,0.003701757,0.0042712167,0.07577712],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014165345,0.000056463276,0.00023450116,0.0002846392,0.00003730429,0.00014440806,0.00032195685,0.00046191743,0.0014428018,0.019958973,0.8299938,0.14692155],"study_design_scores_gemma":[0.000020960062,0.000012669832,0.00041058732,0.00016605991,0.000018176444,0.000095353724,0.00017909065,0.00048740229,0.0005494224,0.009382905,0.9886592,0.000018057232],"about_ca_topic_score_codex":0.005679597,"about_ca_topic_score_gemma":0.008148602,"teacher_disagreement_score":0.14035824,"about_ca_system_score_codex":0.0019694169,"about_ca_system_score_gemma":0.00218947,"threshold_uncertainty_score":0.46954507},"labels":[],"label_agreement":null},{"id":"W2610724063","doi":"10.18608/hla17.030","title":"Linked Data for Learning Analytics: Potentials and Challenges","year":2017,"lang":"en","type":"book-chapter","venue":"Society for Learning Analytics Research (SoLAR) eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Analytics; Computer science; Data science; Data analysis; Learning analytics; Data mining","score_opus":0.23185004642467044,"score_gpt":0.4061054837080487,"score_spread":0.17425543728337828,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2610724063","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0034102928,0.33193415,0.3561254,0.2179625,0.008589462,0.00053557067,0.004368754,0.003980454,0.073093385],"genre_scores_gemma":[0.056936078,0.3977896,0.4525752,0.026437977,0.02142326,0.0013831645,0.008636127,0.002467211,0.032351352],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.98358846,0.008174395,0.000879208,0.0014630349,0.0054733553,0.0004216324],"domain_scores_gemma":[0.9073334,0.07392784,0.0011300235,0.008923366,0.0064349626,0.002250567],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.028524121,0.0017441348,0.0024041713,0.01035177,0.0021658465,0.023991864,0.0068377033,0.006974065,0.010840924],"category_scores_gemma":[0.052323326,0.0015069498,0.0018694397,0.02119253,0.006674841,0.054822147,0.013227881,0.01381653,0.0082893325],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006726669,0.0001338004,0.0011049215,0.0026314359,0.00014095097,0.00016147924,0.0010192029,0.0023884973,0.0004215024,0.34086853,0.1493256,0.5017368],"study_design_scores_gemma":[0.00001661278,0.000022621083,0.00031151876,0.0023089463,0.000020483587,0.00017398085,0.00080381514,0.009961073,0.00043530008,0.45132756,0.5345475,0.00007062314],"about_ca_topic_score_codex":0.0033598708,"about_ca_topic_score_gemma":0.0026099798,"teacher_disagreement_score":0.028524121,"about_ca_system_score_codex":0.004584656,"about_ca_system_score_gemma":0.0055201347,"threshold_uncertainty_score":0.15085173},"labels":[],"label_agreement":null},{"id":"W2610765827","doi":"10.1515/phras-2013-0003","title":"In support of multiword unit classifications: Corpus and human rating data validate phraseological classifications of three different multiword unit types","year":2013,"lang":"en","type":"article","venue":"Yearbook of Phraseology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Computer science; Natural language processing; Linguistics; Artificial intelligence","score_opus":0.10055060976160914,"score_gpt":0.3419018279530505,"score_spread":0.2413512181914414,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2610765827","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.96357393,0.0008414236,0.013791215,0.0002845127,0.00009307463,0.00024027887,0.00072066544,0.000062230916,0.02039273],"genre_scores_gemma":[0.98844916,0.00020274712,0.008964125,0.00013503051,0.00003765149,0.0003343931,0.0009096415,0.00005907276,0.0009080861],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.98763925,0.006457864,0.0014717871,0.0019584463,0.0022542276,0.00021839712],"domain_scores_gemma":[0.84870976,0.09178989,0.015682727,0.022740629,0.019885067,0.0011919618],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013445177,0.00039697878,0.0004528923,0.0021087755,0.0012478699,0.0034952909,0.00092495553,0.00083211745,0.004108733],"category_scores_gemma":[0.094358325,0.0003019713,0.0002503834,0.0024629484,0.0029055055,0.0039948313,0.0023865951,0.0012185101,0.001212037],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0022628047,0.0007001891,0.5300153,0.0020323712,0.00033447298,0.00063481205,0.14459342,0.0011103103,0.04789374,0.015556116,0.008370682,0.24649584],"study_design_scores_gemma":[0.00014247955,0.000745156,0.9067391,0.00063419103,0.00014466987,0.0015650565,0.039774813,0.00856002,0.011923344,0.010050095,0.019514322,0.00020672452],"about_ca_topic_score_codex":0.0029782818,"about_ca_topic_score_gemma":0.0044190916,"teacher_disagreement_score":0.013445177,"about_ca_system_score_codex":0.00047131968,"about_ca_system_score_gemma":0.00035826466,"threshold_uncertainty_score":0.07110578},"labels":[],"label_agreement":null},{"id":"W2611387367","doi":"","title":"Proceedings of the 2nd workshop on Multi-source, Multilingual Information Extraction and Summarization","year":2008,"lang":"en","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"L'Alliance Boviteq","funders":"","keywords":"Automatic summarization; Computer science; Information retrieval; Information extraction; Extraction (chemistry); World Wide Web; Chemistry; Chromatography","score_opus":0.01415274271895416,"score_gpt":0.24982590680400407,"score_spread":0.2356731640850499,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2611387367","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012413939,0.06939391,0.7215786,0.031921715,0.025397979,0.0016655938,0.00773554,0.01519622,0.11469653],"genre_scores_gemma":[0.069597185,0.040556755,0.57044643,0.006454013,0.010658371,0.001651497,0.05573089,0.009582811,0.235322],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99311274,0.0028500487,0.0007169796,0.001224398,0.0016845344,0.0004112419],"domain_scores_gemma":[0.98707426,0.0046643005,0.00036507795,0.0027194507,0.0042308895,0.0009459676],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010018057,0.0020732475,0.0020176913,0.004072415,0.0014479023,0.008884829,0.0041717356,0.003203829,0.04968211],"category_scores_gemma":[0.014071855,0.0008793787,0.002506052,0.0036297508,0.0018485237,0.009061333,0.0059547513,0.0034329533,0.028615797],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004993859,0.0002984342,0.0005771012,0.0016623709,0.00024102652,0.000776006,0.0019319679,0.002721367,0.008730331,0.016819248,0.40996993,0.55577284],"study_design_scores_gemma":[0.00006372964,0.000098628494,0.0009911666,0.00065822277,0.00011087724,0.0004483109,0.0007742454,0.0073078806,0.0038208808,0.021264073,0.9643865,0.00007538167],"about_ca_topic_score_codex":0.0042800126,"about_ca_topic_score_gemma":0.0042101047,"teacher_disagreement_score":0.04968211,"about_ca_system_score_codex":0.00189177,"about_ca_system_score_gemma":0.0033118245,"threshold_uncertainty_score":0.1662032},"labels":[],"label_agreement":null},{"id":"W2611885146","doi":"","title":"Proceedings of the ACL Workshop on Building and Using Parallel Texts","year":2005,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Computer science; Machine translation; Task (project management); Natural language processing; Artificial intelligence; Phrase; Thursday; Presentation (obstetrics); Track (disk drive); World Wide Web; Linguistics; Engineering","score_opus":0.01818337770343694,"score_gpt":0.2855749785700533,"score_spread":0.2673916008666164,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2611885146","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.003760748,0.019792518,0.7268936,0.04007745,0.025484066,0.0019089587,0.024980163,0.04976447,0.10733799],"genre_scores_gemma":[0.031443104,0.018373435,0.57762945,0.010706097,0.008780313,0.00326681,0.14369103,0.025752662,0.18035716],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9817223,0.009478605,0.002008887,0.0027762665,0.0033307956,0.0006831264],"domain_scores_gemma":[0.9683422,0.015120766,0.00082396873,0.008000018,0.006241428,0.0014715604],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.016236411,0.0035151113,0.0033033877,0.0055980897,0.0034228603,0.018221382,0.006585528,0.005908052,0.12402542],"category_scores_gemma":[0.04747858,0.0025737053,0.0035715033,0.0052911504,0.004363134,0.03085373,0.012964179,0.008091636,0.07231643],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00032797692,0.00018489601,0.00029480664,0.0011891334,0.0001495731,0.00036476445,0.0014963873,0.0016816186,0.0021680547,0.039010923,0.7427583,0.21037355],"study_design_scores_gemma":[0.000110147725,0.0000417701,0.00021216391,0.00040063573,0.000050526884,0.0002276966,0.00043929738,0.0037640657,0.0014405979,0.040966563,0.9522744,0.000072115916],"about_ca_topic_score_codex":0.009225939,"about_ca_topic_score_gemma":0.008164442,"teacher_disagreement_score":0.12402542,"about_ca_system_score_codex":0.0027243479,"about_ca_system_score_gemma":0.0053808,"threshold_uncertainty_score":0.41490638},"labels":[],"label_agreement":null},{"id":"W2611973397","doi":"","title":"Proceedings of Ninth Meeting of the ACL Special Interest Group in Computational Morphology and Phonology","year":2007,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":26,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Phonology; Ninth; Linguistics; Computer science; Computational linguistics; sort; Special Interest Group; Artificial intelligence; Philosophy; Information retrieval","score_opus":0.013316314262291703,"score_gpt":0.2619913451206416,"score_spread":0.24867503085834988,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2611973397","genre_codex":"editorial","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010096444,0.035065632,0.067805216,0.19354054,0.4252088,0.0010736851,0.009555574,0.004286083,0.25336805],"genre_scores_gemma":[0.026197981,0.011512761,0.027924793,0.009160391,0.0661069,0.000624853,0.013762547,0.0030464132,0.84166336],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99806315,0.00053716695,0.00016853631,0.0004145918,0.0006318693,0.00018456139],"domain_scores_gemma":[0.99142224,0.001550717,0.00021304817,0.0006774384,0.0040264893,0.0021100584],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0049177925,0.0012769884,0.0015167538,0.0017231151,0.0026873043,0.00750701,0.0021456573,0.0026835306,0.14661516],"category_scores_gemma":[0.008305086,0.0005017579,0.0015283164,0.0011311059,0.001685233,0.0052166823,0.0035966497,0.004339602,0.04902873],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009377253,0.000048374848,0.00018188891,0.000120837794,0.000017982291,0.00012198467,0.000115959265,0.00008362386,0.0006333985,0.0030377996,0.9694862,0.026058206],"study_design_scores_gemma":[0.000037025962,0.000063125924,0.0010608784,0.00013096022,0.000021729109,0.000272279,0.00035593176,0.0005645476,0.0005426577,0.0070810914,0.9898407,0.000029031484],"about_ca_topic_score_codex":0.0034882377,"about_ca_topic_score_gemma":0.00774716,"teacher_disagreement_score":0.14661516,"about_ca_system_score_codex":0.0019447088,"about_ca_system_score_gemma":0.0023020906,"threshold_uncertainty_score":0.4904766},"labels":[],"label_agreement":null},{"id":"W2612919868","doi":"10.21437/interspeech.2017-43","title":"The INTERSPEECH 2017 Computational Paralinguistics Challenge: Addressee, Cold &amp; Snoring","year":2017,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":131,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba","funders":"Engineering and Physical Sciences Research Council; Social Sciences and Humanities Research Council of Canada; National Institutes of Health; National Science Foundation","keywords":"Computer science","score_opus":0.05487383178131913,"score_gpt":0.35326670991496495,"score_spread":0.2983928781336458,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2612919868","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.37361544,0.014576935,0.4111862,0.026795523,0.016232913,0.0018151122,0.071202144,0.023967149,0.060608577],"genre_scores_gemma":[0.51753443,0.0031933454,0.19707473,0.0057037114,0.0035366244,0.00174951,0.19770241,0.0030511878,0.0704541],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9965861,0.0009622745,0.00017743884,0.0008745005,0.0010204581,0.00037916316],"domain_scores_gemma":[0.994894,0.002609482,0.00014436647,0.00078479754,0.001017873,0.0005495526],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019740993,0.0021427171,0.002024884,0.00075275317,0.0012421038,0.0022945877,0.001788707,0.0049697403,0.007635443],"category_scores_gemma":[0.009199498,0.00032176494,0.0011894483,0.00096171943,0.0013899998,0.0026071426,0.0040368503,0.0036472152,0.0073171724],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0024830373,0.0008132494,0.0036546157,0.0016446909,0.00027035855,0.0027708127,0.000731855,0.022082701,0.03264375,0.005749879,0.44355032,0.4836047],"study_design_scores_gemma":[0.00080090546,0.0017249064,0.022500644,0.00038779815,0.00022901682,0.00747487,0.0057273647,0.43196818,0.06115919,0.04205461,0.42543662,0.00053574814],"about_ca_topic_score_codex":0.0072915656,"about_ca_topic_score_gemma":0.015826061,"teacher_disagreement_score":0.007635443,"about_ca_system_score_codex":0.0007870713,"about_ca_system_score_gemma":0.0025840853,"threshold_uncertainty_score":0.025543094},"labels":[],"label_agreement":null},{"id":"W2617355419","doi":"10.71781/10643","title":"Génération automatique de lettres de recrutement","year":2017,"lang":"fr","type":"dissertation","venue":"Open MIND","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Humanities; Art","score_opus":0.04962842029423395,"score_gpt":0.38038848878979387,"score_spread":0.33076006849555994,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2617355419","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.043913793,0.0007064934,0.88790774,0.0005404899,0.00051441224,0.00035459377,0.0021460783,0.05448528,0.009431081],"genre_scores_gemma":[0.15504491,0.00047951465,0.80008197,0.0002446425,0.00010614778,0.00040403605,0.006899393,0.008141766,0.028597618],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99841034,0.00025774553,0.00010858579,0.0005592698,0.0005598985,0.000104156796],"domain_scores_gemma":[0.99544036,0.0022579376,0.00020643357,0.00097147573,0.0009920043,0.00013172129],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014113837,0.0020225844,0.0012593026,0.0019859078,0.001596039,0.0025539668,0.0017793442,0.0014900366,0.01645052],"category_scores_gemma":[0.0071439217,0.0010620311,0.0021561114,0.001212764,0.0010504643,0.00302674,0.002093965,0.0016740215,0.010219376],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010936033,0.0002605456,0.004523465,0.000998519,0.0001556862,0.0014715921,0.00426652,0.026358925,0.107682854,0.019355409,0.047151864,0.78668106],"study_design_scores_gemma":[0.00022702153,0.00059290545,0.005326599,0.0002621796,0.00027476135,0.0016181873,0.0018545145,0.43938097,0.21718542,0.02992421,0.30299494,0.0003583321],"about_ca_topic_score_codex":0.007092797,"about_ca_topic_score_gemma":0.009991105,"teacher_disagreement_score":0.01645052,"about_ca_system_score_codex":0.0009562648,"about_ca_system_score_gemma":0.0011465165,"threshold_uncertainty_score":0.05503249},"labels":[],"label_agreement":null},{"id":"W2617564621","doi":"10.16995/dm.55","title":"Two graphical models for the analysis and comparison of cartularies","year":2015,"lang":"en","type":"article","venue":"Digital Medievalist","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; GRASP; Order (exchange); Visualization; Grid; Theoretical computer science; Data mining; Information retrieval; Data science; Programming language; Mathematics","score_opus":0.04409828269307059,"score_gpt":0.3338599193997451,"score_spread":0.2897616367066745,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2617564621","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0036834467,0.00031963663,0.9877983,0.0005706813,0.000056068904,0.00013112434,0.0012142815,0.0016131668,0.0046132505],"genre_scores_gemma":[0.13813742,0.00076131185,0.85307187,0.00025132397,0.0001595165,0.0007187432,0.0026704893,0.00073128607,0.0034980758],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99465585,0.0024528396,0.00030241592,0.000998272,0.0013664274,0.00022427783],"domain_scores_gemma":[0.9814579,0.013148445,0.0013008727,0.0022784576,0.001474645,0.00033960817],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005186354,0.0014460235,0.00095303945,0.011358946,0.0011214643,0.006155313,0.0028003717,0.0016267427,0.0146022495],"category_scores_gemma":[0.029790543,0.0008600385,0.0030835455,0.008305235,0.0031165013,0.0071842014,0.002669782,0.0018949592,0.002979188],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002859822,0.00015138907,0.0055780057,0.00053154957,0.00027079222,0.00022959811,0.0017260698,0.087403394,0.0019393441,0.6836213,0.01473614,0.20352648],"study_design_scores_gemma":[0.00008195412,0.000079973084,0.0028758384,0.0001888249,0.000114439106,0.0003663948,0.00047850164,0.4084527,0.0011734606,0.5349769,0.05108246,0.00012859007],"about_ca_topic_score_codex":0.008711525,"about_ca_topic_score_gemma":0.009178365,"teacher_disagreement_score":0.0146022495,"about_ca_system_score_codex":0.0024322313,"about_ca_system_score_gemma":0.0015874646,"threshold_uncertainty_score":0.048849344},"labels":[],"label_agreement":null},{"id":"W2617700467","doi":"10.3176/lu.2004.4.04","title":"La réalisation zéro du pronom sujet de première et de deuxième personne  du singulier en finnois et  en estonien  parlés","year":2004,"lang":"fr","type":"article","venue":"Linguistica Uralica","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Eston College","funders":"","keywords":"Humanities; Philosophy","score_opus":0.011155785274669068,"score_gpt":0.29220126626390514,"score_spread":0.28104548098923604,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2617700467","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8734181,0.0014083122,0.014535672,0.0014237101,0.0003820933,0.000028948105,0.0016230086,0.00040049068,0.106779546],"genre_scores_gemma":[0.97156876,0.00027907925,0.0026274277,0.000077047625,0.00004894424,0.00002595527,0.0010299972,0.00023855917,0.024104126],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9988932,0.00031929841,0.00006709103,0.00033826017,0.00024076684,0.00014142376],"domain_scores_gemma":[0.99763656,0.0012023381,0.00025300292,0.0002703622,0.00054332364,0.00009443189],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009240559,0.00043675932,0.00026992167,0.0012287878,0.0019655274,0.0026338538,0.00043620734,0.0006613277,0.016600354],"category_scores_gemma":[0.004893916,0.0004666543,0.00028831657,0.0012959846,0.0018109478,0.0025958477,0.0016376256,0.0012871491,0.0022011937],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017389162,0.00008186206,0.13683924,0.0015798758,0.00011067879,0.003067173,0.25181118,0.00066387653,0.074447446,0.19095674,0.025401182,0.3133018],"study_design_scores_gemma":[0.00007639826,0.0002468564,0.42408127,0.00076005916,0.0001296659,0.005135441,0.111932255,0.0040330705,0.030456189,0.01851885,0.40437084,0.00025916385],"about_ca_topic_score_codex":0.021662764,"about_ca_topic_score_gemma":0.027452117,"teacher_disagreement_score":0.021662764,"about_ca_system_score_codex":0.0020595754,"about_ca_system_score_gemma":0.0008207603,"threshold_uncertainty_score":0.055533648},"labels":[],"label_agreement":null},{"id":"W2617958976","doi":"10.1017/9781316339732.035","title":"Computational Resources: FrameNet and Constructicon","year":2017,"lang":"en","type":"book-chapter","venue":"Cambridge University Press eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"FrameNet; Computer science; Natural language processing; Parsing","score_opus":0.015487689171609689,"score_gpt":0.21396209352385706,"score_spread":0.19847440435224736,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2617958976","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.001233594,0.009249287,0.3504462,0.003762685,0.0009238408,0.00013694918,0.017468132,0.026670013,0.5901093],"genre_scores_gemma":[0.051893707,0.025240354,0.47816384,0.0023128323,0.0013142563,0.0012855252,0.10222966,0.024963312,0.3125965],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99950254,0.000115190436,0.000041328836,0.00012707277,0.00017136896,0.000042491447],"domain_scores_gemma":[0.99944216,0.00020416497,0.000017784818,0.0001839437,0.00010971406,0.00004226672],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00092782197,0.0014968585,0.0009305598,0.003938313,0.0012803522,0.0043324567,0.0024069555,0.0014573754,0.18128824],"category_scores_gemma":[0.0031856056,0.0008385006,0.0007919303,0.0073433444,0.0016490535,0.010570917,0.0026264486,0.0017030119,0.09210734],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00004680621,0.000017128095,0.00006497742,0.0005441998,0.0000108474005,0.000093768744,0.00021541228,0.0011958126,0.00051513826,0.35036555,0.41157985,0.23535061],"study_design_scores_gemma":[0.000008799118,0.000003906045,0.00009446304,0.00021982049,0.0000069257335,0.00012389987,0.00006505985,0.0023006934,0.0006608756,0.15770642,0.83879644,0.000012533758],"about_ca_topic_score_codex":0.0074598943,"about_ca_topic_score_gemma":0.011395658,"teacher_disagreement_score":0.18128824,"about_ca_system_score_codex":0.0020944302,"about_ca_system_score_gemma":0.002020345,"threshold_uncertainty_score":0.60646963},"labels":[],"label_agreement":null},{"id":"W2618498460","doi":"10.1017/9781316339732.030","title":"Metaphor, Simulation, and Fictive Motion","year":2017,"lang":"en","type":"book-chapter","venue":"Cambridge University Press eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Metaphor; Motion (physics); Computer science; Philosophy; Artificial intelligence; Linguistics","score_opus":0.0245039428882623,"score_gpt":0.23836752914488216,"score_spread":0.21386358625661986,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2618498460","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008083957,0.062660895,0.035809536,0.004848175,0.0010209415,0.000030401823,0.000113742004,0.000302266,0.8871301],"genre_scores_gemma":[0.5022487,0.055988155,0.036887318,0.001627412,0.0008347333,0.00018110641,0.00050961185,0.0003018522,0.4014211],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9998362,0.00008044442,0.000005676185,0.000020168014,0.000041242765,0.000016203678],"domain_scores_gemma":[0.9997838,0.00015465323,0.000008167357,0.000026443671,0.000012456309,0.000014409623],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002715476,0.00051629567,0.00029201776,0.0007360318,0.0008278086,0.0021870877,0.00055191654,0.00096927484,0.021951407],"category_scores_gemma":[0.0012045779,0.00016855131,0.0004225365,0.0006740296,0.004052765,0.0027258834,0.0008823079,0.0012431559,0.002320601],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000017560695,0.0000075715834,0.00003967279,0.00026962397,0.000004990171,0.00008073044,0.00120511,0.0012962888,0.0003730669,0.94734895,0.016509574,0.032846883],"study_design_scores_gemma":[0.000011100505,0.000027649226,0.00026256908,0.00038556085,0.0000058415326,0.0004622911,0.0007258357,0.0028430289,0.00052085007,0.54041153,0.4543269,0.000016806978],"about_ca_topic_score_codex":0.0012855283,"about_ca_topic_score_gemma":0.0016562758,"teacher_disagreement_score":0.021951407,"about_ca_system_score_codex":0.0012139439,"about_ca_system_score_gemma":0.0004795431,"threshold_uncertainty_score":0.07343477},"labels":[],"label_agreement":null},{"id":"W2620781045","doi":"10.17613/m65h2m","title":"AIHEC American Indian Collections Portal","year":2012,"lang":"en","type":"article","venue":"Humanities Commons CORE (Modern Language Association / Columbia University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science","score_opus":0.01446258907770077,"score_gpt":0.21875401456759022,"score_spread":0.20429142548988946,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2620781045","genre_codex":"other","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0012654713,0.0004911871,0.002163282,0.0017276014,0.0007058737,0.00031429672,0.11874543,0.015676966,0.8589099],"genre_scores_gemma":[0.017546348,0.0025548479,0.01625172,0.0024076859,0.0010401261,0.000924044,0.29840127,0.012098582,0.6487753],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99889416,0.00010781436,0.00007351445,0.0001138827,0.0006239475,0.00018679441],"domain_scores_gemma":[0.995239,0.0003882152,0.00019788138,0.0013106581,0.0018974378,0.0009667437],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0016858355,0.000833538,0.0006844123,0.008508237,0.0050784852,0.007482128,0.003466809,0.00083265215,0.5077367],"category_scores_gemma":[0.005239411,0.0006193618,0.00041799122,0.019600917,0.00070055085,0.0049704784,0.0060875285,0.0018159705,0.28318107],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000046727826,0.00003455639,0.00030236415,0.00014179717,0.000004674902,0.000082117964,0.0003013267,0.000029836296,0.00034828464,0.0049587744,0.9567049,0.037044533],"study_design_scores_gemma":[0.000004375805,0.0000013006069,0.00045029828,0.000020614516,0.00000198264,0.000032627493,0.00013589152,0.000034524728,0.00010237146,0.00036439567,0.9988424,0.000009283329],"about_ca_topic_score_codex":0.097259305,"about_ca_topic_score_gemma":0.13610148,"teacher_disagreement_score":0.5077367,"about_ca_system_score_codex":0.0023338876,"about_ca_system_score_gemma":0.009364211,"threshold_uncertainty_score":0.70215386},"labels":[],"label_agreement":null},{"id":"W2620853571","doi":"","title":"Stratégies de réécriture probabiliste dans ELAN4","year":2004,"lang":"fr","type":"article","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Prevention of Organ Failure","funders":"","keywords":"Humanities; Art","score_opus":0.011687579553690442,"score_gpt":0.23649294752367242,"score_spread":0.224805367969982,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2620853571","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03929829,0.0014388302,0.93150884,0.0027611735,0.00019693577,0.00015885166,0.00026547626,0.0018854401,0.022486128],"genre_scores_gemma":[0.45394868,0.0009863125,0.48272344,0.0013303312,0.00040755002,0.0006134129,0.00091439264,0.0016155852,0.057460424],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9923528,0.0038477376,0.00030018427,0.0010332403,0.0018231187,0.00064295833],"domain_scores_gemma":[0.98301,0.013140245,0.0003260145,0.0013749107,0.0016228552,0.000526017],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00888136,0.0020823434,0.0019982422,0.0017548191,0.0018538225,0.0053561963,0.0031800023,0.0031823064,0.01186097],"category_scores_gemma":[0.02084067,0.001079955,0.0020694959,0.0010059299,0.0023977652,0.006538862,0.0049258075,0.0060898974,0.0024732584],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011703951,0.00044584484,0.0013440406,0.0003113881,0.00025250003,0.00025124472,0.0007456534,0.16125596,0.0044479035,0.667467,0.008562695,0.1537453],"study_design_scores_gemma":[0.00014552362,0.00017567333,0.00030955192,0.00005761707,0.00006932912,0.000099814046,0.00015339634,0.75902104,0.0057552992,0.2198049,0.014363577,0.000044342425],"about_ca_topic_score_codex":0.010411365,"about_ca_topic_score_gemma":0.02030062,"teacher_disagreement_score":0.01186097,"about_ca_system_score_codex":0.0053231213,"about_ca_system_score_gemma":0.004892198,"threshold_uncertainty_score":0.04696971},"labels":[],"label_agreement":null},{"id":"W2620909954","doi":"","title":"PHMSA rule sets US-Canadian standards","year":2017,"lang":"en","type":"article","venue":"Transport topics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Truck; Business; Engineering; Transport engineering; Automotive engineering","score_opus":0.014731871608444462,"score_gpt":0.2896850168329379,"score_spread":0.27495314522449343,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2620909954","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020286703,0.0009526436,0.11423893,0.0062605324,0.0015957414,0.0015828839,0.08003386,0.012435276,0.7626134],"genre_scores_gemma":[0.26508492,0.0019818936,0.26957202,0.004482006,0.00043961048,0.0012538896,0.12054155,0.004472931,0.33217123],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9851736,0.001308778,0.0013083835,0.0016017844,0.009033871,0.0015735513],"domain_scores_gemma":[0.9762403,0.0024801171,0.00037375168,0.0034979412,0.016808102,0.0005997395],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0069604465,0.0009672915,0.0010194727,0.0067326888,0.0072385594,0.0090606185,0.0054328763,0.0022325793,0.045075856],"category_scores_gemma":[0.017913511,0.0010006917,0.0018732642,0.0055602402,0.0022150783,0.004170456,0.0028170445,0.0032618896,0.016084889],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000490632,0.00019794585,0.0049778456,0.0006376064,0.00007226815,0.00035056684,0.0009833277,0.0049640364,0.0035856096,0.338157,0.48043838,0.1651448],"study_design_scores_gemma":[0.000060375867,0.000028362603,0.0032341075,0.00033498497,0.00008856203,0.00014275976,0.00059858075,0.007301054,0.009643345,0.03937217,0.93909144,0.00010428542],"about_ca_topic_score_codex":0.81237423,"about_ca_topic_score_gemma":0.8666353,"teacher_disagreement_score":0.18762577,"about_ca_system_score_codex":0.019260228,"about_ca_system_score_gemma":0.069895945,"threshold_uncertainty_score":0.37746143},"labels":[],"label_agreement":null},{"id":"W2621151880","doi":"10.1515/jjl-2015-0105","title":"Comments on verb doubling construction in Japanese","year":2015,"lang":"en","type":"article","venue":"Journal of Japanese Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Predicate (mathematical logic); Verb; Sentence; Computer science; Linguistics; Speech recognition; Natural language processing; Mathematics; Artificial intelligence; Philosophy; Programming language","score_opus":0.03421661444261363,"score_gpt":0.3101609212287422,"score_spread":0.2759443067861286,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2621151880","genre_codex":"commentary","genre_gemma":"commentary","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"commentary","genre_consensus":"commentary","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10182096,0.020131478,0.021486234,0.6427232,0.046635907,0.000099129866,0.0004712388,0.0006197517,0.1660121],"genre_scores_gemma":[0.786854,0.0064920713,0.0036379835,0.1403112,0.029013142,0.00012190774,0.0002243174,0.00041788758,0.032927424],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9977386,0.0010374337,0.00017099964,0.0003903302,0.0004114543,0.00025120005],"domain_scores_gemma":[0.98991525,0.0068080556,0.0006335556,0.0004413463,0.0019351379,0.00026665535],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031066153,0.00046278487,0.00045987716,0.0011020708,0.004221685,0.0018521688,0.0014099765,0.0054151323,0.0043083704],"category_scores_gemma":[0.011443432,0.0003156174,0.0003795354,0.0007849124,0.0069169668,0.0020893896,0.001739948,0.007061912,0.00085274177],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00045361888,0.00002276916,0.003824589,0.0008655875,0.0000520618,0.01600801,0.053158846,0.00084877986,0.006388233,0.3918377,0.49614188,0.030397907],"study_design_scores_gemma":[0.000020740166,0.00004885753,0.0032271922,0.0003553839,0.00004843721,0.001868888,0.012985143,0.0009408708,0.0025339015,0.01842893,0.95945674,0.00008492729],"about_ca_topic_score_codex":0.01985967,"about_ca_topic_score_gemma":0.0131232105,"teacher_disagreement_score":0.01985967,"about_ca_system_score_codex":0.003307015,"about_ca_system_score_gemma":0.0018524174,"threshold_uncertainty_score":0.039488137},"labels":[],"label_agreement":null},{"id":"W2621252586","doi":"","title":"5th International Workshop on Computational Terminology (Computerm) joined to the COLING conference 2016","year":2016,"lang":"en","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Terminology; Computer science; Library science; Linguistics; Philosophy","score_opus":0.02608521626844079,"score_gpt":0.28092334690065945,"score_spread":0.25483813063221866,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2621252586","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.027388182,0.029140849,0.6240895,0.07308845,0.06549206,0.0012676086,0.016051972,0.018990114,0.14449123],"genre_scores_gemma":[0.12542799,0.014796223,0.4913588,0.010216338,0.011443258,0.0012811362,0.077987745,0.014041297,0.25344718],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9918257,0.003873495,0.00076539896,0.0013759342,0.0014778629,0.00068160443],"domain_scores_gemma":[0.98787326,0.002935619,0.00028502574,0.0036434962,0.0035572012,0.0017054491],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009753015,0.0014009365,0.0022248854,0.0063169408,0.0031603917,0.010699217,0.0032561123,0.0025980412,0.065233454],"category_scores_gemma":[0.019589897,0.00077868864,0.0029742646,0.0059493035,0.0024794575,0.014156385,0.011930993,0.0053308425,0.032675017],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003447861,0.00022059103,0.0011338709,0.0009315268,0.00009836534,0.0002791967,0.0014468542,0.001021374,0.0063961144,0.11477671,0.51012105,0.36322954],"study_design_scores_gemma":[0.000022301885,0.000050847204,0.0006858928,0.00032712476,0.000040924497,0.0002634543,0.0006197885,0.0038674048,0.0031358174,0.04429048,0.9466555,0.000040436258],"about_ca_topic_score_codex":0.0052717132,"about_ca_topic_score_gemma":0.007359866,"teacher_disagreement_score":0.065233454,"about_ca_system_score_codex":0.003876547,"about_ca_system_score_gemma":0.008176282,"threshold_uncertainty_score":0.21822762},"labels":[],"label_agreement":null},{"id":"W2621304514","doi":"","title":"5Ws: What Went Wrong With Word embeddings","year":2017,"lang":"en","type":"article","venue":"Research Explorer (The University of Manchester)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Open Text (Canada)","funders":"","keywords":"Word (group theory); Linguistics; Computer science; Natural language processing; History; Mathematics; Philosophy","score_opus":0.0686927805555229,"score_gpt":0.3182465839800637,"score_spread":0.2495538034245408,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2621304514","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07925393,0.010323579,0.68973225,0.12244771,0.020572005,0.00025581094,0.008292153,0.025011458,0.04411112],"genre_scores_gemma":[0.47454602,0.0062065674,0.43028864,0.015744945,0.0046976665,0.00039580453,0.01421716,0.013104452,0.040798694],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99260193,0.0029860195,0.00074866903,0.0013564483,0.0019663235,0.00034058592],"domain_scores_gemma":[0.9705483,0.013140679,0.0006937733,0.0066631995,0.00832114,0.00063298125],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008126334,0.0013560667,0.0011864563,0.0025095388,0.0016470288,0.0067920513,0.002497147,0.0034311672,0.020860223],"category_scores_gemma":[0.07384108,0.0009189344,0.0011586165,0.0027078008,0.0029918018,0.03425113,0.004162167,0.006803844,0.010666224],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035234916,0.00011135187,0.005096699,0.0010780665,0.00022345569,0.00023993834,0.0011736099,0.0046459907,0.0029693025,0.15495214,0.14725977,0.6818974],"study_design_scores_gemma":[0.0001120526,0.00014422099,0.0020802664,0.0010061549,0.00019457165,0.0007298146,0.0027739513,0.07527708,0.01315597,0.7028792,0.20145181,0.00019492455],"about_ca_topic_score_codex":0.0052299504,"about_ca_topic_score_gemma":0.0058437157,"teacher_disagreement_score":0.020860223,"about_ca_system_score_codex":0.0012229972,"about_ca_system_score_gemma":0.0018698456,"threshold_uncertainty_score":0.0697844},"labels":[],"label_agreement":null},{"id":"W2626206547","doi":"10.33011/lilt.v2i.1207","title":"The Same Semantic Relations Link Structurally Different Realizations of Concepts","year":2009,"lang":"en","type":"article","venue":"Linguistic Issues in Language Technology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Linguistics; Computer science; Expression (computer science); Noun phrase; Utterance; Realization (probability); Natural language processing; Relation (database); Phrase; Noun; Artificial intelligence; Mathematics","score_opus":0.005994770244023823,"score_gpt":0.30923417278682286,"score_spread":0.303239402542799,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2626206547","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20593598,0.0068530994,0.5452226,0.011090365,0.001449005,0.00072186824,0.0019091002,0.0011625906,0.22565538],"genre_scores_gemma":[0.7891672,0.0028616514,0.19076253,0.0013572539,0.00049812294,0.0006652733,0.0020464743,0.00058674417,0.01205472],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9953868,0.0019438337,0.00036986414,0.0010947628,0.0008506246,0.00035405462],"domain_scores_gemma":[0.99540246,0.0023598275,0.0005243648,0.0010042489,0.00051002635,0.00019904447],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027789965,0.0011745626,0.00068015204,0.0043178643,0.0026787326,0.006722165,0.0019419121,0.0030191462,0.009440796],"category_scores_gemma":[0.010835924,0.0009462869,0.0012557978,0.0030638115,0.018030375,0.022962444,0.0055347,0.0033746636,0.0021667227],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006671355,0.00002878578,0.0010281997,0.0002673341,0.00007532921,0.00051887066,0.015801929,0.00024235652,0.0056222635,0.9522398,0.0012625885,0.022845715],"study_design_scores_gemma":[0.000023395176,0.00004957097,0.0034381652,0.00022133184,0.00011106955,0.0008028514,0.008748771,0.0010537453,0.00256351,0.9083529,0.07454218,0.00009257067],"about_ca_topic_score_codex":0.0023677903,"about_ca_topic_score_gemma":0.0015197629,"teacher_disagreement_score":0.009440796,"about_ca_system_score_codex":0.0023956993,"about_ca_system_score_gemma":0.0017316706,"threshold_uncertainty_score":0.031582654},"labels":[],"label_agreement":null},{"id":"W2626241606","doi":"","title":"Termination of ELAN strategies by simplification - Extended version -","year":2003,"lang":"en","type":"article","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Prevention of Organ Failure","funders":"","keywords":"Mathematical proof; Rewriting; Process (computing); Programming language; Computer science; Transformation (genetics); Calculus (dental); Arithmetic; Mathematics; Algorithm; Theoretical computer science","score_opus":0.007096228704139363,"score_gpt":0.23386613702054787,"score_spread":0.2267699083164085,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2626241606","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.084517196,0.0012742907,0.8024763,0.0014924745,0.0004338298,0.00051700376,0.0008102009,0.0053880224,0.10309067],"genre_scores_gemma":[0.66087735,0.00089430076,0.28395727,0.0011278437,0.00028974182,0.0005436832,0.0021329757,0.0028966325,0.04728016],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9939872,0.0017300223,0.0004621464,0.0006925139,0.0018721544,0.0012559788],"domain_scores_gemma":[0.9855188,0.009032031,0.00036049943,0.003146319,0.001631801,0.00031048464],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003945637,0.0009934424,0.0012120804,0.0013493745,0.0012846458,0.0038518219,0.0025217899,0.0011740257,0.012453869],"category_scores_gemma":[0.016724875,0.00086261064,0.0020805725,0.0012580005,0.0023384653,0.006554083,0.0050111874,0.003323974,0.0029848733],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001123978,0.0003836881,0.0020827942,0.00070821005,0.00022372551,0.00129629,0.0022692068,0.029551782,0.014081528,0.7576659,0.020068895,0.17054404],"study_design_scores_gemma":[0.0002246004,0.000119739816,0.00065194955,0.00012717243,0.00017763069,0.00062143285,0.0003616078,0.12458342,0.020242024,0.82303137,0.029776283,0.00008282816],"about_ca_topic_score_codex":0.002781671,"about_ca_topic_score_gemma":0.002654483,"teacher_disagreement_score":0.012453869,"about_ca_system_score_codex":0.0017240603,"about_ca_system_score_gemma":0.0018796892,"threshold_uncertainty_score":0.041662395},"labels":[],"label_agreement":null},{"id":"W2626778328","doi":"10.65215/2q58a426","title":"Attention Is All You Need","year":2025,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6569,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Machine translation; Transformer; BLEU; Encoder; Artificial intelligence; Parallelizable manifold; Parsing; Language model; Natural language processing; Decoding methods; Task (project management); Convolutional neural network; Speech recognition; Machine learning; Algorithm","score_opus":0.01958296515034699,"score_gpt":0.3049060813157011,"score_spread":0.28532311616535416,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2626778328","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012937478,0.009356464,0.10324918,0.1765424,0.009979519,0.00018219894,0.0029445998,0.010122034,0.6746862],"genre_scores_gemma":[0.21727535,0.00843675,0.04344456,0.07747998,0.0052340976,0.0003113099,0.0038600494,0.005544854,0.6384131],"study_design_codex":"not_applicable","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9988675,0.00026978084,0.000043140488,0.00027601453,0.00033244968,0.00021118995],"domain_scores_gemma":[0.9979519,0.0006074531,0.00012240882,0.000474212,0.00047256416,0.0003715249],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012900358,0.00091858604,0.0005766302,0.00085491885,0.002028028,0.0041229757,0.0011066751,0.0024986858,0.12623918],"category_scores_gemma":[0.008535541,0.00045291043,0.0006130619,0.0008899282,0.0018910087,0.010175814,0.0039558113,0.0037103046,0.06712462],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015855968,0.00006500304,0.0016135972,0.00028408112,0.00004232508,0.00031492152,0.0015501674,0.0005070161,0.0024926187,0.07725223,0.6155868,0.30013272],"study_design_scores_gemma":[0.000013124931,0.000028820856,0.0009472517,0.00017008046,0.000029480922,0.00039146026,0.000664629,0.00097133714,0.0009812773,0.0611787,0.93458533,0.000038523267],"about_ca_topic_score_codex":0.0045779278,"about_ca_topic_score_gemma":0.006797184,"teacher_disagreement_score":0.12623918,"about_ca_system_score_codex":0.001192543,"about_ca_system_score_gemma":0.0012533291,"threshold_uncertainty_score":0.42231214},"labels":[],"label_agreement":null},{"id":"W2649134529","doi":"","title":"An Anytime Deduction Algorithm for the Probabilistic Logic and Entailment Problems","year":2006,"lang":"en","type":"article","venue":"Les Cahiers du GERAD","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University; Group for Research in Decision Analysis","funders":"","keywords":"Probabilistic logic; Logical consequence; Probabilistic argumentation; Consistency (knowledge bases); Range (aeronautics); Set (abstract data type); Algorithm; Computer science; Probabilistic CTL; Imprecise probability; Mathematics; Artificial intelligence; Theoretical computer science; Probabilistic analysis of algorithms; Probability distribution; Statistics; Programming language","score_opus":0.007834545499640567,"score_gpt":0.23277187697653298,"score_spread":0.2249373314768924,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2649134529","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0033523482,0.00009518327,0.9943514,0.00013402732,0.00003052557,0.0000681396,0.000081750906,0.00073624117,0.0011503136],"genre_scores_gemma":[0.046642337,0.00010700439,0.9515881,0.00008995253,0.00005414657,0.00008846627,0.00026848668,0.000103265724,0.0010582504],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.998643,0.00032129377,0.0001152757,0.00030280466,0.00051738723,0.00010030544],"domain_scores_gemma":[0.99699795,0.0020454968,0.00014531042,0.0003691579,0.00036916495,0.00007293454],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015516654,0.00092626264,0.000932856,0.0011199744,0.000857718,0.0018723585,0.0020414935,0.0009033952,0.005461719],"category_scores_gemma":[0.0069319583,0.00049714284,0.0016746058,0.0013261896,0.00089318986,0.003168903,0.0018667296,0.0023008576,0.0011054828],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023556298,0.00027000371,0.00082538754,0.00054121285,0.0000907931,0.0002968769,0.00042249926,0.08159616,0.006130777,0.24587525,0.008689128,0.6550263],"study_design_scores_gemma":[0.00012278381,0.00010548429,0.00020400844,0.00004432941,0.000078579986,0.00035026722,0.00007563027,0.6423429,0.007671183,0.33694914,0.012024851,0.000030856587],"about_ca_topic_score_codex":0.0021268446,"about_ca_topic_score_gemma":0.0034787084,"teacher_disagreement_score":0.005461719,"about_ca_system_score_codex":0.0011511463,"about_ca_system_score_gemma":0.002046056,"threshold_uncertainty_score":0.018271267},"labels":[],"label_agreement":null},{"id":"W2650750770","doi":"","title":"An exploration of the translation of MEND 5-7 for the BC context using the RE-AIM framework","year":2014,"lang":"en","type":"dissertation","venue":"UVic’s Research and Learning Repository (University of Victoria)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Context (archaeology); Translation (biology); Computer science; Geography; Archaeology; Chemistry","score_opus":0.056445931044394715,"score_gpt":0.3343189868475464,"score_spread":0.2778730558031517,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2650750770","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8034607,0.003153667,0.038558044,0.02187721,0.0007874485,0.025716944,0.0026459526,0.00061995594,0.10318013],"genre_scores_gemma":[0.7175259,0.0029352459,0.23540668,0.0030743869,0.000065335946,0.023236988,0.0015664112,0.00026459421,0.015924562],"study_design_codex":"design_other","study_design_gemma":"qualitative","domain_scores_codex":[0.9691733,0.02259142,0.0011829216,0.00072948897,0.0047747134,0.0015480778],"domain_scores_gemma":[0.963508,0.02160151,0.0013018505,0.0019194415,0.009653501,0.0020158025],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.044668898,0.00069217087,0.0007811472,0.0012351914,0.004616901,0.0041979677,0.0027980197,0.00083216676,0.005508932],"category_scores_gemma":[0.046347678,0.0005308449,0.00067970133,0.0017129529,0.0020386833,0.0010113462,0.0044016275,0.0024712568,0.0004735969],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016036564,0.0026172649,0.02369244,0.0068066744,0.00011637541,0.0019855986,0.2987183,0.0025885566,0.008172883,0.032783803,0.021086534,0.599828],"study_design_scores_gemma":[0.00092561677,0.0037561485,0.105409004,0.015250172,0.00043451018,0.00084489444,0.3350458,0.00903083,0.009483266,0.0067055738,0.51277936,0.0003347544],"about_ca_topic_score_codex":0.30411586,"about_ca_topic_score_gemma":0.5898482,"teacher_disagreement_score":0.6958841,"about_ca_system_score_codex":0.043447275,"about_ca_system_score_gemma":0.0815918,"threshold_uncertainty_score":0.6046914},"labels":[],"label_agreement":null},{"id":"W26623053","doi":"10.3892/mco.2015.580","title":"Generating Coherent Extracts of Single Documents Using Latent Semantic Analysis","year":2003,"lang":"en","type":"article","venue":"Molecular and Clinical Oncology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"University of Toronto; University of Pennsylvania","keywords":"Latent semantic analysis; Computer science; Natural language processing; Coherence (philosophical gambling strategy); Artificial intelligence; Information retrieval; Similarity (geometry); Identification (biology); Semantic similarity; Semantics (computer science); Probabilistic latent semantic analysis; Vector space model; Topic model; Statistics; Mathematics; Programming language","score_opus":0.04939897924259892,"score_gpt":0.39150648412221184,"score_spread":0.34210750487961294,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W26623053","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.047830895,0.0050279074,0.89573234,0.0020750398,0.0008162554,0.0012564507,0.023678381,0.0191537,0.004429057],"genre_scores_gemma":[0.3060127,0.0025704722,0.6361556,0.00044076654,0.001020184,0.0013675728,0.04746623,0.0007529341,0.0042136107],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99637526,0.0007725254,0.0005419923,0.0013211532,0.00069171534,0.00029735485],"domain_scores_gemma":[0.99457186,0.0033431211,0.000431199,0.00059297244,0.0008935699,0.00016722645],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003418788,0.0032705783,0.0019223848,0.011358217,0.0012574226,0.0031061927,0.0020813113,0.00301629,0.0077842474],"category_scores_gemma":[0.010368568,0.00066059857,0.0038323582,0.0073022526,0.0011037829,0.004948129,0.0025566076,0.0021989348,0.0053364662],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0025332205,0.0007875846,0.0042662285,0.0021559063,0.0006620771,0.0012281986,0.00067741075,0.022456974,0.026571188,0.007850055,0.030295534,0.9005156],"study_design_scores_gemma":[0.00042486278,0.000995213,0.006271019,0.00046582174,0.0008450155,0.0015466125,0.0014551856,0.844759,0.027224107,0.071811296,0.043928076,0.00027372144],"about_ca_topic_score_codex":0.004193766,"about_ca_topic_score_gemma":0.0042768335,"teacher_disagreement_score":0.011358217,"about_ca_system_score_codex":0.0012878248,"about_ca_system_score_gemma":0.0029054296,"threshold_uncertainty_score":0.026040912},"labels":[],"label_agreement":null},{"id":"W2685066591","doi":"","title":"Found in Translation","year":2009,"lang":"en","type":"article","venue":"Bristol Research (University of Bristol)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; RSS; Machine translation; World Wide Web; Support vector machine; European union; Information retrieval; Artificial intelligence; Natural language processing; Data science","score_opus":0.06430890233622087,"score_gpt":0.3348878534340906,"score_spread":0.2705789510978697,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2685066591","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007656396,0.008920703,0.09681344,0.01044391,0.014342051,0.00058895675,0.0354684,0.024463516,0.80130255],"genre_scores_gemma":[0.08375014,0.011485215,0.10646691,0.0060743983,0.0031980553,0.00056479505,0.07291491,0.013846495,0.7016991],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.996482,0.0006871988,0.0005451986,0.00081301125,0.0010692443,0.00040329967],"domain_scores_gemma":[0.9945161,0.0008855413,0.00024529017,0.0021302006,0.0019074631,0.00031546503],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0019707736,0.0018782287,0.0013574952,0.0069140852,0.0037473177,0.008962491,0.001760742,0.001970149,0.2989434],"category_scores_gemma":[0.01027133,0.0007073585,0.0011952514,0.009292076,0.0015353624,0.007770684,0.0067778905,0.0018427015,0.22364697],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018671154,0.000084735875,0.0014379008,0.0010810833,0.00004073096,0.00091520255,0.0015316289,0.00025392728,0.0036247172,0.050832514,0.42452082,0.51549006],"study_design_scores_gemma":[0.000012909729,0.00002119097,0.00040637495,0.00017666024,0.000022254528,0.00041669258,0.00037545024,0.0002846206,0.001628322,0.008848277,0.9877908,0.000016526807],"about_ca_topic_score_codex":0.0033406848,"about_ca_topic_score_gemma":0.003981538,"teacher_disagreement_score":0.2989434,"about_ca_system_score_codex":0.0015785978,"about_ca_system_score_gemma":0.0032528376,"threshold_uncertainty_score":0.99997216},"labels":[],"label_agreement":null},{"id":"W26866788","doi":"10.1088/0957-4484/27/10/105204","title":"Experiments for HARD and Enterprise Tracks.","year":2005,"lang":"en","type":"article","venue":"Text REtrieval Conference","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Fundamental Research Funds for the Central Universities; National Natural Science Foundation of China","keywords":"Cohesion (chemistry); Computer science; Natural language processing; Lexical database; Social connectedness; Artificial intelligence; Information retrieval; Selection (genetic algorithm); Lexical item; Schema (genetic algorithms); Linguistics; Psychology; WordNet","score_opus":0.034836813912769604,"score_gpt":0.3130115114656667,"score_spread":0.2781746975528971,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W26866788","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.24833947,0.0026586056,0.071541,0.0041103824,0.0028633245,0.0017370501,0.020257149,0.004861486,0.6436316],"genre_scores_gemma":[0.4012441,0.00140186,0.054445487,0.0018165715,0.00013577624,0.0011330479,0.013788679,0.0005466238,0.52548784],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99860185,0.00015467677,0.00006078235,0.0004987445,0.00047670113,0.00020724388],"domain_scores_gemma":[0.9985348,0.00018451219,0.000093671515,0.0004250526,0.0004840334,0.000278014],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010964929,0.00046042638,0.00032137067,0.00073641277,0.0018929753,0.002027692,0.0008902908,0.0013052505,0.10026174],"category_scores_gemma":[0.0020425587,0.00026721953,0.00031592717,0.0012389248,0.0004859813,0.0020799292,0.0014434445,0.0010459552,0.03216751],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0038063251,0.0019848002,0.016645005,0.0012259298,0.00014708085,0.0013921518,0.0035004034,0.0014712538,0.28721002,0.1077391,0.20186695,0.37301105],"study_design_scores_gemma":[0.0002561547,0.0012704212,0.010447539,0.00012746923,0.000066863955,0.0008813381,0.0019023834,0.0025605836,0.15892836,0.014852965,0.8086368,0.00006919235],"about_ca_topic_score_codex":0.0033070398,"about_ca_topic_score_gemma":0.0052233906,"teacher_disagreement_score":0.10026174,"about_ca_system_score_codex":0.00094163354,"about_ca_system_score_gemma":0.0010094013,"threshold_uncertainty_score":0.335409},"labels":[],"label_agreement":null},{"id":"W2695433898","doi":"10.29173/cais726","title":"Évaluation de la contribution des termes composés sytaxiques pour l'indexation automatique","year":2013,"lang":"fr","type":"article","venue":"Proceedings of the Annual Conference of CAIS / Actes du congrès annuel de l ACSI","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":true,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Humanities; Philosophy; Mathematics; Political science","score_opus":0.021362434020459425,"score_gpt":0.27890605857165973,"score_spread":0.2575436245512003,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2695433898","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7049333,0.008335826,0.25857735,0.00088598003,0.00041072944,0.00091921113,0.0032372018,0.008181338,0.014519098],"genre_scores_gemma":[0.5022319,0.0026926615,0.46532032,0.00014306707,0.00017423138,0.00048213906,0.006586287,0.001469462,0.020899951],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99588734,0.001048959,0.0004915368,0.000616095,0.0018058149,0.00015031612],"domain_scores_gemma":[0.9822374,0.011378319,0.0005919718,0.0012587669,0.0042556655,0.00027783393],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005056286,0.0013168468,0.0013346628,0.0042855516,0.00082143594,0.0025783828,0.0011830229,0.00082145043,0.00656601],"category_scores_gemma":[0.019387322,0.00043653513,0.0008843446,0.0038007281,0.00065904815,0.0020211767,0.0013500551,0.00074115547,0.0023952436],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017968704,0.00021742737,0.009332725,0.0014016541,0.00031231614,0.00021805677,0.0013053326,0.009060325,0.09693903,0.0016485873,0.0047603543,0.87300736],"study_design_scores_gemma":[0.0007086589,0.0033262244,0.091294244,0.0005878824,0.0014630029,0.0017550853,0.0028516606,0.42646527,0.37521714,0.005466582,0.09053222,0.000331978],"about_ca_topic_score_codex":0.017294211,"about_ca_topic_score_gemma":0.023789478,"teacher_disagreement_score":0.017294211,"about_ca_system_score_codex":0.0013843621,"about_ca_system_score_gemma":0.0017809863,"threshold_uncertainty_score":0.034387052},"labels":[],"label_agreement":null},{"id":"W2698016272","doi":"","title":"Aboriginal Canada Portal: Download Inuktitut Fonts","year":2007,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Download; Computer science; World Wide Web","score_opus":0.0047962746445585,"score_gpt":0.2634558427620721,"score_spread":0.2586595681175136,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2698016272","genre_codex":"other","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0019405963,0.0007895659,0.00067441986,0.0012007016,0.00069754873,0.0001481819,0.1741033,0.010010533,0.8104351],"genre_scores_gemma":[0.0048629865,0.0009109529,0.0010642514,0.00022798855,0.000078909405,0.00006917747,0.039611958,0.0037306598,0.9494431],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9995509,0.000020552356,0.00001253272,0.00004152304,0.00021033232,0.00016425452],"domain_scores_gemma":[0.9969982,0.00024729932,0.000053219082,0.0002363046,0.00197462,0.0004904053],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0005182481,0.001393548,0.0011876605,0.0039584516,0.0046372856,0.005927671,0.0016564297,0.0009689517,0.8539235],"category_scores_gemma":[0.00272388,0.00067659107,0.0009617442,0.008468519,0.0008279596,0.0022509345,0.0019604166,0.001091015,0.63044196],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000047083235,0.000017330884,0.0002152925,0.00013818762,0.000003743701,0.000041146228,0.00018817082,0.000033173255,0.00016186938,0.0009994917,0.9799477,0.01820674],"study_design_scores_gemma":[0.000023094397,0.000007745878,0.001968251,0.00010880174,0.0000101655805,0.00004014431,0.0004083545,0.000111279216,0.0004783801,0.0006208554,0.99619544,0.000027506998],"about_ca_topic_score_codex":0.8287151,"about_ca_topic_score_gemma":0.884373,"teacher_disagreement_score":0.8539235,"about_ca_system_score_codex":0.0066610607,"about_ca_system_score_gemma":0.017957328,"threshold_uncertainty_score":0.34458727},"labels":[],"label_agreement":null},{"id":"W271245702","doi":"","title":"Software Language Engineering: First International Conference, SLE 2008, Toulouse, France, September 29-30, 2008. Revised Selected Papers","year":2009,"lang":"en","type":"book","venue":"Springer eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Athabasca University","funders":"","keywords":"Library science; Software engineering; Engineering; Computer science","score_opus":0.007069556943934446,"score_gpt":0.22296956040711832,"score_spread":0.2159000034631839,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W271245702","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022617256,0.22420707,0.42601874,0.030305043,0.03658903,0.00066298794,0.005756556,0.019617079,0.23422632],"genre_scores_gemma":[0.021725848,0.057253506,0.07318434,0.0017313835,0.0033366669,0.00023290589,0.0067294263,0.003349613,0.8324563],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9989492,0.00018092754,0.00006295608,0.00021475986,0.00049142336,0.00010061024],"domain_scores_gemma":[0.997974,0.00049780286,0.000059572754,0.00020381936,0.0009631598,0.00030166475],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031763962,0.0015874086,0.0014376286,0.0025916127,0.000807049,0.0035208147,0.0014199723,0.0013034206,0.059343383],"category_scores_gemma":[0.0031252494,0.0006667468,0.00091971626,0.0023451452,0.00089161174,0.0030972974,0.0018269714,0.002165019,0.030706493],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000104250146,0.00010200151,0.00033175907,0.0004949753,0.000028789344,0.00013010495,0.00022555431,0.0011310092,0.0036212925,0.004936838,0.62006456,0.3688289],"study_design_scores_gemma":[0.000041761286,0.000157714,0.0011390637,0.0003988506,0.000029650122,0.00055496936,0.0001732214,0.0026653432,0.003650481,0.004631535,0.98652864,0.0000287432],"about_ca_topic_score_codex":0.0036489624,"about_ca_topic_score_gemma":0.008247662,"teacher_disagreement_score":0.059343383,"about_ca_system_score_codex":0.0017853378,"about_ca_system_score_gemma":0.0027497592,"threshold_uncertainty_score":0.1985234},"labels":[],"label_agreement":null},{"id":"W2714493922","doi":"10.4018/ijsvr.2017010107","title":"Identifying the Meta-Forms of Situations","year":2017,"lang":"en","type":"article","venue":"International Journal of Semiotics and Visual Rhetoric","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Representation (politics); Semantics (computer science); Meaning (existential); Context (archaeology); Similarity (geometry); Function (biology); Identification (biology); Artificial intelligence; Domain (mathematical analysis); Value (mathematics); Space (punctuation); Natural language processing; Programming language; Mathematics; Epistemology; Machine learning","score_opus":0.05886108197296353,"score_gpt":0.3920343389944102,"score_spread":0.3331732570214467,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2714493922","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12474956,0.0011222126,0.7725422,0.0025051078,0.00024033707,0.0004952619,0.00075959606,0.0010548613,0.096530914],"genre_scores_gemma":[0.70805085,0.00042252167,0.2863501,0.00012138698,0.000042081803,0.00022964331,0.00058283086,0.00018733362,0.0040132226],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99704915,0.001170728,0.00032055835,0.00066702854,0.0006174694,0.00017508886],"domain_scores_gemma":[0.9968767,0.0012741493,0.00031839748,0.00088114897,0.0004687479,0.00018087668],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002135904,0.0007949709,0.00048993813,0.0044494774,0.001791517,0.0062888768,0.0013918463,0.0014786582,0.00979823],"category_scores_gemma":[0.010287864,0.0006531787,0.0012973932,0.0017553271,0.004474818,0.013144923,0.0037277406,0.0014908889,0.0014457571],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000089890025,0.000034196717,0.001997881,0.0003232725,0.00003797065,0.0004871357,0.008872336,0.001370277,0.0047313864,0.93843675,0.0012462642,0.04237264],"study_design_scores_gemma":[0.000020518974,0.000059963942,0.0021080885,0.00021743956,0.000049071165,0.000938751,0.00914491,0.016440827,0.0044178925,0.9304527,0.03609659,0.000053257507],"about_ca_topic_score_codex":0.0007572572,"about_ca_topic_score_gemma":0.0010204619,"teacher_disagreement_score":0.00979823,"about_ca_system_score_codex":0.0013443045,"about_ca_system_score_gemma":0.0010481067,"threshold_uncertainty_score":0.032778323},"labels":[],"label_agreement":null},{"id":"W2724469086","doi":"","title":"Classification de transcriptions automatiques imparfaites : Doit-on adapter le calcul du taux d'erreur-mot ?","year":2014,"lang":"fr","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Humanities; Philosophy; Adapter (computing); Computer science","score_opus":0.023465184916341367,"score_gpt":0.25637767931964855,"score_spread":0.2329124944033072,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2724469086","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.43091616,0.007081978,0.49896112,0.004066035,0.002492697,0.00063751725,0.015188132,0.02516592,0.015490502],"genre_scores_gemma":[0.7612893,0.0010494345,0.1981537,0.00047189638,0.0008376781,0.0002655744,0.02294618,0.0021734287,0.012812857],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99030626,0.002449093,0.0012117242,0.0024795756,0.0027925915,0.0007608492],"domain_scores_gemma":[0.9454507,0.031369176,0.0039267875,0.0051616603,0.013076416,0.0010151631],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006589494,0.0016985922,0.0015774041,0.0052420455,0.0011692903,0.005469625,0.0022772697,0.003531259,0.009119629],"category_scores_gemma":[0.055128787,0.0006882423,0.0011919843,0.003812121,0.0010920734,0.0036184485,0.001674821,0.0024386668,0.01427484],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021241657,0.0002860953,0.049677268,0.0012336584,0.00042839732,0.00063472503,0.0011974385,0.010748878,0.050377477,0.0029057655,0.028808508,0.85157764],"study_design_scores_gemma":[0.00041950247,0.0009267372,0.09318732,0.00095407816,0.001064234,0.0031823255,0.0061230604,0.678217,0.13797794,0.020511769,0.057143476,0.0002926014],"about_ca_topic_score_codex":0.008581521,"about_ca_topic_score_gemma":0.010733912,"teacher_disagreement_score":0.009119629,"about_ca_system_score_codex":0.0016211972,"about_ca_system_score_gemma":0.0020249372,"threshold_uncertainty_score":0.03484893},"labels":[],"label_agreement":null},{"id":"W2726154201","doi":"","title":"Giving back to the language community: lessons from the Kiowa and Ojibwe peoples","year":2017,"lang":"en","type":"article","venue":"The COCOON platform (University of Paris)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Commit; Tribe; Linguistics; History; Computer science; Sociology; Anthropology; Philosophy","score_opus":0.033979421795715836,"score_gpt":0.269982291153701,"score_spread":0.23600286935798515,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2726154201","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.85247225,0.0060110353,0.0008408392,0.08611809,0.002105776,0.00020543255,0.000091369286,0.00005970878,0.052095473],"genre_scores_gemma":[0.9455201,0.005100255,0.0009538496,0.01898396,0.0003350986,0.0002498989,0.000088064764,0.00015305704,0.028615704],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.99203885,0.0037637886,0.00016463902,0.0005270339,0.0006243409,0.0028814054],"domain_scores_gemma":[0.98766476,0.0043156287,0.00055725407,0.00037397374,0.001066456,0.0060218778],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009448759,0.0011626297,0.001384451,0.0028145877,0.08679691,0.017544078,0.0051158983,0.0073084356,0.007852593],"category_scores_gemma":[0.012804556,0.0012879518,0.0006896809,0.0025281587,0.036157105,0.022765137,0.029645693,0.016130283,0.0011848062],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000059997346,0.000035681773,0.00075494166,0.000029557767,0.0000017008897,0.00045905568,0.993248,0.0000032606,0.00005328106,0.0010612537,0.0014536236,0.0028935804],"study_design_scores_gemma":[0.0000019670076,0.0000068854574,0.00037852867,0.00005755998,0.0000013288702,0.000053433407,0.9905577,0.0000036641,0.00001302491,0.00035767132,0.0085631395,0.0000051390252],"about_ca_topic_score_codex":0.16412532,"about_ca_topic_score_gemma":0.34770265,"teacher_disagreement_score":0.16412532,"about_ca_system_score_codex":0.010483362,"about_ca_system_score_gemma":0.019907499,"threshold_uncertainty_score":0.32634002},"labels":[],"label_agreement":null},{"id":"W2730286782","doi":"10.1093/geroni/igx004.2476","title":"ENABLING KNOWLEDGE TRANSLATION THROUGH THE CANADIAN DEPRESCRIBING NETWORK","year":2017,"lang":"en","type":"article","venue":"Innovation in Aging","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Alberta Health; Alberta Health Services; Women's College Hospital; Bruyère; University of British Columbia; Université de Montréal","funders":"","keywords":"Deprescribing; Medical prescription; Set (abstract data type); Action plan; Enabling; Plan (archaeology); Public relations; Business; Nursing; Medicine; Polypharmacy; Political science; Computer science; Management; Psychiatry; History; Pharmacology","score_opus":0.07744380998483788,"score_gpt":0.3356384969583223,"score_spread":0.25819468697348447,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2730286782","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.101062536,0.0065472606,0.18139265,0.14192024,0.0022338587,0.0040926253,0.010659431,0.01123533,0.54085606],"genre_scores_gemma":[0.44603652,0.007274453,0.4122293,0.014264821,0.00052527105,0.0024082018,0.015360534,0.0017109342,0.10018991],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9779273,0.009245302,0.0013201324,0.0021960104,0.006995575,0.0023156235],"domain_scores_gemma":[0.94246745,0.02086311,0.0015597445,0.0055763293,0.023865541,0.0056677395],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.027955806,0.0008711004,0.0006168657,0.006849027,0.011464824,0.008592924,0.0026546158,0.0020602038,0.013277948],"category_scores_gemma":[0.052220486,0.000523076,0.0006438623,0.0070666485,0.004421664,0.0063259834,0.014689293,0.0022750208,0.0035387974],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":true,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003093073,0.00030400645,0.006258672,0.0010171623,0.00005029895,0.0009778187,0.028499233,0.0023456954,0.004568491,0.062112167,0.26918927,0.62436795],"study_design_scores_gemma":[0.000084637235,0.000061787425,0.0055108,0.00058478,0.000038561666,0.00014246398,0.012941868,0.008362048,0.0025744557,0.020015078,0.9495444,0.00013910474],"about_ca_topic_score_codex":0.8586037,"about_ca_topic_score_gemma":0.8975957,"teacher_disagreement_score":0.95832133,"about_ca_system_score_codex":0.041678663,"about_ca_system_score_gemma":0.16162758,"threshold_uncertainty_score":0.3024013},"labels":[],"label_agreement":null},{"id":"W2732177421","doi":"10.1017/cnj.2017.35","title":"What's mine is yours: Stable variation and language change in Ancient Egyptian possessive constructions","year":2017,"lang":"en","type":"article","venue":"The Canadian Journal of Linguistics / La revue canadienne de linguistique","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Possessive; Variation (astronomy); Clitic; Possession (linguistics); Noun phrase; Linguistics; Language change; Phrase; Computer science; Noun; Philosophy","score_opus":0.01717365647774312,"score_gpt":0.2816798911099879,"score_spread":0.2645062346322448,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2732177421","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9984281,0.000053548567,0.0006001157,0.000039033763,6.7869877e-7,0.0000014656805,0.000033639546,0.0000030381711,0.0008403261],"genre_scores_gemma":[0.9997179,0.000012674447,0.00015949701,0.000001921118,6.5835195e-7,8.8275857e-7,0.000020047853,0.0000023938678,0.00008388662],"study_design_codex":"observational","study_design_gemma":"not_applicable","domain_scores_codex":[0.99934417,0.00024580915,0.000059959122,0.00012926897,0.00015259131,0.000068250214],"domain_scores_gemma":[0.99684864,0.0018179653,0.00071511,0.0002497647,0.00029603607,0.00007242171],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011439802,0.00012943012,0.00018981929,0.001237997,0.00067279,0.0012743227,0.00026731534,0.00023800896,0.0015057563],"category_scores_gemma":[0.0037371737,0.00014088067,0.00014834556,0.0020376232,0.0029573361,0.0012624898,0.0008786027,0.0003849396,0.00008684711],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012062592,0.00007885815,0.64369404,0.00029877378,0.00022199086,0.0018058654,0.1754346,0.0019674571,0.047155272,0.041334953,0.0004899677,0.08631198],"study_design_scores_gemma":[0.0000106975,0.00009277319,0.9430781,0.00003174147,0.000057199693,0.000794535,0.036919985,0.002390188,0.0053705554,0.007830524,0.0033658259,0.000057819336],"about_ca_topic_score_codex":0.009750016,"about_ca_topic_score_gemma":0.011478538,"teacher_disagreement_score":0.009750016,"about_ca_system_score_codex":0.0011735047,"about_ca_system_score_gemma":0.00043628033,"threshold_uncertainty_score":0.01938653},"labels":[],"label_agreement":null},{"id":"W2732829562","doi":"10.18653/v1/w17-5501","title":"Automatic Mapping of French Discourse Connectives to PDTB Discourse Relations","year":2017,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Lexicon; Exploit; Phrase; Computer science; Linguistics; Natural language processing; Recall; Artificial intelligence; Order (exchange); Philosophy","score_opus":0.02631852927834459,"score_gpt":0.34108689339226606,"score_spread":0.31476836411392145,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2732829562","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.18512014,0.005711392,0.6786552,0.0014776514,0.00053120597,0.0007844777,0.034699623,0.04714979,0.0458706],"genre_scores_gemma":[0.50771636,0.0013636969,0.44498897,0.0003018026,0.00018257218,0.00042315925,0.03644181,0.0025192115,0.006062508],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9982565,0.0006140898,0.00010690278,0.0006496611,0.00028825237,0.00008468989],"domain_scores_gemma":[0.99610573,0.0023312673,0.0003210792,0.0003501413,0.0008016075,0.00009014455],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012803858,0.0012192571,0.0007101978,0.006817588,0.0013028478,0.0026381814,0.00053353416,0.00061697216,0.009708765],"category_scores_gemma":[0.006554797,0.0006138594,0.0005660809,0.0025085923,0.00067764905,0.0027097592,0.0016113229,0.0010512999,0.0037250358],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005578442,0.00017583102,0.0106957005,0.0022991104,0.000113217335,0.0010461095,0.0042096768,0.00433085,0.09002827,0.04543497,0.03668004,0.80442846],"study_design_scores_gemma":[0.00029711687,0.00040570844,0.037326798,0.00087830645,0.00036733763,0.0031712544,0.0050903987,0.19008093,0.17987218,0.07202862,0.5101419,0.00033947724],"about_ca_topic_score_codex":0.011373118,"about_ca_topic_score_gemma":0.015917495,"teacher_disagreement_score":0.011373118,"about_ca_system_score_codex":0.0015297367,"about_ca_system_score_gemma":0.002080311,"threshold_uncertainty_score":0.032479048},"labels":[],"label_agreement":null},{"id":"W2734827374","doi":"","title":"Synthesizing B substitutions for EB 3 attribute definitions","year":2004,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Sherbrooke","funders":"","keywords":"Computer science; Psychology","score_opus":0.08015744295860115,"score_gpt":0.30720775025286784,"score_spread":0.22705030729426667,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2734827374","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.060846172,0.00012983492,0.9276231,0.0001929591,0.00009817207,0.00027005267,0.00037740738,0.0031817385,0.007280521],"genre_scores_gemma":[0.23961431,0.00019162724,0.75380594,0.0002685516,0.000019389081,0.00026431197,0.00070275995,0.001430968,0.0037021323],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9979842,0.0005710625,0.0002542712,0.00025868288,0.00072176685,0.00021000924],"domain_scores_gemma":[0.9967129,0.0014028577,0.00029745876,0.00069347816,0.0008078059,0.000085511645],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016942186,0.00071794016,0.0005167671,0.0011184344,0.00059316546,0.0013987755,0.00095799466,0.0010919721,0.0029131877],"category_scores_gemma":[0.006728,0.00078246853,0.0011318156,0.0008227763,0.0010773564,0.0018914514,0.0021079008,0.0013065534,0.0011056535],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00048245196,0.0003397281,0.0070281387,0.0008821303,0.00012362772,0.001998057,0.0020103813,0.058216006,0.13891971,0.5362056,0.006095348,0.24769886],"study_design_scores_gemma":[0.00022124882,0.00035073532,0.0012172568,0.00028917525,0.0002260668,0.0011960736,0.0009302876,0.28999725,0.3603088,0.2180077,0.12711336,0.0001420067],"about_ca_topic_score_codex":0.0022262214,"about_ca_topic_score_gemma":0.0030625882,"teacher_disagreement_score":0.0029131877,"about_ca_system_score_codex":0.0007336927,"about_ca_system_score_gemma":0.0017425565,"threshold_uncertainty_score":0.009745538},"labels":[],"label_agreement":null},{"id":"W2735378068","doi":"10.29173/cais398","title":"La modélisation de l'analyse documentaire: à la sémiotique, de la psychologie cognitive et de l'intelligence artificielle","year":2013,"lang":"fr","type":"article","venue":"Proceedings of the Annual Conference of CAIS / Actes du congrès annuel de l ACSI","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Humanities; Semiotics; Cognition; Psychology; Linguistics; Philosophy","score_opus":0.030972673487859155,"score_gpt":0.3397371955615567,"score_spread":0.30876452207369753,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2735378068","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011935241,0.0015167756,0.9621315,0.0029995078,0.000114017,0.00019358596,0.0002667418,0.00047325756,0.02036924],"genre_scores_gemma":[0.3192439,0.0031929868,0.66006935,0.00059679133,0.00024968974,0.0012367503,0.0008214102,0.0002752663,0.014313894],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99511504,0.0027337084,0.00030393244,0.000776568,0.0009457202,0.00012508348],"domain_scores_gemma":[0.989489,0.007737075,0.00049998687,0.001210588,0.000890782,0.00017244028],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0050738733,0.0009835374,0.00094188616,0.002797304,0.0011541152,0.011314933,0.0021164566,0.0025240628,0.0056375074],"category_scores_gemma":[0.014578878,0.0006701598,0.0018800899,0.002573054,0.008313809,0.008991953,0.0024299577,0.0025371066,0.0015757454],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000082964914,0.000055021486,0.0014443055,0.00074783305,0.00008134099,0.00030686354,0.0064371587,0.020726608,0.0038509872,0.8985718,0.0017051869,0.06598981],"study_design_scores_gemma":[0.000050864386,0.00009623082,0.0016391048,0.00043379987,0.000074456635,0.0005692319,0.0021106824,0.1574312,0.003666426,0.7649922,0.06883187,0.00010397215],"about_ca_topic_score_codex":0.00780501,"about_ca_topic_score_gemma":0.0038869109,"teacher_disagreement_score":0.011314933,"about_ca_system_score_codex":0.00286701,"about_ca_system_score_gemma":0.0037747195,"threshold_uncertainty_score":0.026833534},"labels":[],"label_agreement":null},{"id":"W2736098086","doi":"","title":"Community-based corpus-building: Three case studies","year":2017,"lang":"en","type":"article","venue":"The COCOON platform (University of Paris)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Lemmatisation; Linguistics; Corpus linguistics; Annotation; Focus (optics); Ethos; Narrative; Documentation; Artificial intelligence; Natural language processing","score_opus":0.07443160432197264,"score_gpt":0.3087848184024849,"score_spread":0.23435321408051224,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2736098086","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.82385594,0.0028782526,0.07362028,0.0102987625,0.00037893496,0.0055873594,0.0024314562,0.00071516656,0.08023395],"genre_scores_gemma":[0.87212235,0.002023081,0.09897958,0.0013982742,0.00010363979,0.0033614377,0.0028891186,0.0007448094,0.018377693],"study_design_codex":"qualitative","study_design_gemma":"observational","domain_scores_codex":[0.9797395,0.01448117,0.0007965866,0.0012337286,0.0023978401,0.0013512622],"domain_scores_gemma":[0.92300427,0.049539197,0.002266927,0.008576131,0.011128757,0.0054847626],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.024192903,0.0007943053,0.0006263213,0.0039000155,0.016087033,0.0054421388,0.005823713,0.004498527,0.0062777298],"category_scores_gemma":[0.062469304,0.0006885398,0.0007004514,0.006923052,0.0077238423,0.006229883,0.011259192,0.0030348385,0.0014776348],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00046000947,0.00149901,0.027066875,0.0016356945,0.000071561284,0.025057565,0.70131946,0.003471382,0.0035617035,0.051630862,0.030659579,0.1535664],"study_design_scores_gemma":[0.0001995967,0.0002826125,0.012561138,0.0014486419,0.00007245717,0.0080282865,0.57382673,0.00586474,0.0056490498,0.016703775,0.37517202,0.00019086353],"about_ca_topic_score_codex":0.071214505,"about_ca_topic_score_gemma":0.14083,"teacher_disagreement_score":0.071214505,"about_ca_system_score_codex":0.0075646336,"about_ca_system_score_gemma":0.009988043,"threshold_uncertainty_score":0.14159995},"labels":[],"label_agreement":null},{"id":"W2736986829","doi":"10.18653/v1/w17-4704","title":"Modeling Target-Side Inflection in Neural Machine Translation","year":2017,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Nutrasource","funders":"European Commission","keywords":"Inflection; Computer science; Lemma (botany); Machine translation; Generalization; Natural language processing; Vocabulary; Artificial intelligence; Word (group theory); Czech; Simple (philosophy); Translation (biology); Byte; German; Transfer-based machine translation; Example-based machine translation; Speech recognition; Linguistics; Mathematics; Programming language","score_opus":0.038414173087790465,"score_gpt":0.31892469438134835,"score_spread":0.2805105212935579,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2736986829","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11533447,0.0027411035,0.8690375,0.0013082075,0.00043121408,0.00007741085,0.0007841741,0.0029093106,0.0073765954],"genre_scores_gemma":[0.84507334,0.0011403614,0.14387752,0.00026927513,0.00029794045,0.00016115443,0.0016296724,0.0008388842,0.0067117848],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99918133,0.00043100351,0.00006260694,0.0001889023,0.0000860911,0.00004998286],"domain_scores_gemma":[0.9957847,0.0030807308,0.00017833021,0.0004864184,0.0004123611,0.000057493027],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020180438,0.000773057,0.00068884273,0.00078444235,0.00060818286,0.0015222643,0.0009474782,0.0013685034,0.0035232343],"category_scores_gemma":[0.012976591,0.0005410719,0.00064259826,0.0012896869,0.00056303554,0.0031413035,0.00090381835,0.0017538731,0.0022588496],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007244125,0.00014957922,0.0042515625,0.0006861772,0.00023924142,0.0004992441,0.00044715838,0.52795076,0.01771432,0.047074202,0.011310575,0.38895276],"study_design_scores_gemma":[0.0000202824,0.0000387409,0.0005177097,0.000023025903,0.000023154967,0.00007816777,0.000032763164,0.94883335,0.0045194263,0.04421968,0.0016788247,0.000014843366],"about_ca_topic_score_codex":0.004078183,"about_ca_topic_score_gemma":0.0067484537,"teacher_disagreement_score":0.004078183,"about_ca_system_score_codex":0.0006825142,"about_ca_system_score_gemma":0.00093040784,"threshold_uncertainty_score":0.011786401},"labels":[],"label_agreement":null},{"id":"W2737257934","doi":"","title":"Verba sonandi des cris des animaux: métaphorisation et traitement automatique des données","year":2017,"lang":"fr","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Linguistic Association","funders":"","keywords":"Computer science","score_opus":0.04295273310298369,"score_gpt":0.28564241949866204,"score_spread":0.24268968639567834,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2737257934","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.058810547,0.0028595298,0.8862033,0.0066097286,0.0011801469,0.00015362723,0.00048382167,0.0033969032,0.040302485],"genre_scores_gemma":[0.4941417,0.0017154317,0.44830927,0.0008954291,0.00028263906,0.0001567044,0.0006485728,0.001218813,0.0526314],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9983469,0.0007459942,0.00006032753,0.0003692513,0.00039512463,0.00008232328],"domain_scores_gemma":[0.99795854,0.0013193418,0.000091391616,0.00023329655,0.0003189224,0.00007846717],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020102647,0.00087573385,0.0006931303,0.0010253789,0.0012582439,0.003554412,0.00064507604,0.0018716475,0.01152339],"category_scores_gemma":[0.007953596,0.00051812147,0.00091891084,0.0007352322,0.0027803022,0.004175844,0.0018821904,0.002471827,0.0027207122],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009395299,0.00012771825,0.0023717808,0.0012897049,0.0001504517,0.002144625,0.023264669,0.0060747215,0.13228366,0.48223567,0.033680886,0.3154366],"study_design_scores_gemma":[0.00016052743,0.0004740611,0.005790311,0.00056187855,0.0001886953,0.0061053815,0.009512331,0.14706612,0.054892436,0.25119486,0.5237572,0.0002962092],"about_ca_topic_score_codex":0.0036371874,"about_ca_topic_score_gemma":0.00264364,"teacher_disagreement_score":0.01152339,"about_ca_system_score_codex":0.0010430715,"about_ca_system_score_gemma":0.0009250953,"threshold_uncertainty_score":0.038549602},"labels":[],"label_agreement":null},{"id":"W2738120697","doi":"","title":"Biblical Hebrew: A formal perspective on the left periphery","year":2017,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Specifier; Linguistics; Verb; Biblical Hebrew; Dependent clause; Polarity (international relations); Hebrew; Head (geology); Subject (documents); Hebrew Bible; Philosophy; Mathematics; Computer science; Biblical studies; Sentence; Noun phrase; Noun","score_opus":0.01971125101510716,"score_gpt":0.3074707632717693,"score_spread":0.28775951225666213,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2738120697","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03121503,0.00428857,0.07461502,0.007957688,0.0003112505,0.00004341933,0.0002350856,0.00016780091,0.8811661],"genre_scores_gemma":[0.9103748,0.0021218413,0.025203142,0.0024959852,0.0006111663,0.00016245442,0.00030280228,0.0001744112,0.058553297],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9988385,0.0006768647,0.00004424142,0.00017175543,0.00013635051,0.00013231486],"domain_scores_gemma":[0.9991886,0.00039291143,0.000110474204,0.00013834545,0.000118639815,0.000051108458],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001433367,0.00054779375,0.000400697,0.0025692326,0.0037515156,0.0058125122,0.0011814305,0.001771011,0.014604666],"category_scores_gemma":[0.0022546186,0.00040178697,0.00036577252,0.00241853,0.010620231,0.008233531,0.0027667943,0.0026262226,0.0021972863],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000003374205,0.0000024308958,0.000028972809,0.000010751577,7.930018e-7,0.000014643773,0.0004380387,0.00006313294,0.000065011256,0.9977112,0.00032301707,0.001338723],"study_design_scores_gemma":[0.000017675375,0.000012736353,0.00039817544,0.00011416386,0.000006352342,0.000060481056,0.0010283829,0.0011962773,0.00043189767,0.93102336,0.06569834,0.000012057479],"about_ca_topic_score_codex":0.011612731,"about_ca_topic_score_gemma":0.008197743,"teacher_disagreement_score":0.014604666,"about_ca_system_score_codex":0.005308956,"about_ca_system_score_gemma":0.0012516851,"threshold_uncertainty_score":0.04885745},"labels":[],"label_agreement":null},{"id":"W2738790619","doi":"10.7202/1040469ar","title":"Genre and Register in Comparable Corpora: An English/Spanish Contrastive Analysis","year":2017,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Junta de Castilla y León","keywords":"Representativeness heuristic; Computer science; Register (sociolinguistics); Linguistics; Natural language processing; Corpus linguistics; Selection (genetic algorithm); Artificial intelligence; Mathematics; Statistics","score_opus":0.0464162821350621,"score_gpt":0.30249124470619554,"score_spread":0.25607496257113344,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2738790619","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8987891,0.0044870246,0.034548137,0.0008432554,0.00025676127,0.00078245613,0.006089624,0.0003447566,0.05385889],"genre_scores_gemma":[0.96573424,0.0005628741,0.023949603,0.00011533438,0.00013666027,0.0006817249,0.0059015877,0.00031836433,0.002599578],"study_design_codex":"design_other","study_design_gemma":"qualitative","domain_scores_codex":[0.9929929,0.004033349,0.00073365506,0.0009487509,0.0010341015,0.00025726858],"domain_scores_gemma":[0.962141,0.028420446,0.0019654157,0.002293785,0.0047464645,0.00043284878],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009229497,0.00045306672,0.0005318416,0.010146083,0.0016074313,0.00360185,0.0006698179,0.0005462176,0.0045216293],"category_scores_gemma":[0.04449507,0.00027070436,0.00061192823,0.011062166,0.0017649604,0.0021342013,0.0023274932,0.0011275895,0.00067728874],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004792975,0.0009379698,0.19466285,0.0048081055,0.0009995393,0.0056495112,0.21781799,0.0016044881,0.04877174,0.06399195,0.025882931,0.43008006],"study_design_scores_gemma":[0.0005260454,0.0008437979,0.5930588,0.0014065532,0.0012757723,0.0032547647,0.09691714,0.0104912,0.014246698,0.01737645,0.26038647,0.00021620466],"about_ca_topic_score_codex":0.0057420763,"about_ca_topic_score_gemma":0.0059453347,"teacher_disagreement_score":0.010146083,"about_ca_system_score_codex":0.0015624893,"about_ca_system_score_gemma":0.00069342105,"threshold_uncertainty_score":0.04881084},"labels":[],"label_agreement":null},{"id":"W2739178775","doi":"10.3389/fpsyg.2017.01236","title":"Pronoun Interpretation in the Second Language: Effects of Computational Complexity","year":2017,"lang":"en","type":"article","venue":"Frontiers in Psychology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Social Sciences and Humanities Research Council of Canada; Fonds de Recherche du Québec-Société et Culture","keywords":"Pronoun; Antecedent (behavioral psychology); Subject pronoun; Personal pronoun; Psychology; Linguistics; Reflexive pronoun; Object pronoun; Interpretation (philosophy); Task (project management); Set (abstract data type); Computer science; Social psychology","score_opus":0.013052113024059236,"score_gpt":0.3326842147517208,"score_spread":0.31963210172766154,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2739178775","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9971029,0.0001498138,0.0006025035,0.00010057869,0.000006638966,0.000018132858,0.00007317396,0.000028191116,0.0019181526],"genre_scores_gemma":[0.99746966,0.00007208648,0.00126078,0.00005090574,0.000012907173,0.00003049511,0.00013121609,0.00007771301,0.0008941501],"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9977475,0.0007492441,0.00017341464,0.0005731857,0.00058508076,0.00017157944],"domain_scores_gemma":[0.86068034,0.12251614,0.00824217,0.004457152,0.0016264906,0.0024777048],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024067801,0.00070211856,0.00068690785,0.0006739393,0.000415831,0.0030389833,0.00064171385,0.00081924634,0.0059327255],"category_scores_gemma":[0.04392806,0.00066701724,0.0006343304,0.00047079075,0.0011917561,0.003091014,0.0020488736,0.0017220031,0.00049201277],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.024523439,0.004404182,0.37619162,0.0013301026,0.0010697229,0.0018580552,0.016295506,0.011618533,0.4566213,0.0046870518,0.0014235107,0.09997688],"study_design_scores_gemma":[0.00034797934,0.0023826766,0.92743325,0.00006746865,0.00041100016,0.000916744,0.001343681,0.0145115545,0.045566596,0.0056948024,0.0011855095,0.00013866342],"about_ca_topic_score_codex":0.002176896,"about_ca_topic_score_gemma":0.0017430675,"teacher_disagreement_score":0.0059327255,"about_ca_system_score_codex":0.0006967494,"about_ca_system_score_gemma":0.00046065374,"threshold_uncertainty_score":0.019846916},"labels":[],"label_agreement":null},{"id":"W2739491707","doi":"10.18653/v1/s17-1006","title":"Deep Learning Models For Multiword Expression Identification","year":2017,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of New Brunswick","funders":"Natural Sciences and Engineering Research Council of Canada; New Brunswick Innovation Foundation","keywords":"Computer science; Artificial intelligence; Convolutional neural network; Deep learning; Identification (biology); Security token; Component (thermodynamics); Feed forward; Feedforward neural network; Artificial neural network; Recurrent neural network; Expression (computer science); Pattern recognition (psychology); Machine learning","score_opus":0.03916598333718744,"score_gpt":0.32032577958983305,"score_spread":0.2811597962526456,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2739491707","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02553094,0.0021581014,0.966157,0.0005048379,0.00015664754,0.000050077208,0.0007587933,0.0025746496,0.0021089003],"genre_scores_gemma":[0.5870853,0.002658372,0.38736045,0.00048720927,0.00016699701,0.00024229659,0.004470502,0.00038964217,0.017139213],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99949074,0.000119438686,0.000037755697,0.00016175564,0.00011915335,0.000071137634],"domain_scores_gemma":[0.9991762,0.00041385004,0.00008886228,0.000107253814,0.00018480315,0.000028892751],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00085378165,0.0010324888,0.0007618622,0.0007414766,0.00030698295,0.0010473732,0.0014032712,0.0011333574,0.0030290587],"category_scores_gemma":[0.0028443402,0.0005030871,0.0007113127,0.0009981085,0.00040991686,0.002542898,0.00091922295,0.0021724799,0.0019439037],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00031690026,0.00019694827,0.0025304134,0.00036744666,0.00024082809,0.0002562101,0.00015233735,0.40375078,0.024876159,0.025010418,0.009608062,0.53269356],"study_design_scores_gemma":[0.00000424547,0.000017866494,0.00023979496,0.000012951626,0.0000148487015,0.000022766088,0.000011506294,0.9854635,0.0028094368,0.009880833,0.0015133634,0.000008853941],"about_ca_topic_score_codex":0.0060312147,"about_ca_topic_score_gemma":0.008607999,"teacher_disagreement_score":0.0060312147,"about_ca_system_score_codex":0.0010660406,"about_ca_system_score_gemma":0.0008487101,"threshold_uncertainty_score":0.011992216},"labels":[],"label_agreement":null},{"id":"W2739740656","doi":"10.18653/v1/w17-2509","title":"BUCC 2017 Shared Task: a First Attempt Toward a Deep Learning Framework for Identifying Parallel Sentences in Comparable Corpora","year":2017,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Leverage (statistics); Preprocessor; Natural language processing; Parallel corpora; Task (project management); Artificial intelligence; Semantics (computer science); Distributional semantics; Programming language; Semantic similarity; Machine translation","score_opus":0.08433113270639955,"score_gpt":0.35244922604400714,"score_spread":0.2681180933376076,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2739740656","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09855772,0.009213673,0.66867137,0.006580901,0.006760496,0.006885438,0.09824838,0.074938014,0.030144118],"genre_scores_gemma":[0.12700681,0.0010866439,0.5489208,0.0018972924,0.0016532326,0.0053263637,0.2935874,0.005376274,0.015145261],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.981821,0.0073624244,0.00160064,0.004755681,0.0033073737,0.0011528317],"domain_scores_gemma":[0.9615038,0.012236767,0.0015114355,0.013836162,0.008601356,0.002310585],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01841049,0.005215203,0.004155893,0.008196627,0.005782546,0.00585564,0.008147839,0.006269599,0.018102933],"category_scores_gemma":[0.053516205,0.0016788298,0.0030319863,0.0051075662,0.0027660287,0.011288723,0.0142611265,0.01043128,0.013451567],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015159616,0.002007039,0.0038084802,0.003422798,0.0011839808,0.0015329152,0.0017377594,0.008692129,0.03080738,0.017371874,0.4189496,0.50897],"study_design_scores_gemma":[0.002113886,0.002770012,0.012255282,0.0010336032,0.0014564324,0.0033473608,0.0038670604,0.3416841,0.06422002,0.08905054,0.47716096,0.001040835],"about_ca_topic_score_codex":0.013771705,"about_ca_topic_score_gemma":0.01728555,"teacher_disagreement_score":0.01841049,"about_ca_system_score_codex":0.0032409008,"about_ca_system_score_gemma":0.009478502,"threshold_uncertainty_score":0.09736514},"labels":[],"label_agreement":null},{"id":"W2739752763","doi":"10.18653/v1/e17-4006","title":"A Computational Model of Human Preferences for Pronoun Resolution","year":2017,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Atomic Energy of Canada Limited; Agence Nationale de la Recherche","keywords":"Pronoun; Antecedent (behavioral psychology); Subject pronoun; Computer science; Resolution (logic); Natural language processing; Probabilistic logic; Artificial intelligence; Computational linguistics; Cognitive model; Reading (process); Computational model; Cognition; Linguistics; Psychology; Social psychology","score_opus":0.06613439630745732,"score_gpt":0.3454038756156842,"score_spread":0.2792694793082269,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2739752763","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.18454629,0.00023298946,0.797433,0.0014419353,0.000036687383,0.0000824085,0.0003375592,0.00046746957,0.015421539],"genre_scores_gemma":[0.8604711,0.0001687267,0.13520499,0.0001738854,0.000026731192,0.00015864927,0.00025139944,0.000074542375,0.0034699177],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99948275,0.00020631735,0.000019935505,0.0001387308,0.000103542225,0.000048750528],"domain_scores_gemma":[0.99744296,0.0018807912,0.00014875381,0.00025421573,0.00017927091,0.000094024224],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009153024,0.00034350314,0.00046827763,0.0005883115,0.00046871637,0.0015409815,0.0014067757,0.0009339596,0.0054534967],"category_scores_gemma":[0.006758081,0.00045784502,0.0008075171,0.00075534766,0.0009010337,0.0023581928,0.00054926344,0.0009762569,0.0007028131],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018960719,0.00023905143,0.0062682033,0.00018865267,0.0001631716,0.0004553696,0.0013729371,0.63016224,0.008334026,0.2844357,0.003199152,0.064991996],"study_design_scores_gemma":[0.000016510017,0.000027315828,0.00093332585,0.0000056550098,0.00001526442,0.00010488607,0.00006239253,0.8976989,0.0004893047,0.09979793,0.0008311882,0.000017308126],"about_ca_topic_score_codex":0.004926076,"about_ca_topic_score_gemma":0.005600588,"teacher_disagreement_score":0.0054534967,"about_ca_system_score_codex":0.00092696,"about_ca_system_score_gemma":0.0009776838,"threshold_uncertainty_score":0.01824379},"labels":[],"label_agreement":null},{"id":"W2739833967","doi":"10.18653/v1/e17-2038","title":"A Parallel Corpus for Evaluating Machine Translation between Arabic and European Languages","year":2017,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"York University; New York University Abu Dhabi","keywords":"Arabic; Computer science; Machine translation; Natural language processing; Computational linguistics; Linguistics; Artificial intelligence; Translation (biology); Translation studies; Philosophy","score_opus":0.06611473451122685,"score_gpt":0.37098334064760036,"score_spread":0.3048686061363735,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2739833967","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6878119,0.008541274,0.11495277,0.0018420101,0.003080848,0.0042187762,0.0801379,0.01552143,0.08389312],"genre_scores_gemma":[0.5665702,0.0017088073,0.18188542,0.00033545206,0.00038342504,0.00491045,0.22915462,0.0015773978,0.013474205],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9953555,0.0022291776,0.0006688744,0.0006443468,0.0009057318,0.00019644704],"domain_scores_gemma":[0.99204165,0.0028937464,0.00026809497,0.001471216,0.0029847736,0.00034053216],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004743995,0.0016772065,0.0013998495,0.005646795,0.0027726856,0.0018569025,0.0018128944,0.0018933277,0.0133118145],"category_scores_gemma":[0.013854284,0.0007568471,0.0007019299,0.0052775103,0.00085908646,0.0037082832,0.0047577983,0.0015591467,0.0091342265],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0053865425,0.002875528,0.011555507,0.003922966,0.00067399174,0.0024300616,0.0023925123,0.017313981,0.033481706,0.0123038655,0.19775291,0.7099105],"study_design_scores_gemma":[0.0047444766,0.004760033,0.07771722,0.0015941779,0.0014956768,0.0074625043,0.009708818,0.28953704,0.1257037,0.025409324,0.45129606,0.00057095883],"about_ca_topic_score_codex":0.0036459074,"about_ca_topic_score_gemma":0.0051255594,"teacher_disagreement_score":0.0133118145,"about_ca_system_score_codex":0.0008214469,"about_ca_system_score_gemma":0.0018122827,"threshold_uncertainty_score":0.04453242},"labels":[],"label_agreement":null},{"id":"W2740061553","doi":"10.24963/ijcai.2017/668","title":"Concerning Referring Expressions in Query Answers","year":2017,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Expression (computer science); Computer science; Noun phrase; Context (archaeology); Phrase; Constant (computer programming); Natural language processing; Base (topology); Artificial intelligence; Linguistics; Information retrieval; Noun; Mathematics; Programming language","score_opus":0.03956770085474896,"score_gpt":0.32841322118007216,"score_spread":0.2888455203253232,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2740061553","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03135301,0.0053002136,0.92964107,0.0080159595,0.00039424747,0.00022346165,0.0004534977,0.00059102534,0.024027526],"genre_scores_gemma":[0.6035109,0.009315993,0.3582474,0.0046830196,0.0027434088,0.000699014,0.0023556696,0.001292477,0.017152019],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9669877,0.01617757,0.0031081082,0.0047388603,0.0068600755,0.002127661],"domain_scores_gemma":[0.9365989,0.04903681,0.0028749204,0.004962495,0.0059322757,0.00059455965],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01894177,0.0012505923,0.0022409342,0.0050107143,0.005857295,0.011828233,0.0035823411,0.006924629,0.0061426],"category_scores_gemma":[0.08487713,0.0019427143,0.002613033,0.0077154194,0.011287542,0.036848076,0.006960093,0.0057026753,0.0016213044],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009030727,0.000016860704,0.0005501466,0.00021608309,0.000032293847,0.0006657092,0.0035957072,0.0023183008,0.0012604827,0.9728055,0.0024062165,0.016042398],"study_design_scores_gemma":[0.000028666673,0.000041381896,0.0002814127,0.00013779175,0.000085003565,0.00082765025,0.0011829638,0.017878678,0.0034877975,0.95057786,0.025407903,0.00006288677],"about_ca_topic_score_codex":0.0055657825,"about_ca_topic_score_gemma":0.0024911405,"teacher_disagreement_score":0.01894177,"about_ca_system_score_codex":0.0038299959,"about_ca_system_score_gemma":0.0020955577,"threshold_uncertainty_score":0.100174904},"labels":[],"label_agreement":null},{"id":"W2740072742","doi":"10.18653/v1/w17-1906","title":"Supervised and unsupervised approaches to measuring usage similarity","year":2017,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of New Brunswick","funders":"Natural Sciences and Engineering Research Council of Canada; New Brunswick Innovation Foundation","keywords":"Similarity (geometry); Computer science; Artificial intelligence; Context (archaeology); Lemma (botany); Word (group theory); Pattern recognition (psychology); Scale (ratio); Machine learning; Unsupervised learning; Meaning (existential); Natural language processing; Mathematics; Image (mathematics)","score_opus":0.1696989937941329,"score_gpt":0.2777366257560457,"score_spread":0.1080376319619128,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2740072742","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.31277657,0.0036059336,0.65355927,0.00031488214,0.00021173155,0.00096205226,0.007256891,0.0072864154,0.014026241],"genre_scores_gemma":[0.7401325,0.00064603746,0.24150515,0.00018329987,0.000219808,0.0010690517,0.012491902,0.00050094933,0.0032513398],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9917058,0.0026355623,0.0010693999,0.0022020182,0.0020476598,0.0003395018],"domain_scores_gemma":[0.9882941,0.004256408,0.0020802522,0.0024204683,0.0026446995,0.00030402138],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024327892,0.001297138,0.0013499485,0.008702244,0.0007932444,0.0015258169,0.0016842162,0.001169845,0.0014862841],"category_scores_gemma":[0.013091654,0.0003431166,0.0012206958,0.006372459,0.001143589,0.003880313,0.0022845056,0.0010670268,0.0013370902],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00073311414,0.0013126531,0.07865272,0.0015559751,0.0009472737,0.000333999,0.0013860362,0.018384425,0.035752494,0.013660404,0.015030789,0.8322502],"study_design_scores_gemma":[0.00013133441,0.0012612741,0.12860833,0.0002824581,0.00048164316,0.002318899,0.0026494341,0.70179164,0.062233634,0.06290352,0.036967922,0.00036989522],"about_ca_topic_score_codex":0.0016114279,"about_ca_topic_score_gemma":0.004410913,"teacher_disagreement_score":0.008702244,"about_ca_system_score_codex":0.0007208154,"about_ca_system_score_gemma":0.00097446615,"threshold_uncertainty_score":0.01286602},"labels":[],"label_agreement":null},{"id":"W2740213693","doi":"10.18653/v1/k17-2008","title":"If you can't beat them, join them: the University of Alberta system description","year":2017,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Innovates","keywords":"Computer science; Discriminative model; Join (topology); Task (project management); Artificial intelligence; Context (archaeology); Natural language processing; Speech recognition","score_opus":0.02441929272480663,"score_gpt":0.222698383479858,"score_spread":0.19827909075505135,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2740213693","genre_codex":"software","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022325074,0.0035815209,0.2666466,0.004675951,0.0009925929,0.0029071928,0.19056231,0.33762327,0.17068547],"genre_scores_gemma":[0.19412205,0.0025862504,0.24598372,0.0024042665,0.00019683347,0.0019666255,0.3131204,0.022376977,0.21724288],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99912494,0.00008816051,0.00004728947,0.00024333263,0.00031293076,0.00018336119],"domain_scores_gemma":[0.9992446,0.00009892043,0.000028654014,0.00016024506,0.00028367853,0.00018398171],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012228884,0.0013763799,0.00090113806,0.0015377267,0.001353312,0.0037045255,0.003103319,0.0012482413,0.07149852],"category_scores_gemma":[0.0023624664,0.00078313873,0.0005234403,0.0018109583,0.00080650125,0.0016718586,0.0016082695,0.0016186013,0.058787003],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":true,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00078545674,0.00012139889,0.003129435,0.0010138561,0.00006091438,0.0006913641,0.0005217082,0.017264405,0.017957453,0.015721744,0.7753224,0.16740982],"study_design_scores_gemma":[0.00013375908,0.0001013655,0.0037872428,0.00014277142,0.00006381974,0.0005542209,0.00022595462,0.05509845,0.016734099,0.0071822032,0.9157402,0.00023587835],"about_ca_topic_score_codex":0.43873665,"about_ca_topic_score_gemma":0.4733698,"teacher_disagreement_score":0.99418324,"about_ca_system_score_codex":0.005816777,"about_ca_system_score_gemma":0.00910286,"threshold_uncertainty_score":0.87236583},"labels":[],"label_agreement":null},{"id":"W2740290989","doi":"10.18653/v1/e17-2098","title":"Bootstrapping Unsupervised Bilingual Lexicon Induction","year":2017,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Innovates","keywords":"Bootstrapping (finance); Computer science; Lexicon; Natural language processing; Artificial intelligence; Context (archaeology); Similarity (geometry); Task (project management); Basis (linear algebra); Speech recognition; Mathematics","score_opus":0.037906062422588434,"score_gpt":0.3149337540185355,"score_spread":0.2770276915959471,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2740290989","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.043096494,0.00029240796,0.9421199,0.00016262403,0.00006993418,0.00022577247,0.0007527307,0.009261082,0.0040191435],"genre_scores_gemma":[0.39631918,0.00023904209,0.58430654,0.00032564317,0.00011839751,0.0008471315,0.011239592,0.0013614871,0.005243069],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99851376,0.0005779612,0.00007955038,0.0004133101,0.00029091025,0.00012456733],"domain_scores_gemma":[0.99687,0.0014468671,0.00018076078,0.0008364091,0.00058689504,0.00007910474],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001194388,0.0010218613,0.0012095926,0.0018420201,0.0010475527,0.0008614632,0.0017366847,0.0008394243,0.0039348914],"category_scores_gemma":[0.0065485,0.0005776899,0.000999675,0.0019333527,0.0009188976,0.0021793232,0.0025416198,0.0012975933,0.0037578638],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003201571,0.00030102977,0.002776724,0.00035268706,0.00016687567,0.000455386,0.00042692295,0.038835596,0.04090313,0.014333292,0.01844779,0.8826803],"study_design_scores_gemma":[0.00014488283,0.00016079519,0.0024892373,0.000043603184,0.00008276435,0.0006259712,0.00023761223,0.8714167,0.047070473,0.06352021,0.014145755,0.000062076666],"about_ca_topic_score_codex":0.0019332469,"about_ca_topic_score_gemma":0.004939321,"teacher_disagreement_score":0.0039348914,"about_ca_system_score_codex":0.0005615202,"about_ca_system_score_gemma":0.0015069548,"threshold_uncertainty_score":0.013163507},"labels":[],"label_agreement":null},{"id":"W2740718109","doi":"10.18653/v1/w17-3205","title":"Cost Weighting for Neural Machine Translation Domain Adaptation","year":2017,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":87,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Weighting; Computer science; Machine translation; Classifier (UML); Domain adaptation; Artificial intelligence; Machine learning; Pattern recognition (psychology); Data mining","score_opus":0.06137940363727862,"score_gpt":0.3309722368791409,"score_spread":0.2695928332418623,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2740718109","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018814512,0.0005345897,0.97771674,0.0001581755,0.00008878741,0.00007616854,0.000041276042,0.00045379347,0.0021159642],"genre_scores_gemma":[0.4786248,0.0007278362,0.5138705,0.00031607348,0.00019001812,0.00043915532,0.0005494992,0.00040326724,0.004878915],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99892455,0.00046919615,0.00007255017,0.00016029063,0.00032186374,0.000051563315],"domain_scores_gemma":[0.99791473,0.00094800966,0.00011985805,0.00040172564,0.00056711264,0.00004848646],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017377489,0.00074420375,0.0006663758,0.0008450848,0.0003797053,0.0006107771,0.0010877044,0.000678451,0.0022173822],"category_scores_gemma":[0.0074229003,0.00024027764,0.0004475434,0.0010725617,0.00048258284,0.001972686,0.001281828,0.0011568994,0.0007051021],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024839808,0.00020916818,0.0012449212,0.00022257959,0.000109704466,0.00013011262,0.00014568519,0.21484841,0.020549083,0.030196853,0.0047029504,0.7273921],"study_design_scores_gemma":[0.00002426664,0.000072229355,0.00062819076,0.000014709144,0.000033083554,0.000110184104,0.000034485412,0.9650603,0.0075636073,0.021058403,0.005380058,0.000020549196],"about_ca_topic_score_codex":0.0014498947,"about_ca_topic_score_gemma":0.001714826,"teacher_disagreement_score":0.0022173822,"about_ca_system_score_codex":0.00069625553,"about_ca_system_score_gemma":0.00059775,"threshold_uncertainty_score":0.009190202},"labels":[],"label_agreement":null},{"id":"W2740947917","doi":"10.18653/v1/p17-2039","title":"Argumentation Quality Assessment: Theory vs. Practice","year":2017,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":65,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"World Wildlife Fund Canada","funders":"Natural Sciences and Engineering Research Council of Canada; Volkswagen Foundation; Deutsche Forschungsgemeinschaft","keywords":"Argumentation theory; Quality (philosophy); Quality assessment; Computer science; Association (psychology); Epistemology; Linguistics; Library science; Natural language processing; Philosophy; Engineering; Evaluation methods; Reliability engineering","score_opus":0.036179491453623036,"score_gpt":0.4224867882984318,"score_spread":0.38630729684480875,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2740947917","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.045839176,0.20929137,0.47324604,0.1934723,0.0024702705,0.0009161968,0.0005145727,0.0009009591,0.073349185],"genre_scores_gemma":[0.7223078,0.03371771,0.23364292,0.0043155095,0.0018022379,0.0007616593,0.0003053106,0.00034613712,0.0028006444],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.85845923,0.09057289,0.008612566,0.008187303,0.03305217,0.001115891],"domain_scores_gemma":[0.5560224,0.3540245,0.016480062,0.028991327,0.039684456,0.0047972067],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.14431618,0.0014577028,0.0026526558,0.010837192,0.001807461,0.0220854,0.0041453145,0.005776028,0.0059039006],"category_scores_gemma":[0.29994804,0.0011458631,0.00090714166,0.0078090397,0.01354567,0.01763887,0.009101149,0.0055762227,0.0021297308],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00040871138,0.00024453452,0.008852075,0.005251495,0.00039031514,0.00011494597,0.004744304,0.0031749755,0.0006060301,0.4168961,0.01698979,0.5423268],"study_design_scores_gemma":[0.00025200308,0.00029784348,0.005167879,0.012193497,0.00023945156,0.00044735952,0.0058442415,0.024924813,0.0017659555,0.8981782,0.050567627,0.000121133315],"about_ca_topic_score_codex":0.0016363685,"about_ca_topic_score_gemma":0.001842918,"teacher_disagreement_score":0.14431618,"about_ca_system_score_codex":0.007068208,"about_ca_system_score_gemma":0.0069766114,"threshold_uncertainty_score":0.76322603},"labels":[],"label_agreement":null},{"id":"W2741774630","doi":"10.18653/v1/e17-2034","title":"Morphological Analysis without Expert Annotation","year":2017,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Innovates","keywords":"Computer science; Annotation; Inflection; Discriminative model; Natural language processing; Artificial intelligence; String (physics); Word (group theory); Task (project management); Lemma (botany); Substring; Information retrieval; Mathematics; Programming language; Data structure","score_opus":0.028216850783038,"score_gpt":0.341827736828401,"score_spread":0.313610886045363,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2741774630","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025950436,0.00024680537,0.8891107,0.00024119485,0.00017007894,0.00016689465,0.004369249,0.06586202,0.0138826445],"genre_scores_gemma":[0.17060223,0.00029893086,0.78994,0.0005448345,0.00010576996,0.00015487615,0.011675106,0.0069193295,0.01975903],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9980597,0.00024626893,0.00026971876,0.00073848234,0.00056087464,0.000124909],"domain_scores_gemma":[0.9933401,0.0019280156,0.00039203145,0.0029302554,0.001303463,0.000106029685],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013062578,0.0013168581,0.000971634,0.003269352,0.000882902,0.0022563096,0.0016167437,0.0012507639,0.028134676],"category_scores_gemma":[0.0063787936,0.0007705402,0.00092500274,0.002417443,0.00070844765,0.0035774151,0.002685457,0.0011565054,0.03085912],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00041090616,0.00011742831,0.0044006445,0.00089024584,0.00010394319,0.0010783825,0.00070069346,0.0018116521,0.22114378,0.012834573,0.031042943,0.7254648],"study_design_scores_gemma":[0.00007711651,0.00030645225,0.01545796,0.00024956718,0.0002549792,0.010182123,0.0008619487,0.105888866,0.4052246,0.060934626,0.40028647,0.00027532518],"about_ca_topic_score_codex":0.0005530548,"about_ca_topic_score_gemma":0.0011964628,"teacher_disagreement_score":0.028134676,"about_ca_system_score_codex":0.0003758592,"about_ca_system_score_gemma":0.0010165229,"threshold_uncertainty_score":0.09411985},"labels":[],"label_agreement":null},{"id":"W2742019041","doi":"10.18653/v1/w17-0101","title":"A Morphological Parser for Odawa","year":2017,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Alberta","funders":"Social Sciences and Humanities Research Council of Canada; University of Alberta","keywords":"Parsing; Computer science; Natural language processing; Finite state; Linguistics; Artificial intelligence; Simple (philosophy); Philosophy","score_opus":0.04268798043559041,"score_gpt":0.3338137133694118,"score_spread":0.2911257329338214,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2742019041","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.117801026,0.0010897088,0.6557543,0.003285067,0.00085602317,0.0009291867,0.033276264,0.049724672,0.13728376],"genre_scores_gemma":[0.41227144,0.0008656274,0.5260348,0.00070057635,0.00008279129,0.00034019724,0.015047285,0.0049586426,0.03969868],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9995697,0.0000424128,0.00006432139,0.00015723411,0.00010538898,0.000060903258],"domain_scores_gemma":[0.9989178,0.0003034237,0.00008258548,0.00017423724,0.00047217286,0.000049780847],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008715758,0.00055521424,0.0005362112,0.0017355871,0.0025667066,0.0030111636,0.00090567215,0.0007719782,0.012750841],"category_scores_gemma":[0.0021266767,0.0010160315,0.0007012489,0.001638452,0.0011738613,0.002555504,0.0023176433,0.0014486933,0.0037291788],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005255866,0.00018437249,0.014604266,0.001437028,0.000106996624,0.0025330596,0.011145556,0.005097971,0.112524174,0.26169044,0.09652213,0.49362838],"study_design_scores_gemma":[0.00008847215,0.00008068551,0.017001716,0.00039950208,0.00024082753,0.0020314546,0.0049980595,0.025089134,0.059400693,0.069288254,0.821065,0.0003161832],"about_ca_topic_score_codex":0.064640395,"about_ca_topic_score_gemma":0.17588942,"teacher_disagreement_score":0.064640395,"about_ca_system_score_codex":0.0025072324,"about_ca_system_score_gemma":0.008084926,"threshold_uncertainty_score":0.1285283},"labels":[],"label_agreement":null},{"id":"W2742098346","doi":"10.18653/v1/e17-3002","title":"Common Round: Application of Language Technologies to Large-Scale Web Debates","year":2017,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Banting and Best Diabetes Centre, University of Toronto; Bundesministerium für Bildung und Forschung; European Commission","keywords":"Computer science; Art history; Scale (ratio); Artificial intelligence; Library science; Natural language processing; Linguistics; Philosophy; Art; Cartography","score_opus":0.008471289718478846,"score_gpt":0.2940793592581881,"score_spread":0.2856080695397093,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2742098346","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.18922411,0.0056302114,0.56626016,0.023273777,0.002147869,0.0011392527,0.019429328,0.03684466,0.15605068],"genre_scores_gemma":[0.725264,0.0015932656,0.23580432,0.0012578954,0.00068559585,0.00072082004,0.012812484,0.0033936026,0.018467968],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9954774,0.0029288474,0.00020120355,0.00060188305,0.0006020759,0.00018860775],"domain_scores_gemma":[0.9750182,0.018674416,0.0009709987,0.0034810225,0.0011177011,0.00073770207],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0049583926,0.00074568536,0.00059406506,0.0063804467,0.0023637817,0.0072316825,0.0018858031,0.0017173028,0.024293914],"category_scores_gemma":[0.034625627,0.0005870389,0.0010747813,0.005852024,0.0023125252,0.02017218,0.0067506237,0.002104929,0.003939301],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009942529,0.0004682975,0.013525717,0.0010011215,0.00028583815,0.001869273,0.024062337,0.008312446,0.0042192102,0.44730273,0.13438979,0.36356896],"study_design_scores_gemma":[0.0002236138,0.00014359734,0.0065750177,0.00033476687,0.00011681586,0.00082307257,0.017577244,0.12003231,0.0065025263,0.53044605,0.31706294,0.0001620708],"about_ca_topic_score_codex":0.0053731115,"about_ca_topic_score_gemma":0.0084580025,"teacher_disagreement_score":0.024293914,"about_ca_system_score_codex":0.0016268457,"about_ca_system_score_gemma":0.0012813592,"threshold_uncertainty_score":0.08127129},"labels":[],"label_agreement":null},{"id":"W2742155240","doi":"10.18653/v1/w17-2512","title":"Overview of the Second BUCC Shared Task: Spotting Parallel Sentences in Comparable Corpora","year":2017,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":61,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Nautical Research Society","funders":"European Commission","keywords":"Sentence; Computer science; German; Task (project management); Natural language processing; Artificial intelligence; Parallel corpora; Gold standard (test); Spotting; Sample (material); Speech recognition; Linguistics; Statistics; Mathematics; Machine translation","score_opus":0.055964154626907146,"score_gpt":0.31497795324294187,"score_spread":0.25901379861603474,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2742155240","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06616517,0.011261624,0.4810588,0.004064674,0.0040429006,0.014980063,0.17847623,0.20178223,0.038168296],"genre_scores_gemma":[0.04043092,0.0009584395,0.37763625,0.0010143287,0.00077187404,0.011147667,0.5458996,0.011391377,0.010749508],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9759168,0.0074463887,0.002512222,0.006645875,0.005772847,0.001705902],"domain_scores_gemma":[0.95881057,0.007658236,0.0011192831,0.01608398,0.013703789,0.0026240645],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.020643476,0.0065516024,0.005794964,0.013744971,0.0057747927,0.0070756315,0.008214633,0.0052785715,0.022375017],"category_scores_gemma":[0.042170487,0.002505239,0.0042381184,0.008373969,0.002308825,0.008519266,0.012793482,0.007732494,0.029658815],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012673446,0.0013587445,0.0043162694,0.0042304485,0.0009860911,0.0007559671,0.0012948548,0.0074032727,0.04239356,0.004568111,0.5255759,0.40584952],"study_design_scores_gemma":[0.001654974,0.002456453,0.02873482,0.0011187199,0.00088234723,0.003966978,0.002206533,0.112519294,0.0888447,0.016295023,0.7400359,0.001284249],"about_ca_topic_score_codex":0.022694325,"about_ca_topic_score_gemma":0.027227154,"teacher_disagreement_score":0.022694325,"about_ca_system_score_codex":0.00400853,"about_ca_system_score_gemma":0.011694949,"threshold_uncertainty_score":0.10917443},"labels":[],"label_agreement":null},{"id":"W2743287021","doi":"","title":"Proceedings of the Seventh Workshop on Building Educational Applications Using NLP, BEA@NAACL-HLT 2012, June 7, 2012, Montréal, Canada","year":2012,"lang":"en","type":"article","venue":"The Association for Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Artificial intelligence; Natural language processing; Computer science; History","score_opus":0.013421041840080503,"score_gpt":0.2803863278671192,"score_spread":0.2669652860270387,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2743287021","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009529459,0.004685534,0.73777956,0.013830015,0.004068796,0.0014059466,0.02175389,0.08189101,0.1250558],"genre_scores_gemma":[0.058398962,0.0046030465,0.6604826,0.002557015,0.00071876484,0.0010759436,0.07883336,0.011604311,0.181726],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99595046,0.0018208608,0.000340781,0.00072681846,0.0008645353,0.00029658168],"domain_scores_gemma":[0.9940441,0.00210225,0.00008663825,0.0013080331,0.0018339105,0.0006251442],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006808115,0.0017278848,0.0017102853,0.0022412764,0.0026023658,0.010298614,0.0034750274,0.0022529005,0.0913913],"category_scores_gemma":[0.0136967935,0.001400691,0.0016402785,0.0022308836,0.0014154724,0.008754619,0.0076211235,0.003840907,0.0405173],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036073432,0.0005298079,0.0014890889,0.00078914606,0.00013151036,0.00037438472,0.0018608243,0.0026965009,0.0072533577,0.018575031,0.4874198,0.47851983],"study_design_scores_gemma":[0.00010510963,0.00006590125,0.0015033196,0.00028806867,0.00012655955,0.00018152832,0.0011070718,0.026668759,0.009085535,0.02319689,0.937595,0.00007635441],"about_ca_topic_score_codex":0.07227346,"about_ca_topic_score_gemma":0.11832014,"teacher_disagreement_score":0.0913913,"about_ca_system_score_codex":0.0028207465,"about_ca_system_score_gemma":0.007352564,"threshold_uncertainty_score":0.30573434},"labels":[],"label_agreement":null},{"id":"W2748411071","doi":"10.1017/9781316416013.012","title":"Appendix: <i>Corpora and Text Collections</i>","year":2017,"lang":"en","type":"other","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Appendix; Computer science; Information retrieval; Natural language processing; Biology; Paleontology","score_opus":0.010294229641413032,"score_gpt":0.25676475330461956,"score_spread":0.24647052366320654,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2748411071","genre_codex":"dataset","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00020866901,0.0001863638,0.0028527617,0.00041972505,0.00056839595,0.00028856637,0.93237525,0.005530853,0.057569485],"genre_scores_gemma":[0.0011308481,0.00032515402,0.008548921,0.00023789739,0.0002578498,0.00087223895,0.9433371,0.0048110504,0.040479083],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9983486,0.00027604745,0.00028529565,0.00030618234,0.0006600288,0.0001238596],"domain_scores_gemma":[0.98059344,0.0058292835,0.0009062854,0.0033215415,0.008255041,0.0010943435],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0018317706,0.0017398206,0.0015534682,0.010567767,0.0012650725,0.0030042843,0.0028504287,0.0012062979,0.72571903],"category_scores_gemma":[0.017316494,0.0010776541,0.0008155626,0.017609378,0.000627496,0.0032760592,0.0025751132,0.0016885751,0.628998],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00001915437,0.000012344909,0.000035134082,0.00021121967,0.0000023643322,0.00000957984,0.00000995347,0.000040109884,0.00008993939,0.00046040755,0.99227506,0.0068348264],"study_design_scores_gemma":[0.00008292001,0.000014963605,0.0006571199,0.00017220287,0.000010063974,0.000071641625,0.000042993892,0.0002366839,0.00069054397,0.0020982756,0.99590236,0.000020176407],"about_ca_topic_score_codex":0.009120999,"about_ca_topic_score_gemma":0.010448897,"teacher_disagreement_score":0.72571903,"about_ca_system_score_codex":0.0012884309,"about_ca_system_score_gemma":0.0035027363,"threshold_uncertainty_score":0.3912285},"labels":[],"label_agreement":null},{"id":"W2752124967","doi":"10.7939/r31d1p","title":"Large-scale semi-supervised learning for natural language processing","year":2010,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Artificial intelligence; Computer science; Machine learning; Semi-supervised learning; Supervised learning; Natural language processing; Bottleneck; Exploit; Test set; Set (abstract data type); Context (archaeology); Labeled data; Artificial neural network","score_opus":0.006815353062016282,"score_gpt":0.2714176253257657,"score_spread":0.2646022722637494,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2752124967","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0010280097,0.0005423897,0.9958807,0.0002837868,0.000056124103,0.00012253044,0.00015909072,0.0013613632,0.0005660916],"genre_scores_gemma":[0.07441909,0.000948604,0.9187171,0.00037006428,0.00035079778,0.0013907003,0.0018679302,0.0004538916,0.0014818242],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.988867,0.005616839,0.0006688657,0.001944025,0.0027050104,0.00019830908],"domain_scores_gemma":[0.9643222,0.025310185,0.0017281759,0.004878123,0.0033577722,0.0004036278],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010879029,0.0022283525,0.0026238945,0.0024857598,0.0016504069,0.002931815,0.004877859,0.0024224783,0.0038306105],"category_scores_gemma":[0.03027497,0.0014013662,0.0023841066,0.0030659167,0.0034758104,0.0044548926,0.004366711,0.0058977967,0.002574616],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002159532,0.00036828924,0.001382889,0.0013266527,0.00043095168,0.0003027521,0.00042988115,0.4831835,0.0033571753,0.09568637,0.024213364,0.38910228],"study_design_scores_gemma":[0.000014172739,0.000023233039,0.0001171747,0.000031331852,0.000009520927,0.000025425215,0.00002048432,0.922073,0.00069863297,0.07446082,0.0025094089,0.000016828168],"about_ca_topic_score_codex":0.0038794263,"about_ca_topic_score_gemma":0.0053108977,"teacher_disagreement_score":0.010879029,"about_ca_system_score_codex":0.0030876454,"about_ca_system_score_gemma":0.0034316892,"threshold_uncertainty_score":0.057534456},"labels":[],"label_agreement":null},{"id":"W2756726036","doi":"10.18653/v1/w17-4732","title":"NRC Machine Translation System for WMT 2017","year":2017,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Google (Canada); National Research Council Canada","funders":"","keywords":"Machine translation; Computer science; Translation (biology); Machine translation system; Artificial intelligence; Natural language processing; Chemistry","score_opus":0.03566946681279727,"score_gpt":0.3180401626370161,"score_spread":0.28237069582421886,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2756726036","genre_codex":"software","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.110590406,0.00369491,0.16702062,0.0031727797,0.004989375,0.0065318807,0.1884655,0.32070467,0.19482984],"genre_scores_gemma":[0.14705865,0.000828652,0.27382672,0.0013188719,0.00043639189,0.004285586,0.48029745,0.013885035,0.07806267],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9967816,0.0005897281,0.00032324065,0.0007487519,0.0011962625,0.0003605336],"domain_scores_gemma":[0.99464226,0.00040697565,0.00018119124,0.0016089847,0.0027448258,0.00041583888],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004044409,0.0018205323,0.001618206,0.002410653,0.0023738502,0.0020815213,0.00209485,0.0015750006,0.035901386],"category_scores_gemma":[0.007975426,0.0006381025,0.0008229931,0.0023045694,0.0006559248,0.002517029,0.0029057541,0.002228485,0.05916378],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005400531,0.00041213122,0.0019103391,0.00086769817,0.00009486777,0.0007494921,0.0011980481,0.0013693216,0.02642152,0.004047533,0.73150456,0.23088443],"study_design_scores_gemma":[0.0003648745,0.0005370548,0.0070502716,0.00017823948,0.00010814411,0.0018469738,0.0006288082,0.021282146,0.046728298,0.003457242,0.9175649,0.00025302902],"about_ca_topic_score_codex":0.023568658,"about_ca_topic_score_gemma":0.035594862,"teacher_disagreement_score":0.035901386,"about_ca_system_score_codex":0.0021153502,"about_ca_system_score_gemma":0.0072523993,"threshold_uncertainty_score":0.12010217},"labels":[],"label_agreement":null},{"id":"W2757109549","doi":"","title":"Compositional Semantics of Coordination using Synchronous Tree Adjoining Grammar.","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Syntax; Semantics (computer science); Sentence; Conjunction (astronomy); Tree (set theory); Grammar; Argument (complex analysis); Quantifier (linguistics); Meaning (existential); Natural language processing; Artificial intelligence; Linguistics; Mathematics; Programming language; Philosophy; Combinatorics","score_opus":0.023558457096241932,"score_gpt":0.26893599171539223,"score_spread":0.2453775346191503,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2757109549","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0074458267,0.00013993416,0.9821358,0.00041677576,0.000099983925,0.000077169614,0.00009197188,0.0006001828,0.008992369],"genre_scores_gemma":[0.48272184,0.00038456803,0.50976616,0.0004412472,0.00022110392,0.00030882953,0.00031544425,0.0004520722,0.0053888056],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99801207,0.00074322673,0.00017127895,0.00040372423,0.00048318153,0.00018647556],"domain_scores_gemma":[0.9983699,0.0006520673,0.00020269604,0.000351339,0.000298517,0.0001255211],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020947403,0.0007176306,0.00052778947,0.0012592066,0.001481872,0.002368731,0.0017254502,0.0013665134,0.004932971],"category_scores_gemma":[0.0037413533,0.00049612933,0.0015552575,0.0010039461,0.0044344156,0.0067187427,0.0027152011,0.0017032819,0.0007602865],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000009657389,0.000008310061,0.00011038349,0.000036941972,0.000008550889,0.00017727616,0.00051891775,0.0015336348,0.0017086269,0.9905241,0.00043921632,0.004924505],"study_design_scores_gemma":[0.000025418993,0.000026861791,0.00015850656,0.000025831006,0.000031619005,0.00030546822,0.0002492284,0.030504422,0.0024698004,0.9484578,0.017725248,0.000019741508],"about_ca_topic_score_codex":0.0017583349,"about_ca_topic_score_gemma":0.0015938954,"teacher_disagreement_score":0.004932971,"about_ca_system_score_codex":0.001166494,"about_ca_system_score_gemma":0.0015097844,"threshold_uncertainty_score":0.01650244},"labels":[],"label_agreement":null},{"id":"W2757364855","doi":"10.5539/ijel.v7n6p109","title":"The Impact of the Data-Driven Learning Approach on ESL Writers’ Citation Patterns","year":2017,"lang":"en","type":"article","venue":"International Journal of English Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Citation; Computer science; Test (biology); Set (abstract data type); Sample (material); Mathematics education; Natural language processing; Mathematics; Linguistics; Library science","score_opus":0.038311973793618184,"score_gpt":0.34840097664450204,"score_spread":0.31008900285088387,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2757364855","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"reporting","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"reporting","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9848626,0.00025796643,0.009239891,0.00039312043,0.00003611939,0.00020420199,0.00014074982,0.00027773823,0.004587533],"genre_scores_gemma":[0.9820166,0.00015751833,0.015327057,0.00012665226,0.00003893639,0.00019175089,0.0002654936,0.000038786773,0.0018371625],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.98843294,0.005840545,0.0008559481,0.0013272687,0.0032616402,0.000281632],"domain_scores_gemma":[0.89104456,0.082915306,0.0070406124,0.008148192,0.007820623,0.0030306736],"candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.010074463,0.0004440247,0.00058282225,0.002406351,0.00073157065,0.0030361344,0.0012985639,0.000665648,0.0012378531],"category_scores_gemma":[0.061176803,0.00022867776,0.00046196242,0.0022717933,0.0006713233,0.0018519519,0.0026695305,0.0009016245,0.0006071366],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012288797,0.0039029326,0.144003,0.00058237,0.000116874624,0.00019766758,0.014135721,0.0032193658,0.010843863,0.00097264594,0.0009943679,0.8198023],"study_design_scores_gemma":[0.0007630721,0.017202592,0.7451473,0.00058459473,0.00054202107,0.0014543119,0.037822213,0.056814685,0.073670484,0.012321756,0.053200908,0.00047616995],"about_ca_topic_score_codex":0.00092161883,"about_ca_topic_score_gemma":0.0016895473,"teacher_disagreement_score":0.99759364,"about_ca_system_score_codex":0.0010698986,"about_ca_system_score_gemma":0.0015101942,"threshold_uncertainty_score":0.05327952},"labels":[],"label_agreement":null},{"id":"W2757376562","doi":"10.18653/v1/d17-1073","title":"Don't Throw Those Morphological Analyzers Away Just Yet: Neural Morphological Disambiguation for Arabic","year":2017,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":56,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"York University; New York University Abu Dhabi","keywords":"Computer science; Artificial intelligence; Ranking (information retrieval); Arabic; Feature (linguistics); Pattern recognition (psychology); Feature engineering; Approximation error; Artificial neural network; Embedding; Recurrent neural network; Vocabulary; Feature extraction; Speech recognition; Natural language processing; Deep learning; Algorithm","score_opus":0.07278266145795757,"score_gpt":0.3595368032119135,"score_spread":0.2867541417539559,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2757376562","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.083258934,0.0022516847,0.8810535,0.0014555034,0.0006086949,0.000079319725,0.0006127702,0.021532578,0.009147074],"genre_scores_gemma":[0.4935045,0.0011221712,0.4919145,0.0006594633,0.00011080746,0.000042736545,0.0010948868,0.00084809144,0.01070283],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995827,0.00008009925,0.000039851908,0.0001628224,0.00010439472,0.000030231462],"domain_scores_gemma":[0.99941564,0.0001702784,0.000085530846,0.00011811567,0.00018088674,0.00002962937],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00068995636,0.00091062026,0.00052375573,0.000950351,0.0007583159,0.0013514148,0.0010874299,0.00089597324,0.004036325],"category_scores_gemma":[0.0020923028,0.0003667418,0.0006689876,0.00070909946,0.0006020906,0.00306448,0.000984098,0.001124623,0.00485068],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00045913045,0.0000524915,0.002212575,0.0002891229,0.00009988477,0.00042716172,0.0003922556,0.017128067,0.059505817,0.006529733,0.0101911025,0.90271264],"study_design_scores_gemma":[0.000038684204,0.00017722671,0.0031178906,0.00015400673,0.00020324375,0.0013554733,0.00076405157,0.7750937,0.13448897,0.0340655,0.050417952,0.00012325794],"about_ca_topic_score_codex":0.003380641,"about_ca_topic_score_gemma":0.0062142415,"teacher_disagreement_score":0.004036325,"about_ca_system_score_codex":0.00042457102,"about_ca_system_score_gemma":0.0006594276,"threshold_uncertainty_score":0.013502896},"labels":[],"label_agreement":null},{"id":"W2758547268","doi":"","title":"Binding Variables in English: An Analysis Using Delayed Tree Locality","year":2010,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Locality; Upper and lower bounds; Equivalence (formal languages); Mathematics; Tree (set theory); Variable (mathematics); Component (thermodynamics); Constraint (computer-aided design); Set (abstract data type); Computer science; Algorithm; Discrete mathematics; Combinatorics; Linguistics; Geometry","score_opus":0.01593258944550329,"score_gpt":0.289820269867007,"score_spread":0.2738876804215037,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2758547268","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20356299,0.00031087035,0.7791668,0.0003376727,0.000033866425,0.000068712376,0.000120472025,0.00030963757,0.01608893],"genre_scores_gemma":[0.8818591,0.0002763912,0.11105697,0.00010491927,0.000031630785,0.000051833213,0.0000836759,0.0002654436,0.0062700342],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9991565,0.00026156197,0.00006013223,0.00014935792,0.00026584393,0.00010657194],"domain_scores_gemma":[0.99836427,0.000851124,0.00015126533,0.00024442468,0.0003411147,0.00004780209],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009058616,0.00028481567,0.00033673106,0.0007358529,0.00096932065,0.0013164807,0.0008710623,0.00043558388,0.0029619064],"category_scores_gemma":[0.0026401659,0.00045292117,0.00058201706,0.0006761186,0.0013981009,0.003336709,0.0014538643,0.0010136855,0.00036903247],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000058819263,0.00002689157,0.0022254747,0.00009877014,0.000021967295,0.0007724502,0.001842294,0.014393863,0.01383599,0.94557476,0.0004026896,0.020745935],"study_design_scores_gemma":[0.000048574864,0.000117458345,0.002541441,0.00004866974,0.00015834473,0.0012390112,0.0015374494,0.17535555,0.04205245,0.7304265,0.046408173,0.000066337896],"about_ca_topic_score_codex":0.0035598103,"about_ca_topic_score_gemma":0.0035513225,"teacher_disagreement_score":0.0035598103,"about_ca_system_score_codex":0.0010787895,"about_ca_system_score_gemma":0.0008020186,"threshold_uncertainty_score":0.009908557},"labels":[],"label_agreement":null},{"id":"W2758909929","doi":"10.18653/v1/w17-5041","title":"Exploring Optimal Voting in Native Language Identification","year":2017,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"National Research Council Canada","funders":"National Research Council Canada","keywords":"Voting; Preprocessor; Computer science; Identification (biology); Track (disk drive); Artificial intelligence; Natural language processing; Speech recognition; Political science","score_opus":0.0739580548976712,"score_gpt":0.33200639820877137,"score_spread":0.25804834331110016,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2758909929","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3225535,0.0019525241,0.6621129,0.0015067388,0.00020831729,0.00021768782,0.00023860089,0.0018245796,0.0093851965],"genre_scores_gemma":[0.8698122,0.00016987738,0.12459511,0.00030549936,0.00008548548,0.00013043387,0.00045625577,0.00029845632,0.004146801],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9940252,0.0034842866,0.00023118363,0.0009973992,0.00052822556,0.000733731],"domain_scores_gemma":[0.9903452,0.0068341983,0.00028667488,0.00088647933,0.0013223502,0.00032506863],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009357159,0.0009534432,0.0020273854,0.00096256274,0.0016180883,0.0016820087,0.0016567009,0.0014145209,0.0027556245],"category_scores_gemma":[0.021109303,0.00058912765,0.000874469,0.0007594502,0.0012239703,0.002888827,0.0027872054,0.0013731623,0.00082018645],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012551302,0.00044349607,0.009777398,0.00041461506,0.00021684915,0.00033644246,0.0011975417,0.36237675,0.015621796,0.034416996,0.008391265,0.5655517],"study_design_scores_gemma":[0.000067627356,0.0002014195,0.0009398128,0.00003109122,0.000037995756,0.00009864264,0.00035113867,0.94637996,0.006611577,0.04236601,0.0028840955,0.000030610157],"about_ca_topic_score_codex":0.004678635,"about_ca_topic_score_gemma":0.006818276,"teacher_disagreement_score":0.009357159,"about_ca_system_score_codex":0.0012507656,"about_ca_system_score_gemma":0.0018807397,"threshold_uncertainty_score":0.04948598},"labels":[],"label_agreement":null},{"id":"W2759181158","doi":"10.18653/v1/2023.acl-short","title":"Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 2: Short Papers)","year":2023,"lang":"en","type":"paratext","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":201,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"Atomic Energy of Canada Limited","keywords":"Volume (thermodynamics); Computational linguistics; Computer science; Association (psychology); Natural language processing; Psychology; Physics","score_opus":0.012214328987654597,"score_gpt":0.27600368312666973,"score_spread":0.2637893541390151,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2759181158","genre_codex":"editorial","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007923364,0.14872523,0.03583389,0.1830549,0.36683813,0.0010005505,0.022587886,0.006088659,0.22794737],"genre_scores_gemma":[0.023627779,0.08117965,0.03749578,0.038014032,0.074130766,0.0017784853,0.041543137,0.0075707003,0.69465965],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99508274,0.0015890952,0.00076207885,0.00092084263,0.0013186267,0.0003267028],"domain_scores_gemma":[0.9701073,0.012146588,0.0011730285,0.0030079125,0.010397433,0.003167707],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012245306,0.0011657192,0.0026683332,0.0055887643,0.0027581996,0.0121259205,0.0026863138,0.0037299357,0.26177445],"category_scores_gemma":[0.032085393,0.0010401485,0.0011934509,0.004587137,0.0021128112,0.010218676,0.003736968,0.004847599,0.2136264],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005466466,0.00003656861,0.00045102538,0.0003574498,0.000028153021,0.00005916535,0.00013622738,0.000025267316,0.00032325884,0.0011692898,0.9388952,0.058463767],"study_design_scores_gemma":[0.000016242237,0.000017234885,0.0009631094,0.0005457109,0.000019864812,0.00013027321,0.0002466537,0.00014163523,0.00011322965,0.0021242225,0.9956631,0.000018840983],"about_ca_topic_score_codex":0.0041202414,"about_ca_topic_score_gemma":0.005488215,"teacher_disagreement_score":0.26177445,"about_ca_system_score_codex":0.0027342674,"about_ca_system_score_gemma":0.0055724345,"threshold_uncertainty_score":0.87572277},"labels":[],"label_agreement":null},{"id":"W2760656271","doi":"10.18653/v1/w17-4717","title":"Findings of the 2017 Conference on Machine Translation (WMT17)","year":2017,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":418,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"European Regional Development Fund; Horizon 2020 Framework Programme; Agence Nationale de la Recherche; Univerzita Karlova v Praze; Trinity College Dublin; Science Foundation Ireland; European Commission","keywords":"Chatterjee; Translation (biology); Machine translation; Art history; Computer science; Art; Artificial intelligence; Chemistry","score_opus":0.04852302411475251,"score_gpt":0.3154419876781849,"score_spread":0.26691896356343237,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2760656271","genre_codex":"editorial","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023678074,0.083333015,0.057961673,0.20460652,0.255079,0.002390971,0.19886978,0.013043599,0.16103736],"genre_scores_gemma":[0.052779607,0.04301537,0.07751177,0.019226363,0.03631927,0.0026797645,0.53807193,0.010991629,0.21940425],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9782381,0.006434896,0.0015989755,0.002060077,0.009632913,0.0020350262],"domain_scores_gemma":[0.93850064,0.011079329,0.0015674811,0.007313407,0.033049565,0.008489511],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.030872004,0.003786767,0.0031948495,0.0146963615,0.004098598,0.012809902,0.003553762,0.004305138,0.06335443],"category_scores_gemma":[0.042726453,0.00090413226,0.0024776808,0.011195258,0.0028892115,0.010541841,0.009254429,0.0066333665,0.069488324],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00038416858,0.00017084485,0.0008766606,0.00051721913,0.00006388574,0.00014514854,0.00012109678,0.00048682722,0.00087994826,0.0027148463,0.93855923,0.055080198],"study_design_scores_gemma":[0.00014842422,0.00015241974,0.003742968,0.00069049385,0.00012469677,0.00027331797,0.0005522036,0.0023631558,0.006834755,0.008806232,0.9762357,0.00007548397],"about_ca_topic_score_codex":0.01813523,"about_ca_topic_score_gemma":0.026219193,"teacher_disagreement_score":0.06335443,"about_ca_system_score_codex":0.0049894066,"about_ca_system_score_gemma":0.014380106,"threshold_uncertainty_score":0.21194172},"labels":[],"label_agreement":null},{"id":"W2763036384","doi":"10.3233/aac-170030","title":"Rhetorical figures, arguments, computation","year":2017,"lang":"en","type":"article","venue":"Argument & Computation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Rhetorical question; Computation; Computer science; Linguistics; Philosophy; Programming language","score_opus":0.027162982335720814,"score_gpt":0.32232926865245237,"score_spread":0.2951662863167316,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2763036384","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.047980312,0.07887578,0.60185826,0.034485284,0.0016305792,0.00013307942,0.00079630775,0.0012663414,0.232974],"genre_scores_gemma":[0.74839544,0.019253928,0.19614397,0.0020140915,0.0030322399,0.00024773463,0.0007909882,0.0004785828,0.029642958],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9979522,0.0010002992,0.00013252831,0.00040855515,0.0004047457,0.00010160435],"domain_scores_gemma":[0.99477464,0.003944766,0.00031308105,0.0004965983,0.0003365773,0.00013441339],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024318786,0.0007170116,0.0011291412,0.0043227687,0.002896288,0.012018177,0.0013354251,0.0033000328,0.014692357],"category_scores_gemma":[0.010740886,0.0009005509,0.00087728596,0.005246506,0.013569112,0.021245101,0.0023506593,0.0042673773,0.0026537674],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000010114877,0.00001304929,0.00012348252,0.00011718125,0.0000070460783,0.00003745259,0.0003629194,0.00037508033,0.0002130935,0.98215365,0.0029093267,0.013677606],"study_design_scores_gemma":[0.0000046818795,0.0000024021203,0.00008893993,0.000029424364,0.0000058505752,0.000061388775,0.000110716144,0.0011748574,0.00009502169,0.98796964,0.010452039,0.000005124909],"about_ca_topic_score_codex":0.002324801,"about_ca_topic_score_gemma":0.0021110463,"teacher_disagreement_score":0.014692357,"about_ca_system_score_codex":0.0031021172,"about_ca_system_score_gemma":0.0013189517,"threshold_uncertainty_score":0.049150884},"labels":[],"label_agreement":null},{"id":"W2763898218","doi":"10.18653/v1/w17-01","title":"Proceedings of the 2nd Workshop on the Use of Computational Methods in the Study of Endangered Languages","year":2017,"lang":"en","type":"paratext","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"Social Sciences and Humanities Research Council of Canada; Universität zu Köln; University of Alberta; National Science Foundation","keywords":"Endangered species; Computer science; Programming language; Ecology; Biology","score_opus":0.09630619699977382,"score_gpt":0.40520610332370166,"score_spread":0.30889990632392783,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2763898218","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007931686,0.04776649,0.71880287,0.079908855,0.037071977,0.0006163507,0.0026183838,0.0030845115,0.10219886],"genre_scores_gemma":[0.07502525,0.059001632,0.53210896,0.009945745,0.020762004,0.0014914577,0.012347479,0.008192041,0.2811255],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9941022,0.0036101085,0.00034541075,0.0006703837,0.0009798596,0.00029212877],"domain_scores_gemma":[0.9762749,0.01614948,0.0003625338,0.003417419,0.0022081838,0.001587428],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01554575,0.0012000604,0.00169621,0.0029771095,0.001747177,0.00920178,0.00363636,0.0025322384,0.047280613],"category_scores_gemma":[0.020545332,0.0009191853,0.0019600024,0.0022621008,0.004581778,0.0076642027,0.0069630113,0.006424239,0.012391823],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003213564,0.0003265227,0.0012225937,0.0011114235,0.0002063505,0.00051654555,0.0025093714,0.0041651814,0.002603404,0.09180016,0.5227404,0.3724766],"study_design_scores_gemma":[0.000047555037,0.000041399555,0.0010845112,0.00083688006,0.00005179482,0.00032269707,0.000516033,0.005855308,0.0015416526,0.10281849,0.88682836,0.000055298853],"about_ca_topic_score_codex":0.005472035,"about_ca_topic_score_gemma":0.011087813,"teacher_disagreement_score":0.047280613,"about_ca_system_score_codex":0.002688989,"about_ca_system_score_gemma":0.0044998196,"threshold_uncertainty_score":0.15816939},"labels":[],"label_agreement":null},{"id":"W2765190181","doi":"10.1145/3133323","title":"Linguistic-Relationships-Based Approach for Improving Word Alignment","year":2017,"lang":"en","type":"article","venue":"ACM Transactions on Asian and Low-Resource Language Information Processing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Vietnamese; Computer science; Word (group theory); Natural language processing; Machine translation; Phrase; Artificial intelligence; Linguistics; Quality (philosophy); Bilingual dictionary","score_opus":0.015150148058170393,"score_gpt":0.26381645205486953,"score_spread":0.24866630399669915,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2765190181","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.027332187,0.0009841961,0.9607783,0.00032247795,0.00011950693,0.00016398435,0.00026590296,0.005452445,0.0045809764],"genre_scores_gemma":[0.22868027,0.0006846948,0.7622442,0.00042715733,0.00015338777,0.00023352377,0.0014387305,0.0007743921,0.0053637372],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99832636,0.00045064115,0.0001673919,0.0005269001,0.0004479648,0.00008067101],"domain_scores_gemma":[0.9989593,0.00022099243,0.0001815753,0.00019266186,0.0004012409,0.000044172484],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00092286186,0.0013031437,0.0007375966,0.0028406573,0.0008682159,0.0007436069,0.0011655266,0.00089403114,0.0034554505],"category_scores_gemma":[0.0029675753,0.00049868535,0.000995458,0.0036130869,0.00045211564,0.0026608987,0.0014199938,0.0013783441,0.0030771464],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021112734,0.00034786828,0.0021738713,0.00042132012,0.00016564054,0.00029322947,0.0006194194,0.05071463,0.092423186,0.012662328,0.009412607,0.8305548],"study_design_scores_gemma":[0.00007171965,0.00025271287,0.0023185217,0.000036365447,0.00023066935,0.0004944015,0.0002695934,0.9049009,0.045455914,0.016511748,0.029373245,0.0000842583],"about_ca_topic_score_codex":0.0039874255,"about_ca_topic_score_gemma":0.0073790336,"teacher_disagreement_score":0.0039874255,"about_ca_system_score_codex":0.0004947456,"about_ca_system_score_gemma":0.0015802295,"threshold_uncertainty_score":0.011559665},"labels":[],"label_agreement":null},{"id":"W2766945157","doi":"10.1109/edocw.2017.24","title":"Message from the VORTE 2017 Workshop Chairs","year":2017,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Conselho Nacional de Desenvolvimento Científico e Tecnológico","keywords":"Computer science; Software engineering; World Wide Web; Data science","score_opus":0.028146450214188073,"score_gpt":0.30243773395254714,"score_spread":0.2742912837383591,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2766945157","genre_codex":"commentary","genre_gemma":"editorial","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"editorial","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0005257868,0.0030467825,0.0008628616,0.59606993,0.38732302,0.00009102198,0.00096014317,0.00032281023,0.0107976915],"genre_scores_gemma":[0.011317604,0.0046836943,0.0017960677,0.60068387,0.14660308,0.00059638196,0.0019274406,0.00085048866,0.23154135],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99335104,0.0012643085,0.00033044026,0.0010736521,0.0028993196,0.001081264],"domain_scores_gemma":[0.97883695,0.0021172788,0.0007144893,0.00055619807,0.009162906,0.0086122565],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009718879,0.0018002532,0.0013319459,0.0011691817,0.004358865,0.009449835,0.0021240038,0.012739591,0.05716415],"category_scores_gemma":[0.022617059,0.0005400255,0.0017084782,0.00090242684,0.0014678297,0.0060535586,0.0070429663,0.022835422,0.047369957],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000028296869,0.000012574347,0.00005879861,0.000028590834,0.0000024640012,0.00003499086,0.00003360173,0.0000121177745,0.00005307546,0.00048440072,0.9968939,0.0023571486],"study_design_scores_gemma":[0.000020248217,0.000025921956,0.00026342564,0.00011728122,0.0000071803447,0.00005034773,0.00035950326,0.00005838863,0.00016993882,0.0007269906,0.9981772,0.000023556897],"about_ca_topic_score_codex":0.0038176335,"about_ca_topic_score_gemma":0.0064056963,"teacher_disagreement_score":0.05716415,"about_ca_system_score_codex":0.0035866348,"about_ca_system_score_gemma":0.009428248,"threshold_uncertainty_score":0.1912331},"labels":[],"label_agreement":null},{"id":"W2767435535","doi":"10.1145/3132847.3133106","title":"Paraphrastic Fusion for Abstractive Multi-Sentence Compression Generation","year":2017,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Lethbridge","funders":"","keywords":"Computer science; Sentence; Natural language processing; Artificial intelligence; Compression (physics); Fusion; Speech recognition; Linguistics; Materials science","score_opus":0.06880851776482023,"score_gpt":0.35683083086691203,"score_spread":0.2880223131020918,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2767435535","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018672364,0.0007875347,0.9662601,0.00027799862,0.00018654301,0.00029687266,0.000652126,0.010078876,0.0027876347],"genre_scores_gemma":[0.22672547,0.00067421445,0.76225495,0.0003386431,0.0003679605,0.00032623223,0.00371485,0.0007336327,0.00486415],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99875665,0.00039243925,0.00012485826,0.0002513142,0.00040942608,0.000065253524],"domain_scores_gemma":[0.9972282,0.000984162,0.00028373484,0.0006206619,0.0008126771,0.000070542155],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015572824,0.001329978,0.0009835608,0.002116518,0.00055338466,0.0010319769,0.0013666573,0.0009203513,0.00559827],"category_scores_gemma":[0.0052023227,0.00030377632,0.00085010874,0.0013030464,0.0005074126,0.0019670697,0.0011710452,0.0014259592,0.0034237418],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003436253,0.00029327333,0.0008697572,0.0006561497,0.00014207092,0.0004512581,0.0004948957,0.016903896,0.101615176,0.01102601,0.01448429,0.8527196],"study_design_scores_gemma":[0.00012504523,0.0008103831,0.002316188,0.00011285093,0.00030655492,0.001589746,0.0002805065,0.6813281,0.25159425,0.022881735,0.038516905,0.00013772263],"about_ca_topic_score_codex":0.0011215477,"about_ca_topic_score_gemma":0.0014395391,"teacher_disagreement_score":0.00559827,"about_ca_system_score_codex":0.0005796158,"about_ca_system_score_gemma":0.00069049833,"threshold_uncertainty_score":0.018728077},"labels":[],"label_agreement":null},{"id":"W2767633796","doi":"10.1016/j.procs.2017.10.121","title":"Arabic Social Media Analysis and Translation","year":2017,"lang":"en","type":"article","venue":"Procedia Computer Science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Computer science; Natural language processing; Normalization (sociology); Artificial intelligence; Arabic; Machine translation; Social media; Unavailability; Domain (mathematical analysis); Focus (optics); Modern Standard Arabic; Context (archaeology); Linguistics; World Wide Web","score_opus":0.023165091138060312,"score_gpt":0.2917618620317705,"score_spread":0.2685967708937102,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2767633796","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20764796,0.0034506747,0.5440771,0.0058515472,0.0034426872,0.0020037687,0.043145046,0.021601124,0.16878018],"genre_scores_gemma":[0.5428882,0.002484203,0.3428852,0.0005963716,0.00076095964,0.0013673904,0.035676807,0.0018738189,0.07146707],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988456,0.00038780482,0.00010742692,0.00020804329,0.00036429465,0.000086891916],"domain_scores_gemma":[0.99810815,0.0004174469,0.000153016,0.00028153238,0.0009789312,0.000061007297],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009113624,0.001252502,0.0003247221,0.0042607505,0.0013281032,0.0020231793,0.0003719669,0.00037518615,0.014432936],"category_scores_gemma":[0.004615332,0.0002058568,0.0006606788,0.0022474602,0.00044705562,0.0014478664,0.0012394334,0.00072578585,0.010695842],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003394982,0.00018613828,0.0050279675,0.0007831753,0.00007950302,0.0009833726,0.0020784808,0.004558799,0.035505332,0.022520773,0.058034163,0.86990285],"study_design_scores_gemma":[0.000051524345,0.00018268883,0.020725843,0.00028105904,0.00010115163,0.001127168,0.004958603,0.13246842,0.13982809,0.028407777,0.6717025,0.00016521112],"about_ca_topic_score_codex":0.004082279,"about_ca_topic_score_gemma":0.003172817,"teacher_disagreement_score":0.014432936,"about_ca_system_score_codex":0.0008484012,"about_ca_system_score_gemma":0.001031088,"threshold_uncertainty_score":0.04828298},"labels":[],"label_agreement":null},{"id":"W2769298630","doi":"10.1162/tacl_a_00011","title":"Modeling Past and Future for Neural Machine Translation","year":2018,"lang":"en","type":"article","venue":"Transactions of the Association for Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":54,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"National Natural Science Foundation of China; National Science Foundation","keywords":"Machine translation; Computer science; Decoding methods; Translation (biology); German; Artificial intelligence; Natural language processing; Mechanism (biology); Speech recognition; Machine learning; Algorithm; Linguistics","score_opus":0.01634169044876612,"score_gpt":0.28153284019936564,"score_spread":0.26519114975059954,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2769298630","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.045944355,0.0041804863,0.94020426,0.0012672617,0.0002635128,0.000035936664,0.0005661241,0.0017767523,0.005761333],"genre_scores_gemma":[0.83075124,0.0025090543,0.1570445,0.00039890656,0.0002552245,0.00015094751,0.0014286898,0.00024995787,0.007211437],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996451,0.00011257095,0.000030885032,0.00012698912,0.00005607917,0.000028349104],"domain_scores_gemma":[0.9990958,0.00048010857,0.000093739785,0.00017221189,0.00012964668,0.000028424707],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00091359636,0.00066192704,0.000573656,0.00070674595,0.0005064672,0.0010373702,0.0011394416,0.000997578,0.0034974127],"category_scores_gemma":[0.0038847865,0.00043964028,0.00077932375,0.0009936851,0.00068613683,0.003852327,0.0009964558,0.0014965514,0.0011962759],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00066745636,0.00013989431,0.0040025655,0.0006286479,0.00022835737,0.00068002805,0.00090586994,0.37664717,0.02355266,0.13701594,0.008923014,0.4466084],"study_design_scores_gemma":[0.000015386975,0.000045348013,0.000562121,0.00003487427,0.000068666304,0.00015421759,0.00003344014,0.9291243,0.0034203343,0.059313953,0.0072031016,0.000024265726],"about_ca_topic_score_codex":0.0052925306,"about_ca_topic_score_gemma":0.008616557,"teacher_disagreement_score":0.0052925306,"about_ca_system_score_codex":0.0007702792,"about_ca_system_score_gemma":0.00081983133,"threshold_uncertainty_score":0.011700034},"labels":[],"label_agreement":null},{"id":"W2771330107","doi":"10.1162/neco_a_01111","title":"An Empirical Evaluation of Rule Extraction from Recurrent Neural Networks","year":2018,"lang":"en","type":"article","venue":"Neural Computation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":58,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Recurrent neural network; Computer science; Artificial intelligence; Black box; Machine learning; Artificial neural network; Rule-based machine translation","score_opus":0.057603525015650484,"score_gpt":0.40381459477433357,"score_spread":0.3462110697586831,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2771330107","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.87574446,0.0066754925,0.10255934,0.0007834165,0.00021779559,0.00029069878,0.0044271206,0.0035822792,0.005719539],"genre_scores_gemma":[0.9127691,0.0008096635,0.07045164,0.00012951813,0.00009533851,0.00018606213,0.013384054,0.00033026992,0.0018442896],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99368435,0.0032139474,0.000634894,0.0009704068,0.0012778173,0.00021859414],"domain_scores_gemma":[0.8657256,0.11637094,0.0028492552,0.008731889,0.005747423,0.00057490706],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011943957,0.0015748723,0.0011333681,0.002360907,0.00061502913,0.0014459223,0.0015676051,0.0021158454,0.0024706714],"category_scores_gemma":[0.079814315,0.00046867036,0.00091665203,0.0017270156,0.0010364401,0.0033667453,0.0013985831,0.0017160648,0.001237031],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0031561789,0.001291438,0.03777664,0.0018999382,0.0010015877,0.000672177,0.0003694158,0.5825722,0.00773206,0.0044236565,0.0113245165,0.3477802],"study_design_scores_gemma":[0.00011991286,0.0006316726,0.0082061635,0.000106702355,0.00011277283,0.00030866696,0.000113596965,0.9746832,0.00930254,0.0041140374,0.0022640123,0.000036754467],"about_ca_topic_score_codex":0.0026031951,"about_ca_topic_score_gemma":0.003817287,"teacher_disagreement_score":0.011943957,"about_ca_system_score_codex":0.00094899465,"about_ca_system_score_gemma":0.0006734219,"threshold_uncertainty_score":0.06316638},"labels":[],"label_agreement":null},{"id":"W2772421893","doi":"10.26615/978-954-452-049-6_070","title":"Classifying Frames at the Sentence Level in News Articles","year":2017,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":32,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Sentence; Computer science; Newspaper; Frame (networking); Artificial intelligence; Baseline (sea); Point (geometry); Natural language processing; Meaning (existential); Information retrieval; Advertising; Mathematics","score_opus":0.07420992841022758,"score_gpt":0.3248417685656015,"score_spread":0.2506318401553739,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2772421893","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5638482,0.0041635865,0.39921802,0.00139277,0.0007474601,0.0006612783,0.009931717,0.0066113966,0.013425604],"genre_scores_gemma":[0.74596584,0.0012978106,0.22633615,0.00027756987,0.00049422,0.00022095953,0.019648641,0.00023620915,0.0055225403],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992949,0.00015338637,0.00006703895,0.00025930215,0.00012974204,0.000095664145],"domain_scores_gemma":[0.9984175,0.0005538889,0.00025047004,0.00016570743,0.00052809046,0.000084447376],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00079613074,0.0010803348,0.0006100909,0.003661885,0.00064668513,0.001084356,0.00068862084,0.0010638396,0.0019593486],"category_scores_gemma":[0.0030380979,0.0001884446,0.0008455584,0.0020144908,0.00038242366,0.0022768292,0.0007002889,0.0011051018,0.0014716535],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00053768384,0.00047475914,0.020482618,0.00060803373,0.00018784907,0.00031500348,0.00073159597,0.009506614,0.05387659,0.0035313575,0.01678326,0.89296466],"study_design_scores_gemma":[0.00007832135,0.00070811476,0.06426463,0.00025080764,0.0004823398,0.00062945124,0.002724864,0.79267794,0.08765354,0.023640938,0.026751686,0.00013742219],"about_ca_topic_score_codex":0.0058539524,"about_ca_topic_score_gemma":0.01505946,"teacher_disagreement_score":0.0058539524,"about_ca_system_score_codex":0.0007836721,"about_ca_system_score_gemma":0.0006927807,"threshold_uncertainty_score":0.011639774},"labels":[],"label_agreement":null},{"id":"W2772828501","doi":"10.5334/kula.3","title":"Modes of Annotation in the Video-Based Corpus FrancoToile: Developing a Design Method","year":2017,"lang":"en","type":"article","venue":"KULA knowledge creation dissemination and preservation studies","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"","keywords":"Computer science; Annotation; Focus (optics); Corpus linguistics; Natural language processing; Process (computing); Linguistics; Software; Text corpus; Artificial intelligence; World Wide Web","score_opus":0.0950684179262708,"score_gpt":0.43982713340973756,"score_spread":0.34475871548346676,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2772828501","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017742729,0.00013156144,0.9594741,0.000675611,0.00012822659,0.010127282,0.00062302354,0.002245591,0.008851928],"genre_scores_gemma":[0.044012025,0.00008138044,0.93228084,0.00017780616,0.000040387567,0.017951613,0.00068810116,0.0007543495,0.004013634],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.96914446,0.022421874,0.001374185,0.0037724504,0.0027467466,0.0005402175],"domain_scores_gemma":[0.93001163,0.045666736,0.0018039001,0.008836031,0.012360807,0.0013208961],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.042887114,0.0012692775,0.0006566897,0.004274032,0.0028224546,0.0054639834,0.0031122775,0.0020073825,0.009465591],"category_scores_gemma":[0.056037568,0.0013949522,0.00087062124,0.0018688454,0.0043005473,0.006571651,0.004934659,0.0019914897,0.0030083607],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015005396,0.0011240619,0.008993391,0.0033130918,0.00013514384,0.001653703,0.12117368,0.0063054664,0.075695135,0.09583211,0.01755276,0.6667209],"study_design_scores_gemma":[0.0012955796,0.002306565,0.014207019,0.002417112,0.0002866273,0.0030583958,0.060596168,0.11428915,0.12800482,0.054982692,0.61758864,0.0009672048],"about_ca_topic_score_codex":0.0031986237,"about_ca_topic_score_gemma":0.004367594,"teacher_disagreement_score":0.042887114,"about_ca_system_score_codex":0.0027772777,"about_ca_system_score_gemma":0.003716689,"threshold_uncertainty_score":0.22681141},"labels":[],"label_agreement":null},{"id":"W2774419307","doi":"10.12685/027.7-5-1-169","title":"Transfer Learning for OCRopus Model Training on Early Printed Books","year":2017,"lang":"en","type":"preprint","venue":"027 7 Zeitschrift für Bibliothekskultur","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Character (mathematics); Computer science; Scratch; Ground truth; Alphabet; Code (set theory); Training set; Test set; Set (abstract data type); Natural language processing; Artificial intelligence; Training (meteorology); Character encoding; Test (biology); Test data; Speech recognition; Linguistics; Mathematics; Programming language","score_opus":0.05356677114805272,"score_gpt":0.3376364533032796,"score_spread":0.2840696821552269,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2774419307","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20312257,0.001425138,0.7528527,0.00043524397,0.0003281999,0.00034169565,0.0010021185,0.03225701,0.008235369],"genre_scores_gemma":[0.7240758,0.00037571808,0.25260288,0.00029888825,0.000120514174,0.00053845,0.0039141346,0.0013733165,0.016700268],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99919015,0.00019101353,0.00005458676,0.0002881106,0.00018116871,0.00009504361],"domain_scores_gemma":[0.99799144,0.0009640037,0.00010608861,0.0004849969,0.00037270962,0.00008073167],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012948378,0.00191898,0.0009798761,0.0010427276,0.00054103014,0.001157343,0.0020822117,0.0013500641,0.005948951],"category_scores_gemma":[0.0047219335,0.0006748838,0.0008032175,0.00096635235,0.0005946057,0.0018153277,0.0020225372,0.0026120797,0.00450159],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00038753744,0.00023866883,0.0015410669,0.00023151666,0.00015161501,0.0003215652,0.0002201912,0.1480245,0.030364169,0.001189337,0.007463806,0.8098661],"study_design_scores_gemma":[0.00002236015,0.00014531147,0.0009454142,0.000023113253,0.000033812732,0.00012307163,0.0000848509,0.9617483,0.0318819,0.0017889827,0.0031745075,0.000028306991],"about_ca_topic_score_codex":0.006155325,"about_ca_topic_score_gemma":0.0060683694,"teacher_disagreement_score":0.006155325,"about_ca_system_score_codex":0.0009768505,"about_ca_system_score_gemma":0.00084564177,"threshold_uncertainty_score":0.019901276},"labels":[],"label_agreement":null},{"id":"W2774524497","doi":"10.5539/elt.v11n1p110","title":"Quantitative Research in Systemic Functional Linguistics","year":2017,"lang":"en","type":"article","venue":"English Language Teaching","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"Department of Education of Guangdong Province","keywords":"Systemic functional linguistics; Applied linguistics; Linguistics; Quantitative linguistics; Psychology; Competence (human resources); Clinical linguistics; Theoretical linguistics; Computational linguistics; Structural linguistics; Qualitative research; Language assessment; Computer science; Sociology; Philosophy; Social science; Social psychology","score_opus":0.06724168365843029,"score_gpt":0.3930830598141428,"score_spread":0.3258413761557125,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2774524497","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024849335,0.017844329,0.79439497,0.020822512,0.0008279303,0.00037235464,0.00068188604,0.00033210864,0.13987462],"genre_scores_gemma":[0.77421236,0.008143423,0.20229131,0.003296605,0.0017113526,0.0017061998,0.00048743875,0.0004113024,0.0077400072],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.95326346,0.036723413,0.0015265599,0.0032222213,0.004645932,0.0006182766],"domain_scores_gemma":[0.7751902,0.19995275,0.0064909304,0.008196686,0.00929601,0.0008734768],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.045888584,0.001385185,0.0014168627,0.009339658,0.0023631835,0.0067195287,0.0014881677,0.0019440923,0.007506164],"category_scores_gemma":[0.0948766,0.0006917343,0.0011192844,0.010691543,0.026790004,0.012970252,0.003963351,0.0031252464,0.0007235163],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00001121066,0.000027703552,0.0006646095,0.00028064847,0.000024749843,0.00002936483,0.0015566561,0.0006791684,0.00014180383,0.98601115,0.0007770847,0.009795846],"study_design_scores_gemma":[0.000013920859,0.00004300269,0.0008829438,0.00027952954,0.000016570573,0.0000653213,0.0010764601,0.003093884,0.00026902868,0.98073876,0.013503843,0.00001664541],"about_ca_topic_score_codex":0.0023689372,"about_ca_topic_score_gemma":0.0010260943,"teacher_disagreement_score":0.045888584,"about_ca_system_score_codex":0.0058869654,"about_ca_system_score_gemma":0.004432782,"threshold_uncertainty_score":0.24268496},"labels":[],"label_agreement":null},{"id":"W2774582842","doi":"10.1162/tacl_a_00076","title":"Joint Prediction of Word Alignment with Alignment Types","year":2017,"lang":"en","type":"article","venue":"Transactions of the Association for Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Word (group theory); Artificial intelligence; Task (project management); Probabilistic logic; Natural language processing; Joint (building); Generative grammar; Pattern recognition (psychology)","score_opus":0.01894919867233834,"score_gpt":0.26651702727514015,"score_spread":0.2475678286028018,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2774582842","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16451982,0.000686596,0.82563275,0.0006548646,0.00021491,0.00013966335,0.0012148536,0.004391225,0.0025453875],"genre_scores_gemma":[0.824299,0.00022443291,0.16918427,0.0002194831,0.00015717106,0.00016710168,0.0027992928,0.0005205815,0.002428737],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9956328,0.0016082249,0.0002634276,0.0016449989,0.00060615275,0.00024436344],"domain_scores_gemma":[0.9849518,0.009473177,0.0017609766,0.001729292,0.0016385877,0.0004461689],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0038249542,0.0012038465,0.0015564909,0.0025898633,0.00072962395,0.001767745,0.0016908104,0.0019418346,0.0024511432],"category_scores_gemma":[0.019834494,0.00089378405,0.001553261,0.0023980662,0.0010780205,0.0068042637,0.0021178517,0.0036238611,0.003123739],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021251147,0.0008137728,0.067475684,0.00056098,0.0005576672,0.0004373915,0.0006687481,0.25853342,0.030395335,0.019005261,0.012799805,0.60662687],"study_design_scores_gemma":[0.000023634098,0.000061279214,0.0033426445,0.000016122834,0.0000353315,0.000082364,0.00005275379,0.96353,0.0037915453,0.028219473,0.00081629294,0.000028436942],"about_ca_topic_score_codex":0.0020358271,"about_ca_topic_score_gemma":0.0038919046,"teacher_disagreement_score":0.0038249542,"about_ca_system_score_codex":0.0007521151,"about_ca_system_score_gemma":0.0013714136,"threshold_uncertainty_score":0.020228505},"labels":[],"label_agreement":null},{"id":"W2779949567","doi":"10.26615/978-2-9701095-2-5_001","title":"A Comparison of Three Metrics for Detecting Cross- Linguistic Variations in Information Volume and Multiword Expressions Between Parallel Bitexts","year":2017,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Trois-Rivières","funders":"","keywords":"Metric (unit); Computer science; Annotation; Python (programming language); Artificial intelligence; Natural language processing; Word (group theory); Variation (astronomy); Volume (thermodynamics); Character (mathematics); Pattern recognition (psychology); Mathematics","score_opus":0.057052341314773596,"score_gpt":0.38817339833243547,"score_spread":0.33112105701766187,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2779949567","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6836404,0.0046520936,0.2869287,0.0004036824,0.00032132427,0.0004995353,0.004610754,0.0095624,0.009381011],"genre_scores_gemma":[0.6945233,0.00069331215,0.2942733,0.000071993214,0.00010454324,0.00055255974,0.0068285433,0.001094706,0.001857631],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9860701,0.0042848596,0.001612156,0.0021976044,0.0052237846,0.0006114639],"domain_scores_gemma":[0.93799895,0.038271144,0.0048533636,0.0049639754,0.012825149,0.0010874943],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010660711,0.00177253,0.0012610435,0.016094046,0.00090369605,0.004099649,0.0015521419,0.0014264012,0.0014497159],"category_scores_gemma":[0.057190325,0.0005022831,0.00088990026,0.009430002,0.0013108704,0.0044329506,0.0027867071,0.0010487001,0.00068210234],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0035172016,0.00053705287,0.098707125,0.0023247318,0.001739639,0.00044087702,0.004083246,0.025988203,0.067017905,0.007945484,0.008323882,0.77937454],"study_design_scores_gemma":[0.00036954167,0.003562645,0.3615327,0.00040046516,0.00073917286,0.0022839305,0.0041699354,0.4003562,0.19041039,0.014443433,0.02088048,0.00085109397],"about_ca_topic_score_codex":0.0043540127,"about_ca_topic_score_gemma":0.004727411,"teacher_disagreement_score":0.016094046,"about_ca_system_score_codex":0.0015703181,"about_ca_system_score_gemma":0.001063634,"threshold_uncertainty_score":0.056379914},"labels":[],"label_agreement":null},{"id":"W2781769144","doi":"10.2514/6.2018-0132","title":"Analysis of Unstructured Meshes from GMGW-1 / HiLiftPW-3","year":2018,"lang":"en","type":"article","venue":"2018  AIAA Aerospace Sciences Meeting","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Polygon mesh; Computer science; Computer graphics (images)","score_opus":0.01449860256580975,"score_gpt":0.2803708158811365,"score_spread":0.26587221331532673,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2781769144","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2969724,0.00027368142,0.6468089,0.0007016893,0.00039478662,0.00032099077,0.00534399,0.0070435912,0.042139962],"genre_scores_gemma":[0.7336495,0.00018857358,0.24243626,0.0002036815,0.000103654915,0.00022122875,0.009072325,0.003474504,0.010650288],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996681,0.00005280736,0.000010653101,0.00003146249,0.00020094373,0.000036109195],"domain_scores_gemma":[0.9994128,0.0002298639,0.00003756507,0.00009474547,0.00019311675,0.00003191967],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003880563,0.0007343849,0.00043924397,0.00070009235,0.00044657922,0.0008734821,0.0010715466,0.0010894472,0.0049413606],"category_scores_gemma":[0.002268534,0.00028781645,0.0006724675,0.0005279077,0.0005191428,0.00044853636,0.0008004382,0.00073344674,0.0010465637],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015892847,0.00011558443,0.002271112,0.0002183604,0.00005204267,0.00035531094,0.0001904911,0.92579895,0.018919865,0.011880862,0.011355793,0.028682599],"study_design_scores_gemma":[0.000010303551,0.000020290161,0.00073413644,0.000013626939,0.000003819512,0.000033245557,0.000041066447,0.9916483,0.0026565646,0.0020893356,0.0027422851,0.0000070739807],"about_ca_topic_score_codex":0.005746423,"about_ca_topic_score_gemma":0.006631101,"teacher_disagreement_score":0.005746423,"about_ca_system_score_codex":0.00033189348,"about_ca_system_score_gemma":0.0006695205,"threshold_uncertainty_score":0.016530514},"labels":[],"label_agreement":null},{"id":"W2781839508","doi":"10.1007/978-3-319-73706-5","title":"Language Technologies for the Challenges of the Digital Age","year":2018,"lang":"en","type":"book","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Bundesministerium für Bildung und Forschung; Horizon 2020 Framework Programme; Deutsche Forschungsgemeinschaft; Bộ Giáo dục và Ðào tạo; Banting and Best Diabetes Centre, University of Toronto; Schweizerischer Nationalfonds zur Förderung der Wissenschaftlichen Forschung; European Commission","keywords":"Computer science; Programming language","score_opus":0.0153390283246302,"score_gpt":0.2661524133120138,"score_spread":0.2508133849873836,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2781839508","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0071817786,0.25023088,0.15329319,0.1502392,0.013005904,0.0001304652,0.0013443341,0.0019819806,0.42259222],"genre_scores_gemma":[0.1423594,0.2749858,0.13302414,0.036449756,0.02069059,0.00055344205,0.0031397818,0.0016066408,0.38719055],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9992637,0.0002587242,0.00006191192,0.0000828733,0.00025211603,0.000080710095],"domain_scores_gemma":[0.99775946,0.0014269784,0.00008661526,0.00021679903,0.00035614547,0.00015404745],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015594527,0.0005986896,0.00049431523,0.0026937977,0.0013991499,0.008710639,0.0011460225,0.0019189479,0.018658986],"category_scores_gemma":[0.0038309218,0.00028232596,0.00062103046,0.002697514,0.004009556,0.019390142,0.003645618,0.0039667324,0.007921793],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000015301755,0.0000145411195,0.00008933428,0.00044889541,0.00000884816,0.00015167026,0.00087086775,0.00019811928,0.0014083424,0.76414734,0.1163507,0.11629589],"study_design_scores_gemma":[0.0000040992973,0.000007817981,0.00009703582,0.00030040264,0.0000056440967,0.0003268678,0.0007261776,0.00041737454,0.00045522713,0.3028063,0.69484234,0.000010812699],"about_ca_topic_score_codex":0.0012463869,"about_ca_topic_score_gemma":0.0017715135,"teacher_disagreement_score":0.018658986,"about_ca_system_score_codex":0.0016897097,"about_ca_system_score_gemma":0.0014162442,"threshold_uncertainty_score":0.062420547},"labels":[],"label_agreement":null},{"id":"W2782126927","doi":"10.21494/iste.op.2018.0206","title":"Les espaces sémantiques de mots-clés : une méthode d’indexation automatique de documents par assignation de mots-clés","year":2017,"lang":"fr","type":"article","venue":"Recherche d’information document et web sémantique","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Indexation; Humanities; Art; Economics","score_opus":0.08990668803295894,"score_gpt":0.3943007640468393,"score_spread":0.30439407601388035,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2782126927","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00871557,0.0004261053,0.9775613,0.0005219731,0.00017733329,0.00014852644,0.0010430238,0.007888484,0.0035177923],"genre_scores_gemma":[0.072550684,0.00040737298,0.9088482,0.00021495871,0.00011205909,0.00026333274,0.0017769121,0.0022050852,0.013621402],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99586064,0.00087369146,0.0004300261,0.0009901316,0.001645093,0.00020051753],"domain_scores_gemma":[0.9931017,0.0027387391,0.000437387,0.0015797426,0.0019387935,0.00020354007],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037163906,0.0013954531,0.0012075262,0.0044815335,0.0018950809,0.006227604,0.0022210819,0.0016159005,0.009088944],"category_scores_gemma":[0.013444062,0.000988668,0.0017754455,0.005502191,0.0020186421,0.005091418,0.0026340291,0.0023806193,0.0058897533],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00058926735,0.00017909847,0.0031927254,0.0007869165,0.0001804863,0.00024940146,0.0025332354,0.010434729,0.041950546,0.055702277,0.024759304,0.85944206],"study_design_scores_gemma":[0.00025011456,0.00023930236,0.007387101,0.00036546992,0.00028814824,0.0014489894,0.0017279629,0.4824096,0.17142585,0.10315497,0.2308664,0.00043608528],"about_ca_topic_score_codex":0.014065703,"about_ca_topic_score_gemma":0.015622443,"teacher_disagreement_score":0.014065703,"about_ca_system_score_codex":0.0017016935,"about_ca_system_score_gemma":0.0027522314,"threshold_uncertainty_score":0.030405521},"labels":[],"label_agreement":null},{"id":"W2782149963","doi":"","title":"LibGuides: Vancouver Referencing Style @ UCT: Reference samples","year":2014,"lang":"en","type":"libguides","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Style (visual arts); Computer science; Artificial intelligence; Geography; Archaeology","score_opus":0.03443742083059551,"score_gpt":0.2812926282621876,"score_spread":0.2468552074315921,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2782149963","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0030382166,0.0015479854,0.06348834,0.0022931586,0.0026463342,0.0003229768,0.19602205,0.17786345,0.5527775],"genre_scores_gemma":[0.01977678,0.002256975,0.061200045,0.0007910066,0.0006997549,0.000498732,0.36220732,0.17639054,0.3761789],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99741083,0.00030257355,0.00020705258,0.0003799363,0.0015058024,0.00019396182],"domain_scores_gemma":[0.9932045,0.00082589936,0.00014846436,0.00204121,0.003464553,0.0003153553],"candidate_categories":["scholarly_communication","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0013916979,0.0022928633,0.0015687787,0.010119563,0.0028391923,0.0071051763,0.0033269501,0.0019104215,0.48006183],"category_scores_gemma":[0.014161365,0.0014362705,0.0010802565,0.015901549,0.0009321857,0.0051410403,0.0039614253,0.002814558,0.42555252],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000049627968,0.000019667732,0.00012871597,0.00017145612,0.000007000291,0.00003791662,0.000105282175,0.0001504872,0.0004111379,0.00410617,0.9389075,0.055905063],"study_design_scores_gemma":[0.000024718842,0.000010133917,0.000357206,0.0001304926,0.000010199986,0.00011320226,0.00010661593,0.0010115909,0.0026379658,0.006454267,0.9891095,0.000034127166],"about_ca_topic_score_codex":0.048613414,"about_ca_topic_score_gemma":0.08239108,"teacher_disagreement_score":0.9928948,"about_ca_system_score_codex":0.0025913834,"about_ca_system_score_gemma":0.004539313,"threshold_uncertainty_score":0.74162865},"labels":[],"label_agreement":null},{"id":"W2782633305","doi":"10.1017/cem.2017.430","title":"A New Chapter for <i><b>CJEM</b></i>","year":2018,"lang":"pl","type":"editorial","venue":"Canadian Journal of Emergency Medicine","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary; Ottawa Hospital; Dalhousie University; Saint John Regional Hospital; University of Ottawa","funders":"","keywords":"Medicine; Content (measure theory); Action (physics); World Wide Web; Information retrieval; Computer science; Mathematics","score_opus":0.05662916149963672,"score_gpt":0.3482965629787479,"score_spread":0.2916674014791112,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2782633305","genre_codex":"editorial","genre_gemma":"editorial","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"editorial","genre_consensus":"editorial","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.000011217037,0.004015168,0.00013216744,0.032151412,0.96246356,0.000010666444,0.000037874117,0.000041564697,0.0011363211],"genre_scores_gemma":[0.00013682435,0.0034820007,0.00017076886,0.021007828,0.9665886,0.000021502481,0.000045562276,0.00006485197,0.008482049],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9919864,0.0010954447,0.0010748507,0.0007922977,0.004494536,0.00055659446],"domain_scores_gemma":[0.9494846,0.018707426,0.003008275,0.0013137287,0.020563781,0.006922154],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008841858,0.0033483477,0.003525217,0.0088665765,0.0046116905,0.012654334,0.0029768634,0.014147199,0.034290284],"category_scores_gemma":[0.042695276,0.001114271,0.0033986,0.0037764395,0.0039255074,0.005431506,0.0028678144,0.026134817,0.020754756],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000010282039,0.0000048941724,0.000007262452,0.00010170496,0.000005080817,0.00002502321,0.000006383429,0.000006489874,0.000021316668,0.00018009738,0.9960247,0.0036067083],"study_design_scores_gemma":[0.000030219502,0.000013492653,0.00012306913,0.0003569137,0.000024769946,0.00012456329,0.000033557746,0.00005851224,0.00004611498,0.0010260295,0.99814725,0.000015498228],"about_ca_topic_score_codex":0.004521535,"about_ca_topic_score_gemma":0.016110642,"teacher_disagreement_score":0.034290284,"about_ca_system_score_codex":0.004850429,"about_ca_system_score_gemma":0.008319621,"threshold_uncertainty_score":0.11471242},"labels":[],"label_agreement":null},{"id":"W2782848721","doi":"10.1515/ling-2017-0032","title":"Reproducible research in linguistics: A position statement on data citation and attribution in our field","year":2017,"lang":"en","type":"article","venue":"Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":172,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"National Science Foundation","keywords":"Attribution; Linguistics; Statement (logic); Citation; Field (mathematics); Applied linguistics; Quantitative linguistics; Authorship attribution; Position statement; Clinical linguistics; Psychology; Computer science; Sociology; Philosophy; Social psychology; Library science; Medicine","score_opus":0.18682001548247734,"score_gpt":0.4697376448389886,"score_spread":0.28291762935651127,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2782848721","genre_codex":"commentary","genre_gemma":"commentary","domain_codex":null,"domain_gemma":"reproducibility","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"commentary","genre_consensus":"commentary","domain_candidate":"reproducibility","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00050865597,0.004595578,0.026413327,0.90869933,0.05399496,0.00013873051,0.00011911522,0.00026714776,0.0052631265],"genre_scores_gemma":[0.0830136,0.014681828,0.118276596,0.55096275,0.20411709,0.0028595997,0.0006792421,0.0022234935,0.023185944],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.42414463,0.2677759,0.093188375,0.03858541,0.16420946,0.01209629],"domain_scores_gemma":[0.08779371,0.5594523,0.060614813,0.1008102,0.16545716,0.025871834],"candidate_categories":["metaresearch","open_science"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.5745385,0.0024990458,0.007002002,0.013322383,0.028915191,0.0797623,0.013923496,0.09473223,0.0064599235],"category_scores_gemma":[0.7133316,0.0032559156,0.0042127846,0.01796038,0.09172913,0.056500528,0.032720536,0.10039718,0.008033013],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000086423184,0.00006725691,0.0007726867,0.0006451564,0.000111146444,0.00025052362,0.0038941354,0.00042353463,0.00041709244,0.6334046,0.32461333,0.035313983],"study_design_scores_gemma":[0.00009334875,0.00007585415,0.00042396138,0.0029823892,0.00011020769,0.00024501927,0.001318174,0.002043078,0.0012670271,0.41271943,0.57836336,0.00035809973],"about_ca_topic_score_codex":0.005914455,"about_ca_topic_score_gemma":0.0043316972,"teacher_disagreement_score":0.9860765,"about_ca_system_score_codex":0.027429618,"about_ca_system_score_gemma":0.09346701,"threshold_uncertainty_score":0.52466977},"labels":[],"label_agreement":null},{"id":"W2782900468","doi":"10.5430/air.v7n1p23","title":"Combining Information Extraction and Text Segmentation methods in Greek Texts","year":2018,"lang":"en","type":"article","venue":"Artificial Intelligence Research","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Segmentation; Computer science; Information extraction; Natural language processing; Artificial intelligence; Information retrieval; Resolution (logic); Text segmentation; Extraction (chemistry); Pattern recognition (psychology)","score_opus":0.15148678799638132,"score_gpt":0.5122084057036119,"score_spread":0.3607216177072306,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2782900468","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17145409,0.0074390904,0.791701,0.0010604858,0.00040682845,0.00063521625,0.0017141548,0.00948768,0.016101442],"genre_scores_gemma":[0.22247665,0.0018373899,0.76528656,0.00018148955,0.000141905,0.00022821121,0.0034358767,0.0007530267,0.00565899],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99695206,0.0012630079,0.0003433501,0.00079917145,0.0005219511,0.00012042107],"domain_scores_gemma":[0.9903916,0.0068348874,0.00052842667,0.0007550581,0.0013851481,0.00010481008],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003061984,0.001225236,0.0010024356,0.005732377,0.0009765746,0.0022730832,0.0006617022,0.0012292824,0.0024025047],"category_scores_gemma":[0.010751608,0.00038116236,0.0008952027,0.0052744043,0.0007562998,0.0034054352,0.0010442074,0.0007580749,0.00260441],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00056307786,0.00013862722,0.0027147033,0.0020089746,0.00018384102,0.0008633395,0.0030665821,0.005783893,0.12675284,0.0039830366,0.003782297,0.85015875],"study_design_scores_gemma":[0.0001918495,0.0011492455,0.038542975,0.0009732205,0.0013659581,0.0040874905,0.005202039,0.12876774,0.59445256,0.023291368,0.20148997,0.00048568926],"about_ca_topic_score_codex":0.0015674118,"about_ca_topic_score_gemma":0.0026424697,"teacher_disagreement_score":0.005732377,"about_ca_system_score_codex":0.00062179123,"about_ca_system_score_gemma":0.0008611604,"threshold_uncertainty_score":0.01619351},"labels":[],"label_agreement":null},{"id":"W2783006698","doi":"10.3765/salt.v27i0.4144","title":"Ambiguous than-clauses and the mention-some reading","year":2017,"lang":"en","type":"article","venue":"Proceedings from Semantics and Linguistic Theory","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Ambiguity; Sentence; Negation; Context (archaeology); Computer science; Reading (process); Semantics (computer science); Linguistics; Operator (biology); Meaning (existential); Modal; Arithmetic; Mathematics; Natural language processing; Programming language; Philosophy","score_opus":0.009176302158488794,"score_gpt":0.25291598597136195,"score_spread":0.24373968381287314,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2783006698","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.36765665,0.0027139387,0.44503948,0.0059753847,0.00044272278,0.000115752904,0.00083536544,0.00091692404,0.17630377],"genre_scores_gemma":[0.9830919,0.0001970934,0.013450014,0.00037882084,0.00011682031,0.000022236514,0.00025193943,0.00014156803,0.0023496107],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99725974,0.0011300221,0.00020849911,0.0005747195,0.0005890881,0.00023789759],"domain_scores_gemma":[0.99576914,0.002603967,0.0003791066,0.00057557004,0.0005532538,0.000118974436],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030649228,0.0008200223,0.00067664875,0.001238936,0.0019473435,0.0030587472,0.0015213767,0.0021092773,0.0067576324],"category_scores_gemma":[0.0059210574,0.00076530874,0.0008082692,0.0010347768,0.0061149397,0.01197872,0.004044981,0.0030275225,0.00058649835],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000057778787,0.000013043873,0.00073180214,0.00010199831,0.0000129458585,0.00031091014,0.0075768614,0.00033293376,0.0027865225,0.9804857,0.0005942135,0.0069952793],"study_design_scores_gemma":[0.000040851337,0.00006792585,0.0024846082,0.000142527,0.00007178233,0.0013863953,0.0061948732,0.0042781467,0.0101669375,0.9296061,0.045464948,0.000094887415],"about_ca_topic_score_codex":0.0008487323,"about_ca_topic_score_gemma":0.00062091235,"teacher_disagreement_score":0.0067576324,"about_ca_system_score_codex":0.0011852336,"about_ca_system_score_gemma":0.0005125854,"threshold_uncertainty_score":0.022606552},"labels":[],"label_agreement":null},{"id":"W2783099878","doi":"10.3765/salt.v27i0.4113","title":"Negative polarity items: a case for questions as licensers","year":2017,"lang":"en","type":"article","venue":"Proceedings from Semantics and Linguistic Theory","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Syntax; Polarity (international relations); Virtue; Linguistics; Argument (complex analysis); Computer science; Psychology; Epistemology; Philosophy","score_opus":0.014674449103973995,"score_gpt":0.2989156968640085,"score_spread":0.2842412477600345,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2783099878","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.276314,0.0010659462,0.1615293,0.013599585,0.00028952313,0.00035948437,0.0004024295,0.0003797263,0.54605997],"genre_scores_gemma":[0.984445,0.00021844235,0.0069945036,0.000995162,0.00020564841,0.00013387694,0.00018000395,0.00017983184,0.006647674],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.98830926,0.005842385,0.00086178054,0.0016618511,0.0024320483,0.0008925942],"domain_scores_gemma":[0.96829695,0.021359408,0.0013639242,0.0045021954,0.0035659613,0.0009115916],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011015867,0.00092072936,0.0010406898,0.0030222496,0.004465797,0.010546463,0.0023620303,0.0073958207,0.019985791],"category_scores_gemma":[0.03490182,0.0010839326,0.0010331997,0.0018618725,0.01816822,0.042973187,0.011773426,0.0054866355,0.0021781318],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006728226,0.00003464038,0.0024196669,0.00007457041,0.0000080217005,0.000639421,0.013557364,0.000068725,0.0022992238,0.97125804,0.0006886846,0.008884272],"study_design_scores_gemma":[0.00007017297,0.00005646445,0.004332198,0.00015654191,0.00004333493,0.0029700727,0.012331847,0.0022849042,0.0031975794,0.9291482,0.045310024,0.00009871863],"about_ca_topic_score_codex":0.0011802416,"about_ca_topic_score_gemma":0.0007118559,"teacher_disagreement_score":0.019985791,"about_ca_system_score_codex":0.0016848993,"about_ca_system_score_gemma":0.0009708136,"threshold_uncertainty_score":0.066859126},"labels":[],"label_agreement":null},{"id":"W2783291346","doi":"10.63317/3e7ocg3pz9e6","title":"Manual vs Automatic Bitext Extraction","year":2018,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Computer science; Artificial intelligence; Extraction (chemistry); Speech recognition; Natural language processing; Pattern recognition (psychology); Chromatography","score_opus":0.011290672237174665,"score_gpt":0.3104030726975384,"score_spread":0.29911240046036375,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2783291346","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5407723,0.008958031,0.28915668,0.0030371554,0.0038252294,0.0013328901,0.03117313,0.032658227,0.089086354],"genre_scores_gemma":[0.7379522,0.0025806157,0.18156624,0.0008297354,0.00072125933,0.00039851823,0.036515262,0.002831857,0.036604304],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9956891,0.001656478,0.00052483735,0.0006309696,0.0010660134,0.0004325682],"domain_scores_gemma":[0.98984295,0.006753612,0.0003431287,0.0018363727,0.0010752303,0.00014879464],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003959907,0.00095754664,0.0010060784,0.0028650423,0.00081773644,0.0033747372,0.0013119214,0.0016692502,0.023115879],"category_scores_gemma":[0.0151235815,0.00048601744,0.0010035923,0.0019515202,0.0005069845,0.0037227557,0.0018843863,0.0014952616,0.00945732],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.005040477,0.00063391554,0.007810285,0.001403786,0.00031761575,0.00053506164,0.00024569998,0.004427084,0.05217815,0.00797875,0.043286815,0.8761423],"study_design_scores_gemma":[0.001789045,0.0032916605,0.09179099,0.0012333154,0.0022062175,0.0069296937,0.0023120223,0.30523235,0.38362312,0.04626388,0.15489593,0.00043177887],"about_ca_topic_score_codex":0.0012292326,"about_ca_topic_score_gemma":0.0020976579,"teacher_disagreement_score":0.023115879,"about_ca_system_score_codex":0.00032566302,"about_ca_system_score_gemma":0.00087542186,"threshold_uncertainty_score":0.07733029},"labels":[],"label_agreement":null},{"id":"W2785621828","doi":"","title":"Lexfom: a lexical functions ontology model","year":2016,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; Université du Québec","funders":"","keywords":"Syntagmatic analysis; Computer science; Lexical item; Natural language processing; Lexical density; Lexical grammar; Lexical choice; Lexical functional grammar; Artificial intelligence; Relation (database); Function (biology); Ontology; Linguistics; Representation (politics); Perspective (graphical); Generative grammar; Phrase structure rules","score_opus":0.0599011662098063,"score_gpt":0.34416787995760867,"score_spread":0.2842667137478024,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2785621828","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008918064,0.0004546549,0.9370622,0.001449426,0.00014755853,0.0004883497,0.014382207,0.017548842,0.019548643],"genre_scores_gemma":[0.12823229,0.0009894954,0.8105661,0.0009022403,0.00012782236,0.0014590529,0.03350372,0.0031998213,0.02101947],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9992539,0.00014803081,0.000117217096,0.00018536262,0.00021961151,0.000075787866],"domain_scores_gemma":[0.9994142,0.00015519239,0.00004883145,0.00017544912,0.00016235864,0.00004393039],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010655649,0.001079211,0.00067672296,0.0033197762,0.0012600374,0.0038797173,0.0025883017,0.0016483237,0.012869513],"category_scores_gemma":[0.003029662,0.00088040996,0.0023644397,0.0021670607,0.00095528405,0.008334832,0.0022469864,0.0020781197,0.005042222],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023571971,0.00026041528,0.0029019113,0.0007968387,0.00014899173,0.00065555546,0.0013348218,0.026373891,0.006025353,0.67964894,0.053366415,0.22825114],"study_design_scores_gemma":[0.000078639205,0.000057818317,0.0009803739,0.00028051002,0.000120520155,0.00057835056,0.00042346134,0.1992463,0.0057584317,0.29769707,0.49468932,0.00008913242],"about_ca_topic_score_codex":0.014732557,"about_ca_topic_score_gemma":0.014305199,"teacher_disagreement_score":0.014732557,"about_ca_system_score_codex":0.0027805679,"about_ca_system_score_gemma":0.0032134494,"threshold_uncertainty_score":0.043052793},"labels":[],"label_agreement":null},{"id":"W2786622573","doi":"","title":"Korean-English MT and S-TAG","year":2002,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Linguistics; History; Computer science; Philosophy","score_opus":0.011819525066339405,"score_gpt":0.22587104277349107,"score_spread":0.21405151770715167,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2786622573","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012915191,0.0059777275,0.82156116,0.0032428033,0.0008689915,0.0002949247,0.0020350388,0.0057561905,0.14734802],"genre_scores_gemma":[0.3281745,0.005204822,0.57617414,0.0015848838,0.0004015963,0.00040597073,0.004085008,0.0015130423,0.08245608],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9995141,0.00021139822,0.000045282886,0.0001244035,0.000068650326,0.000036167283],"domain_scores_gemma":[0.9989272,0.0003452667,0.00008876171,0.00042968622,0.00017413439,0.000035063764],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012984222,0.0006030397,0.00041121824,0.0016396998,0.0009815533,0.001710979,0.0007221572,0.0010424451,0.017698582],"category_scores_gemma":[0.00322244,0.00057766074,0.00041094,0.002702456,0.0008646198,0.006203788,0.0019767052,0.0011433712,0.0137907015],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001350293,0.00010088153,0.001373176,0.00054227666,0.000074963,0.0005955,0.0008305014,0.0032867258,0.0036018512,0.45244008,0.055083983,0.48193505],"study_design_scores_gemma":[0.0000533457,0.0000669338,0.0029061672,0.00019067245,0.00006929329,0.0013312497,0.0005355717,0.047813285,0.010474347,0.4726159,0.46386322,0.00008003083],"about_ca_topic_score_codex":0.0047383094,"about_ca_topic_score_gemma":0.0064123645,"teacher_disagreement_score":0.017698582,"about_ca_system_score_codex":0.00074685126,"about_ca_system_score_gemma":0.0009710366,"threshold_uncertainty_score":0.059207678},"labels":[],"label_agreement":null},{"id":"W2787153175","doi":"","title":"Two Uummarmiutun modals – including a brief comparison with Utkuhikšalingmiutut cognates","year":2017,"lang":"en","type":"article","venue":"PhilPapers (PhilPapers Foundation)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Linguistics; Modal verb; Meaning (existential); Modal; Cognate; Cognitive linguistics; Psychology; Grammaticalization; History; Cognition; Philosophy; Verb","score_opus":0.037060430696843616,"score_gpt":0.3359573929679526,"score_spread":0.29889696227110896,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2787153175","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8213957,0.0029762823,0.005882385,0.00058672973,0.00009830414,0.00007692726,0.00027511193,0.00012042512,0.1685881],"genre_scores_gemma":[0.99361515,0.00042685785,0.0020016187,0.000049902665,0.00000820078,0.000028909391,0.00013796634,0.000044658213,0.0036868316],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"qualitative","domain_scores_codex":[0.9995043,0.00008996373,0.000051002957,0.000099868084,0.00011661686,0.0001382355],"domain_scores_gemma":[0.9997551,0.00008288938,0.00004933571,0.00002713072,0.000062480714,0.000023133041],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003777285,0.00046363237,0.0003586626,0.0014104559,0.0031918956,0.0023490486,0.0005170987,0.0006555198,0.0041966904],"category_scores_gemma":[0.00087176723,0.00031060222,0.00027272475,0.001833999,0.0035595326,0.0021422643,0.0020705243,0.000807766,0.00030602972],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000551704,0.00008343788,0.031748213,0.000902704,0.00006826087,0.003931032,0.4059938,0.00038641712,0.027716229,0.4356191,0.0020358092,0.09096333],"study_design_scores_gemma":[0.000039893937,0.0002534505,0.19140726,0.0007136268,0.00026027512,0.012040733,0.4167204,0.0025151726,0.026803654,0.035953887,0.31301934,0.00027233714],"about_ca_topic_score_codex":0.05667761,"about_ca_topic_score_gemma":0.09460275,"teacher_disagreement_score":0.05667761,"about_ca_system_score_codex":0.003558655,"about_ca_system_score_gemma":0.0014901438,"threshold_uncertainty_score":0.112695396},"labels":[],"label_agreement":null},{"id":"W2787277033","doi":"10.26754/ojs_misc/mj.20176814","title":"A data-driven learning experiment in the legal English classroom using the FLAX platform","year":2017,"lang":"en","type":"article","venue":"Miscelánea A Journal of English and American Studies","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Lexicon; Terminology; Exploit; Computer science; Natural language processing; Artificial intelligence; Group (periodic table); Control (management); Contrast (vision); Lexical diversity; Linguistics; Task (project management); Term (time); English for specific purposes; Psychology; Mathematics education; Engineering","score_opus":0.05413030168228262,"score_gpt":0.3450025758273006,"score_spread":0.290872274145018,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2787277033","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99507195,0.000014230172,0.0018184438,0.000050846753,0.000018742476,0.0016154043,0.00038713228,0.00009517261,0.0009281216],"genre_scores_gemma":[0.9606714,0.00006552583,0.024287144,0.00021085232,0.00004367826,0.006380754,0.0021270933,0.0000748704,0.0061386256],"study_design_codex":"nonrandomized_trial","study_design_gemma":"observational","domain_scores_codex":[0.9963201,0.0016182013,0.00030606566,0.00084113935,0.0006587647,0.00025562258],"domain_scores_gemma":[0.97502214,0.01646626,0.0014441332,0.00224256,0.0031014662,0.001723483],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0056394613,0.0007617358,0.0011351282,0.0007037679,0.0011027231,0.0014699908,0.0015997544,0.0014440791,0.0035944514],"category_scores_gemma":[0.01724521,0.000447117,0.0005252483,0.0005298435,0.0012188631,0.0014899386,0.0013442265,0.0014314756,0.0013799729],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.039872844,0.3110721,0.07196469,0.0029382892,0.00035899377,0.0023909411,0.09195436,0.01884776,0.19135903,0.0037228959,0.0061197113,0.25939846],"study_design_scores_gemma":[0.014468291,0.2953094,0.28386337,0.00046737024,0.00058732356,0.0010045446,0.041122686,0.04675654,0.247099,0.009660103,0.058668125,0.0009933261],"about_ca_topic_score_codex":0.001663252,"about_ca_topic_score_gemma":0.0021682614,"teacher_disagreement_score":0.0056394613,"about_ca_system_score_codex":0.0010103971,"about_ca_system_score_gemma":0.0011312546,"threshold_uncertainty_score":0.029824674},"labels":[],"label_agreement":null},{"id":"W2787587581","doi":"","title":"A Proposal for combining “general” and specialized frames","year":2016,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"FrameNet; Merge (version control); Computer science; Domain (mathematical analysis); Resource (disambiguation); Semantics (computer science); Natural language processing; Representation (politics); Artificial intelligence; Information retrieval; Programming language; Parsing; Mathematics","score_opus":0.034748774668634734,"score_gpt":0.33961479867365135,"score_spread":0.3048660240050166,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2787587581","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004866407,0.00026347817,0.96711975,0.0016262844,0.00027633377,0.00015923071,0.000146613,0.00074069184,0.024801165],"genre_scores_gemma":[0.13194166,0.00032835366,0.8545499,0.000908018,0.00039772197,0.00042963578,0.00055257103,0.00055476493,0.01033742],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99415827,0.0022435223,0.00048807912,0.0017728219,0.00077096676,0.0005664863],"domain_scores_gemma":[0.99515027,0.0012392352,0.0003112397,0.0020190964,0.0008269261,0.00045316305],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006954605,0.001591776,0.0013275703,0.004524835,0.003136967,0.006902109,0.0046980553,0.003696007,0.012064948],"category_scores_gemma":[0.009133885,0.0013727759,0.002319356,0.004154444,0.007219665,0.026086394,0.008903231,0.0040552625,0.0029582474],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000025436231,0.000014825253,0.00023055947,0.00005283504,0.000014457953,0.000066386056,0.00091964705,0.00049752643,0.0007604071,0.9716713,0.0019860237,0.023760637],"study_design_scores_gemma":[0.00003394626,0.00007348066,0.00036333702,0.00016004835,0.00008745105,0.0004133041,0.0012921329,0.01706614,0.0023929859,0.81186587,0.16618837,0.000062989886],"about_ca_topic_score_codex":0.0048107454,"about_ca_topic_score_gemma":0.0050713345,"teacher_disagreement_score":0.012064948,"about_ca_system_score_codex":0.0029235221,"about_ca_system_score_gemma":0.0035683927,"threshold_uncertainty_score":0.040361226},"labels":[],"label_agreement":null},{"id":"W2787737401","doi":"","title":"Description of the Homebank Child/Adult Addressee Corpus (HB-CHAAC).","year":2017,"lang":"en","type":"article","venue":"Conference of the International Speech Communication Association","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba","funders":"","keywords":"Computer science; Natural language processing; Linguistics; Speech recognition; Artificial intelligence; Philosophy","score_opus":0.03006516855866038,"score_gpt":0.2780974164276234,"score_spread":0.24803224786896305,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2787737401","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0069652554,0.0008576934,0.004988508,0.00037152207,0.00035673837,0.0006658372,0.9645161,0.0034464092,0.017831953],"genre_scores_gemma":[0.0066012037,0.00023789672,0.007931275,0.00017163024,0.000075454285,0.0012061711,0.97864944,0.0008416008,0.004285385],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9987,0.000283305,0.00023449055,0.00043021332,0.00024147076,0.000110401445],"domain_scores_gemma":[0.99766076,0.0006541464,0.0001287325,0.0004579668,0.00085019297,0.00024818332],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012981787,0.0017172267,0.0011410215,0.006586926,0.001998093,0.0024251,0.0018694086,0.0014373548,0.112173274],"category_scores_gemma":[0.0030399067,0.00089333626,0.00057758205,0.0068379464,0.00083170965,0.001637037,0.003091621,0.0021273931,0.08400889],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00047305954,0.00025275335,0.0022206868,0.0025500297,0.000059591308,0.0006487863,0.00072059396,0.0004721991,0.014815002,0.0037327656,0.93061376,0.043440834],"study_design_scores_gemma":[0.00031154868,0.00007107376,0.014349115,0.00034560298,0.000065582884,0.001230951,0.0007640599,0.0010849579,0.0064267074,0.0017157713,0.9735454,0.00008917988],"about_ca_topic_score_codex":0.023411853,"about_ca_topic_score_gemma":0.036818285,"teacher_disagreement_score":0.112173274,"about_ca_system_score_codex":0.0010127944,"about_ca_system_score_gemma":0.0029451656,"threshold_uncertainty_score":0.37525702},"labels":[],"label_agreement":null},{"id":"W2789336736","doi":"10.3390/languages3010006","title":"On Recursive Modification in Child L1 French","year":2018,"lang":"en","type":"article","venue":"Languages","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria; University of Toronto","funders":"","keywords":"Recursion (computer science); Merge (version control); Computer science; Universal grammar; Universality (dynamical systems); Grammar; Linguistics; Embedding; Schema (genetic algorithms); Minimalist program; Theoretical computer science; Syntax; Artificial intelligence; Programming language; Generative grammar","score_opus":0.009840454882455323,"score_gpt":0.29384800837508196,"score_spread":0.28400755349262663,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2789336736","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9859811,0.0007026266,0.0006825435,0.00029750797,0.000007628984,0.000012857118,0.00038734198,0.000067660156,0.011860822],"genre_scores_gemma":[0.99536717,0.0004989162,0.0007850779,0.00019810835,0.000006566672,0.000027695225,0.00045232306,0.00004299079,0.0026212],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9986268,0.0003914904,0.00006833182,0.0002852508,0.00040087104,0.00022724443],"domain_scores_gemma":[0.9940253,0.0033125672,0.0011985664,0.00044467126,0.000843728,0.00017523688],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020639806,0.00055844535,0.00064697204,0.0015516292,0.000781886,0.001856648,0.00048681517,0.0010087456,0.008003206],"category_scores_gemma":[0.007227958,0.00035657728,0.00035370624,0.00081131957,0.0021665783,0.0019712406,0.0013484246,0.0009823148,0.0015050403],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008972866,0.00059490185,0.48660478,0.0010779013,0.00013150799,0.0106956065,0.22614835,0.0014669525,0.0540736,0.018439848,0.004729536,0.19513972],"study_design_scores_gemma":[0.000038154707,0.00071325724,0.90282536,0.00033741695,0.00009402681,0.009153781,0.02977443,0.0012263752,0.0107780835,0.0044839354,0.040391352,0.00018384465],"about_ca_topic_score_codex":0.039192647,"about_ca_topic_score_gemma":0.044956584,"teacher_disagreement_score":0.039192647,"about_ca_system_score_codex":0.0016233874,"about_ca_system_score_gemma":0.0008714163,"threshold_uncertainty_score":0.07792902},"labels":[],"label_agreement":null},{"id":"W2789616329","doi":"10.3115/1289189.1289231","title":"TransType","year":2002,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Context (archaeology); Machine translation; Artificial intelligence; Translation (biology); Natural language processing; Human–computer interaction; Machine learning","score_opus":0.017381839144219827,"score_gpt":0.24032465487210647,"score_spread":0.22294281572788666,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2789616329","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018095342,0.0011163047,0.42463297,0.0051512513,0.0045178384,0.0007701661,0.08213911,0.1509555,0.3126215],"genre_scores_gemma":[0.1764854,0.0020649135,0.20347966,0.0070312396,0.0018605775,0.0014025263,0.15653516,0.09540114,0.35573933],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99742717,0.0005901908,0.00031520848,0.0006327096,0.00077640254,0.00025827702],"domain_scores_gemma":[0.9925419,0.001984033,0.0003603188,0.0030234577,0.0018539398,0.00023632075],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022844796,0.0014692903,0.0009286374,0.0018747591,0.0017928062,0.0031280378,0.0025622721,0.002047629,0.13196081],"category_scores_gemma":[0.008188708,0.00086988736,0.0012897551,0.0023778758,0.0011293731,0.008046129,0.0049649724,0.002417212,0.10479527],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010672249,0.00017918262,0.0052784323,0.000932971,0.00008750337,0.0012974681,0.0019014548,0.0015046365,0.011081862,0.15567519,0.6195033,0.20149072],"study_design_scores_gemma":[0.000037811835,0.00003984185,0.0006224862,0.000096052056,0.000024615192,0.00047985927,0.00025448605,0.0025512055,0.007026428,0.024943072,0.9638701,0.000053845597],"about_ca_topic_score_codex":0.003973079,"about_ca_topic_score_gemma":0.0054989094,"teacher_disagreement_score":0.13196081,"about_ca_system_score_codex":0.0014176044,"about_ca_system_score_gemma":0.0014875308,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W2789803393","doi":"10.21248/zaspil.61.2018.501","title":"Decomposing universal projection in questions","year":2018,"lang":"en","type":"article","venue":"ZAS Papers in Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Presupposition; Projection (relational algebra); Typology; Epistemology; Mathematics; Computer science; Sociology; Philosophy; Algorithm","score_opus":0.009895379209969076,"score_gpt":0.29313301540278397,"score_spread":0.2832376361928149,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2789803393","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.27681446,0.0011679726,0.6395804,0.002700924,0.00013135093,0.00019748804,0.0004452096,0.0009413239,0.07802089],"genre_scores_gemma":[0.95656234,0.0002714275,0.039836958,0.00026681353,0.000081013,0.000105406536,0.0003783264,0.0001400842,0.0023577544],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9971877,0.00096697424,0.00020651407,0.0007628993,0.00048865075,0.00038731215],"domain_scores_gemma":[0.99663574,0.0016506118,0.00019504469,0.00096969056,0.00037609262,0.00017279903],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033817512,0.0009263663,0.00072283635,0.0015093308,0.0018567108,0.0034379894,0.0013445639,0.0023789892,0.008530592],"category_scores_gemma":[0.008008008,0.0009054375,0.001563644,0.0011472366,0.00799132,0.0170113,0.00877363,0.0030486393,0.00059820677],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000049664446,0.000023483597,0.0013357181,0.00008977902,0.00002443543,0.0004233787,0.00203555,0.0012660418,0.0020027738,0.9764997,0.00035409263,0.015895337],"study_design_scores_gemma":[0.00000878427,0.000014641464,0.00075533515,0.000019987026,0.000010032009,0.00022602773,0.0003453671,0.004259571,0.0010836255,0.990756,0.0025090578,0.0000114968325],"about_ca_topic_score_codex":0.0009649796,"about_ca_topic_score_gemma":0.0007297303,"teacher_disagreement_score":0.008530592,"about_ca_system_score_codex":0.0014264623,"about_ca_system_score_gemma":0.000897569,"threshold_uncertainty_score":0.028537631},"labels":[],"label_agreement":null},{"id":"W2790100769","doi":"10.3389/fpsyg.2017.02335","title":"Using Neural Networks to Generate Inferential Roles for Natural Language","year":2018,"lang":"en","type":"article","venue":"Frontiers in Psychology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Canada Research Chairs","keywords":"Computer science; Natural language processing; Syntax; Sentence; Semantics (computer science); Artificial intelligence; Set (abstract data type); Inference; Linguistics; Expression (computer science); Representation (politics); Logical consequence; Natural language","score_opus":0.019858011181402336,"score_gpt":0.3481746268889014,"score_spread":0.32831661570749904,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2790100769","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.23046081,0.0006974292,0.75079423,0.002391964,0.00015653654,0.00027292152,0.0021476832,0.0042960565,0.008782341],"genre_scores_gemma":[0.72272664,0.00027395747,0.26943606,0.00033831233,0.00006919843,0.00028257674,0.0040688044,0.00022415376,0.0025802476],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99919754,0.00037995394,0.00004245479,0.0002097947,0.00013141667,0.000038913367],"domain_scores_gemma":[0.9951609,0.0038837227,0.00022221127,0.00041271793,0.00025128067,0.000069247384],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020728132,0.0009871866,0.00031320023,0.0011316302,0.00047672787,0.0010456692,0.0015116854,0.0009868696,0.002623484],"category_scores_gemma":[0.0123828985,0.0004325948,0.00080436177,0.00061148306,0.0008440724,0.0030370785,0.0010052909,0.002123048,0.00060593535],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00052416656,0.00045517713,0.00822293,0.00045347953,0.00020042277,0.00055181293,0.0009456816,0.5936463,0.0122033395,0.08734429,0.012980279,0.28247213],"study_design_scores_gemma":[0.000013761371,0.000014936716,0.0002598148,0.000014204651,0.000008376597,0.000021207925,0.000025980777,0.96199745,0.0016729175,0.035178117,0.0007879069,0.000005325911],"about_ca_topic_score_codex":0.0039899014,"about_ca_topic_score_gemma":0.009349255,"teacher_disagreement_score":0.0039899014,"about_ca_system_score_codex":0.001657594,"about_ca_system_score_gemma":0.0006638802,"threshold_uncertainty_score":0.012026727},"labels":[],"label_agreement":null},{"id":"W2790484720","doi":"10.1109/ialp.2017.8300612","title":"Joint bi-affine parsing and semantic role labeling","year":2017,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Parsing; Joint (building); Natural language processing; Artificial intelligence","score_opus":0.019415394432935752,"score_gpt":0.2738629744058371,"score_spread":0.25444757997290135,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2790484720","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0049339305,0.0003298459,0.97835094,0.00041945744,0.00014290946,0.00011522502,0.0010041535,0.012186511,0.0025170064],"genre_scores_gemma":[0.26457357,0.00072960154,0.7077239,0.0008856001,0.00025830863,0.00044361013,0.008998351,0.0019686292,0.014418333],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9980804,0.0004993993,0.00011369143,0.0008027119,0.00033207875,0.00017172607],"domain_scores_gemma":[0.99726677,0.0008997888,0.00014958298,0.0010953881,0.0004501581,0.00013832345],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003109561,0.0024178573,0.0018083896,0.0017836222,0.0007559241,0.002481174,0.0047302884,0.0027163248,0.00792502],"category_scores_gemma":[0.007070009,0.0012583565,0.0025074387,0.0023276794,0.0013066367,0.006804708,0.0034643635,0.0051179277,0.008506126],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00046480715,0.00055029336,0.0032821875,0.00044172932,0.00026983477,0.0004480309,0.00027499755,0.14968655,0.016640654,0.046756204,0.04031765,0.7408671],"study_design_scores_gemma":[0.00002071768,0.000052234212,0.00040471146,0.000020960671,0.00007337319,0.00016628011,0.000031779968,0.9503206,0.008629049,0.033774953,0.006461109,0.00004427698],"about_ca_topic_score_codex":0.007197344,"about_ca_topic_score_gemma":0.015717078,"teacher_disagreement_score":0.00792502,"about_ca_system_score_codex":0.0013017011,"about_ca_system_score_gemma":0.0037372787,"threshold_uncertainty_score":0.026511908},"labels":[],"label_agreement":null},{"id":"W2790869732","doi":"10.18653/v1/w17-3602","title":"The Good, the Bad, and the Disagreement: Complex ground truth in rhetorical structure analysis","year":2017,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Rhetorical question; Annotation; Computer science; Representation (politics); Linguistics; Natural language processing; German; Epistemology; Artificial intelligence; Philosophy; Political science; Law","score_opus":0.019228887030872905,"score_gpt":0.2963445324012137,"score_spread":0.27711564537034084,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2790869732","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.072506376,0.0008678253,0.90799266,0.007250047,0.00018362289,0.00010350906,0.0002757947,0.00031193154,0.010508291],"genre_scores_gemma":[0.8130123,0.00029949506,0.18331921,0.0005541354,0.0002968031,0.00024330693,0.00042665383,0.0002792915,0.0015688364],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.9676413,0.019833965,0.0017121704,0.0037558633,0.0058777416,0.0011789401],"domain_scores_gemma":[0.9045132,0.07417348,0.0061190627,0.008543074,0.0054068505,0.0012443538],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.031397842,0.0011738356,0.0017612258,0.008761682,0.006647407,0.012055272,0.0033669532,0.0050818087,0.0034317004],"category_scores_gemma":[0.10106168,0.0011698512,0.0018074138,0.0067878836,0.021277925,0.028648918,0.007903769,0.0066028014,0.0006148296],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011189319,0.000037603044,0.0021358095,0.00015309996,0.00004372796,0.0002384411,0.008534163,0.0072037047,0.001042648,0.95137584,0.0012700813,0.02785298],"study_design_scores_gemma":[0.000008380752,0.000008385013,0.00035222858,0.00006563191,0.00001749758,0.000039476177,0.0007295395,0.0295773,0.00069621607,0.96642494,0.002058118,0.000022270842],"about_ca_topic_score_codex":0.002663769,"about_ca_topic_score_gemma":0.002233375,"teacher_disagreement_score":0.031397842,"about_ca_system_score_codex":0.004383622,"about_ca_system_score_gemma":0.002535526,"threshold_uncertainty_score":0.1660496},"labels":[],"label_agreement":null},{"id":"W2791176029","doi":"10.1016/j.neucom.2018.01.007","title":"Fine-grained attention mechanism for neural machine translation","year":2018,"lang":"en","type":"preprint","venue":"Neurocomputing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"Natural Sciences and Engineering Research Council of Canada; National Research Foundation of Korea; National Research Foundation; Compute Canada; Samsung; Canada Research Chairs; Canadian Institute for Advanced Research","keywords":"Machine translation; Computer science; Mechanism (biology); Translation (biology); Artificial intelligence; Context (archaeology); Word (group theory); Natural language processing; Exploit; Task (project management); Machine learning; Linguistics","score_opus":0.030560991087392766,"score_gpt":0.29747321397155,"score_spread":0.2669122228841573,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2791176029","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.060987346,0.0014186163,0.91931957,0.0013949381,0.0005382268,0.00007184536,0.00042178252,0.0042736465,0.011574042],"genre_scores_gemma":[0.87474257,0.0004625344,0.11284898,0.00039894297,0.00028817306,0.00008289781,0.0004777059,0.00022540643,0.010472791],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.999678,0.00008316146,0.000024718618,0.00010344273,0.00004524304,0.00006543248],"domain_scores_gemma":[0.99927694,0.00023800285,0.000045149307,0.00026393952,0.00013300755,0.00004288141],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007030204,0.00043704498,0.0006254075,0.0005362664,0.00055228465,0.0010777136,0.0011771304,0.0012936532,0.009364428],"category_scores_gemma":[0.0026828249,0.00029404627,0.00045079942,0.0007838256,0.00045783498,0.002300085,0.001341197,0.0014573233,0.0017434377],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006887857,0.00035881257,0.001116789,0.00034174262,0.00014295067,0.00033201135,0.0002490623,0.12412219,0.103035174,0.18863694,0.02524029,0.55573535],"study_design_scores_gemma":[0.00003334345,0.00008019446,0.0007228003,0.000012657148,0.000032020675,0.00010335749,0.000028043,0.81698143,0.01377552,0.16372937,0.004478055,0.00002321627],"about_ca_topic_score_codex":0.002176203,"about_ca_topic_score_gemma":0.0030851453,"teacher_disagreement_score":0.009364428,"about_ca_system_score_codex":0.0006431993,"about_ca_system_score_gemma":0.00079610985,"threshold_uncertainty_score":0.03132713},"labels":[],"label_agreement":null},{"id":"W2792194291","doi":"","title":"Multi-way classification of semantic relations between pairs of nominals","year":2009,"lang":"en","type":"article","venue":"North American Chapter of the Association for Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Linguistics; Information retrieval; Philosophy","score_opus":0.025962098988825722,"score_gpt":0.29546025094887,"score_spread":0.2694981519600443,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2792194291","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7173652,0.002325051,0.24110505,0.001283687,0.00065659237,0.0003618425,0.009777943,0.0025874535,0.024537196],"genre_scores_gemma":[0.8806752,0.00033110922,0.10620113,0.000075080476,0.000081331535,0.00014228624,0.008176471,0.00014456538,0.00417287],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99796766,0.00046024224,0.00028758228,0.00058942946,0.00049790705,0.00019727804],"domain_scores_gemma":[0.9942849,0.0024884585,0.00064647134,0.0007958736,0.0014001265,0.0003840893],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017581349,0.00052011554,0.00055176823,0.005344762,0.0016304796,0.0026246782,0.0010068527,0.0012036286,0.006085287],"category_scores_gemma":[0.0070311744,0.00023782252,0.0013446682,0.002873498,0.0009417874,0.005089556,0.002098062,0.0012484405,0.0021865675],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004125077,0.00064230204,0.13687682,0.0012360192,0.00040892628,0.0012798589,0.0043616723,0.0039723134,0.08478189,0.059603985,0.01590038,0.68681073],"study_design_scores_gemma":[0.00022087242,0.001165785,0.23310661,0.00093511457,0.001034435,0.0040039252,0.017154055,0.31689206,0.06904742,0.2690659,0.087013505,0.0003604827],"about_ca_topic_score_codex":0.0021031918,"about_ca_topic_score_gemma":0.0039110156,"teacher_disagreement_score":0.006085287,"about_ca_system_score_codex":0.0008489018,"about_ca_system_score_gemma":0.0010831269,"threshold_uncertainty_score":0.020357251},"labels":[],"label_agreement":null},{"id":"W2792210162","doi":"10.1145/3160488","title":"Expanding Paraphrase Lexicons by Exploiting Generalities","year":2018,"lang":"en","type":"article","venue":"ACM Transactions on Asian and Low-Resource Language Information Processing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"Japan Society for the Promotion of Science","keywords":"Paraphrase; Computer science; Natural language processing; Lexicon; Artificial intelligence; Leverage (statistics); Task (project management); Substitution (logic); Set (abstract data type); Semantic equivalence","score_opus":0.00916648332078006,"score_gpt":0.2578009485067484,"score_spread":0.24863446518596832,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2792210162","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.27230534,0.00091908226,0.70188504,0.00053734274,0.00006360601,0.00085396087,0.002438542,0.008368419,0.012628701],"genre_scores_gemma":[0.558216,0.00072819553,0.42632696,0.00028713877,0.000096267075,0.00041597313,0.009966584,0.0007309787,0.0032319785],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990109,0.00021765864,0.00011696941,0.0003412673,0.0002442394,0.000068933216],"domain_scores_gemma":[0.9969855,0.0013967538,0.00025113733,0.00081269623,0.00047425926,0.00007968773],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00078163494,0.0010448701,0.0007370647,0.0036301678,0.0005391437,0.0011936488,0.00135007,0.0006573736,0.0030725144],"category_scores_gemma":[0.005436913,0.0007088809,0.0012622386,0.0024482463,0.0006769381,0.0028379734,0.0019988585,0.0010078067,0.0017019211],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002627692,0.00047353157,0.011105793,0.0007540233,0.00021753438,0.0020017775,0.001645694,0.020035774,0.14222673,0.014664645,0.010117649,0.79649407],"study_design_scores_gemma":[0.00019527136,0.000760098,0.027483955,0.00023240148,0.0006779254,0.0062768427,0.0018243828,0.69529957,0.09870922,0.10718038,0.061186936,0.00017302742],"about_ca_topic_score_codex":0.0013093736,"about_ca_topic_score_gemma":0.00450177,"teacher_disagreement_score":0.0036301678,"about_ca_system_score_codex":0.0004732434,"about_ca_system_score_gemma":0.00070535217,"threshold_uncertainty_score":0.010278523},"labels":[],"label_agreement":null},{"id":"W2792717671","doi":"","title":"Abductive Logic Grammars","year":2009,"lang":"en","type":"article","venue":"RUCforsk (Roskilde University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Rewriting; Rule-based machine translation; Formalism (music); Sentence; Tree-adjoining grammar; Dynamic logic (digital electronics); Definite clause grammar; Context-sensitive grammar; L-attributed grammar; String (physics); Theoretical computer science; Artificial intelligence; Context-free grammar; Programming language; Mathematics","score_opus":0.0069089387676847455,"score_gpt":0.2105836513681577,"score_spread":0.20367471260047296,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2792717671","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0013785887,0.00025281156,0.9867508,0.0006791164,0.000053942975,0.00009667021,0.00040690752,0.0007556608,0.009625648],"genre_scores_gemma":[0.05253169,0.0005763823,0.9404269,0.00047799363,0.00008200577,0.00023632473,0.0010718236,0.00017023277,0.0044265613],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99687356,0.0010058285,0.00042084345,0.0006564138,0.000910525,0.00013294313],"domain_scores_gemma":[0.9946031,0.0034367526,0.0002751422,0.0010186456,0.0005729716,0.0000932238],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034661007,0.0010320162,0.0006950467,0.003216984,0.00124588,0.0038399254,0.0034407207,0.0013659085,0.0071812836],"category_scores_gemma":[0.009176332,0.0006844851,0.002160182,0.0029458322,0.004155498,0.004002146,0.003941655,0.0026408941,0.0015598076],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000020499194,0.000038246515,0.00032791143,0.00030754026,0.000058066507,0.00035238743,0.0005445073,0.022296559,0.0012631182,0.8747388,0.0045434954,0.095508866],"study_design_scores_gemma":[0.000025833418,0.0000101615715,0.000075575874,0.00008745035,0.000033885473,0.00016031931,0.000087350505,0.054845847,0.0020156354,0.9119993,0.030638205,0.00002050219],"about_ca_topic_score_codex":0.0034598638,"about_ca_topic_score_gemma":0.005148567,"teacher_disagreement_score":0.0071812836,"about_ca_system_score_codex":0.0017391847,"about_ca_system_score_gemma":0.0021002097,"threshold_uncertainty_score":0.024023771},"labels":[{"model":"gemma","categories":[],"domain":null,"study_design":"theoretical_or_conceptual","genre":"methods","about_ca_system":false,"about_ca_topic":false,"confidence":"low"},{"model":"gpt","categories":[],"domain":null,"study_design":"theoretical_or_conceptual","genre":"methods","about_ca_system":false,"about_ca_topic":false,"confidence":"medium"}],"label_agreement":"agree"},{"id":"W2793238213","doi":"10.1075/term.00002.gha","title":"Automatic extraction of specialized verbal units","year":2017,"lang":"en","type":"article","venue":"Terminology International Journal of Theoretical and Applied Issues in Specialized Communication","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal; Université de Montréal","funders":"","keywords":"Computer science; Terminology; Natural language processing; Domain (mathematical analysis); Extractor; Artificial intelligence; Perspective (graphical); Quality (philosophy); Field (mathematics); Arabic; Linguistics; Mathematics","score_opus":0.020295174790708838,"score_gpt":0.34768089114443035,"score_spread":0.3273857163537215,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2793238213","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13453002,0.0015621131,0.8108385,0.00045117325,0.00030294122,0.0010249303,0.008052108,0.0108483955,0.032389775],"genre_scores_gemma":[0.31725904,0.00083734933,0.6543647,0.00010889743,0.000115056246,0.0005642253,0.014918529,0.0013706656,0.010461515],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988857,0.00025394667,0.00015217101,0.0002313371,0.0003802841,0.000096568634],"domain_scores_gemma":[0.99684805,0.0012611075,0.00032571773,0.00041829888,0.0010773701,0.0000694278],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007307443,0.00088113506,0.00064530014,0.0048527317,0.00067136384,0.0020329908,0.0007347847,0.0005634704,0.00851536],"category_scores_gemma":[0.0060499273,0.00029696757,0.0006640089,0.003170982,0.0005507953,0.001776876,0.0013785707,0.00069681415,0.0047519878],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00030793098,0.00009403167,0.005324098,0.0019310687,0.00006968727,0.0012936805,0.0025873648,0.0018364766,0.12962657,0.037768617,0.017252866,0.8019076],"study_design_scores_gemma":[0.0001098003,0.00035981432,0.03179968,0.00096739887,0.00030876326,0.0040136892,0.006924302,0.110081285,0.29562786,0.074317284,0.4752579,0.00023230028],"about_ca_topic_score_codex":0.0009069338,"about_ca_topic_score_gemma":0.0009628917,"teacher_disagreement_score":0.00851536,"about_ca_system_score_codex":0.0004978091,"about_ca_system_score_gemma":0.001099914,"threshold_uncertainty_score":0.028486729},"labels":[],"label_agreement":null},{"id":"W2794627393","doi":"10.48550/arxiv.1803.11034","title":"Automatic Generation of Optimal Reductions of Distributions","year":2018,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Reduction (mathematics); Backtracking; Algorithm; Distribution (mathematics); Mathematics; Computer science; Property (philosophy); Mathematical optimization; Production (economics); Substitution (logic); Programming language","score_opus":0.07251052579286775,"score_gpt":0.22120010525446726,"score_spread":0.1486895794615995,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2794627393","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.039552446,0.00009214458,0.95124185,0.00033130648,0.00004918302,0.00027166976,0.00034886014,0.0036255,0.004487068],"genre_scores_gemma":[0.3997901,0.00016044386,0.59246826,0.00018896956,0.00003880154,0.000464141,0.0011751769,0.0019155892,0.0037985153],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99640816,0.0009921485,0.00019160165,0.000827665,0.0012535255,0.0003269365],"domain_scores_gemma":[0.9901725,0.0064934827,0.00031711115,0.0019119311,0.0009874993,0.00011734882],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024951524,0.0006948039,0.00068850734,0.0012898097,0.00072536786,0.0013266924,0.0017674767,0.0007725356,0.0051953094],"category_scores_gemma":[0.012839459,0.0006622135,0.0016726244,0.00056210574,0.0017046737,0.002186631,0.002534312,0.0015828852,0.0011845739],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007550613,0.0003792695,0.0035092623,0.0009527417,0.0001502539,0.0016382871,0.001487684,0.12709528,0.13955708,0.37465385,0.012080825,0.3377404],"study_design_scores_gemma":[0.00017256771,0.00018460288,0.00080143876,0.0001070913,0.00013860111,0.00060708105,0.00030022557,0.5486648,0.16120067,0.26290193,0.024845576,0.00007533535],"about_ca_topic_score_codex":0.00068548025,"about_ca_topic_score_gemma":0.0012343305,"teacher_disagreement_score":0.0051953094,"about_ca_system_score_codex":0.00093418255,"about_ca_system_score_gemma":0.0018851327,"threshold_uncertainty_score":0.017380059},"labels":[],"label_agreement":null},{"id":"W2794692119","doi":"10.18653/v1/w18-0106","title":"Modeling bilingual word associations as connected monolingual networks","year":2018,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Word (group theory); Natural language processing; Artificial intelligence; Speech recognition; Linguistics","score_opus":0.018416566004137762,"score_gpt":0.29898245079003727,"score_spread":0.2805658847858995,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2794692119","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7415994,0.00029607982,0.24339885,0.00058100425,0.00004619759,0.00006645047,0.00037880524,0.00047244367,0.013160797],"genre_scores_gemma":[0.97191364,0.00020142624,0.021458069,0.000054002554,0.000029355471,0.00010138118,0.000252863,0.00005771608,0.0059314617],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99980754,0.000060681527,0.000007764081,0.00007618285,0.000020386802,0.000027457154],"domain_scores_gemma":[0.9992341,0.00045486647,0.0001078019,0.00007267092,0.000071386086,0.00005927937],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00047559675,0.00047202737,0.0004105473,0.00070194626,0.00047557565,0.0008384994,0.0007788649,0.0008416405,0.004372262],"category_scores_gemma":[0.0031697901,0.00043249744,0.0005398099,0.0006029736,0.0005961512,0.0023933926,0.0009991989,0.000687587,0.0006318035],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00047400422,0.00016976436,0.0111134285,0.00013266238,0.00012972602,0.00057011255,0.00061423716,0.86542857,0.007802145,0.0760256,0.0013743795,0.036165383],"study_design_scores_gemma":[0.000019921728,0.000029316521,0.0009318108,0.00000517767,0.000026279058,0.000070603346,0.000040818068,0.9564703,0.00060592446,0.04108067,0.00071085605,0.000008417519],"about_ca_topic_score_codex":0.0055139526,"about_ca_topic_score_gemma":0.008702466,"teacher_disagreement_score":0.0055139526,"about_ca_system_score_codex":0.00072964205,"about_ca_system_score_gemma":0.0005830934,"threshold_uncertainty_score":0.014626682},"labels":[],"label_agreement":null},{"id":"W2795018103","doi":"10.5339/qfarc.2018.ssahpd880","title":"Building a Rich Lexical Resource for Standard Arabic","year":2018,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Arabic; Modern Standard Arabic; Natural language processing; Resource (disambiguation); Artificial intelligence; Linguistics","score_opus":0.016329767503907754,"score_gpt":0.31442999475722805,"score_spread":0.2981002272533203,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2795018103","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.059236262,0.0044980543,0.6435809,0.0033735954,0.0015026975,0.003970064,0.11139032,0.10834549,0.0641025],"genre_scores_gemma":[0.12929375,0.001853267,0.70584714,0.0010563666,0.0002791527,0.0024742272,0.13763706,0.006487435,0.015071588],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.998398,0.00023782367,0.0003564332,0.00050212047,0.00033509068,0.00017054506],"domain_scores_gemma":[0.9976144,0.0006453197,0.0002029005,0.00048929465,0.0008240225,0.00022405855],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001486592,0.0017420084,0.0018189057,0.01004668,0.00230062,0.0036190175,0.0023515408,0.00130477,0.04230459],"category_scores_gemma":[0.0072284844,0.0013766276,0.001870351,0.005564714,0.00093758793,0.010000324,0.005982567,0.0021977497,0.03175734],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009512009,0.00049991143,0.005357983,0.003939016,0.00020870376,0.004717785,0.0034967791,0.0035238448,0.042323813,0.057381883,0.2288006,0.6487986],"study_design_scores_gemma":[0.00027498472,0.00022536944,0.004591531,0.0011441112,0.000281715,0.0027302685,0.0032420654,0.04100633,0.031238701,0.054324653,0.8605141,0.00042615578],"about_ca_topic_score_codex":0.004418915,"about_ca_topic_score_gemma":0.0065581016,"teacher_disagreement_score":0.04230459,"about_ca_system_score_codex":0.0013608614,"about_ca_system_score_gemma":0.0033057202,"threshold_uncertainty_score":0.141523},"labels":[],"label_agreement":null},{"id":"W2795544197","doi":"10.1007/978-3-319-89656-4_10","title":"Reranking Candidate Lists for Improved Lexical Induction","year":2018,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Natural language processing; Task (project management); Artificial intelligence; Lexicon; Translation (biology); Speech recognition","score_opus":0.016389435755552975,"score_gpt":0.27935371309484636,"score_spread":0.2629642773392934,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2795544197","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.034792755,0.0019873192,0.9266265,0.0006026226,0.0007925492,0.0004422787,0.0038759564,0.022487016,0.008392986],"genre_scores_gemma":[0.17349404,0.0008023885,0.7817201,0.0003307477,0.00072219304,0.00053449476,0.019949667,0.002360958,0.020085432],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99767464,0.0006960312,0.00026412215,0.00039778926,0.0007144797,0.00025295324],"domain_scores_gemma":[0.9938637,0.0030519783,0.00019415157,0.0009298723,0.0017854987,0.00017485852],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014703695,0.0013351116,0.0021304453,0.0064423466,0.0018753542,0.002332154,0.002827083,0.001400667,0.028023649],"category_scores_gemma":[0.009089941,0.0009256261,0.0013588488,0.005198186,0.0005652824,0.0040281597,0.0027144498,0.0021500578,0.021603554],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00045073195,0.00030035127,0.000970258,0.00047035905,0.00008558901,0.00032508888,0.0001377115,0.0074761854,0.018445743,0.008535031,0.04848853,0.9143145],"study_design_scores_gemma":[0.00035197343,0.00045366582,0.0022731822,0.00024465012,0.00038792216,0.0010791309,0.0004168606,0.8168732,0.041767675,0.08359636,0.052405644,0.00014969845],"about_ca_topic_score_codex":0.0027815308,"about_ca_topic_score_gemma":0.009081157,"teacher_disagreement_score":0.028023649,"about_ca_system_score_codex":0.00061597826,"about_ca_system_score_gemma":0.0019757675,"threshold_uncertainty_score":0.09374851},"labels":[],"label_agreement":null},{"id":"W2795677445","doi":"10.1080/01639374.2018.1438551","title":"Dealing with False Friends to Avoid Errors in Subject Analysis in Slavic Cataloging: An Overview of Resources and Strategies","year":2018,"lang":"en","type":"article","venue":"Cataloging & Classification Quarterly","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Cataloging; Slavic languages; Subject (documents); Workflow; Computer science; World Wide Web; Linguistics; Meaning (existential); Psychology; Database","score_opus":0.044217755784321036,"score_gpt":0.33241926268879674,"score_spread":0.2882015069044757,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2795677445","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.056513097,0.024823932,0.80413634,0.026244232,0.001967083,0.001804239,0.00090117747,0.019170022,0.064439856],"genre_scores_gemma":[0.20442802,0.013289229,0.73500913,0.0044856435,0.0011759229,0.0011451809,0.0025822143,0.00529935,0.03258533],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.937604,0.028811019,0.0075496966,0.0056354683,0.018313572,0.0020862285],"domain_scores_gemma":[0.8086567,0.08242824,0.021097885,0.042245485,0.042161137,0.0034106092],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.051831555,0.0023349503,0.002573737,0.02928958,0.009293324,0.017453512,0.0068358453,0.003551537,0.009775029],"category_scores_gemma":[0.17800908,0.0021144666,0.0015936587,0.014974442,0.008726414,0.030552557,0.01753163,0.0038359275,0.014389238],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022199607,0.00018697011,0.013625495,0.0018839898,0.00011876204,0.0014645804,0.04566714,0.0006927682,0.00365292,0.033114534,0.05169894,0.84767187],"study_design_scores_gemma":[0.000070348106,0.0001848492,0.013589619,0.008657952,0.0003734742,0.009348541,0.07100599,0.008666369,0.021839058,0.12751251,0.73809886,0.0006525067],"about_ca_topic_score_codex":0.008232329,"about_ca_topic_score_gemma":0.009923766,"teacher_disagreement_score":0.051831555,"about_ca_system_score_codex":0.0038896238,"about_ca_system_score_gemma":0.011374415,"threshold_uncertainty_score":0.27411473},"labels":[],"label_agreement":null},{"id":"W2797283743","doi":"10.3968/10193","title":"An Empirical Study of English-Chinese Translation of Novel Context-Free Compound Nouns and Phrases","year":2018,"lang":"en","type":"article","venue":"Canadian social science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Paraphrase; Computer science; Natural language processing; Linguistics; Context (archaeology); Affect (linguistics); Noun; Interpretation (philosophy); Artificial intelligence; Comprehension; Machine translation; Noun phrase; History","score_opus":0.02769154184229863,"score_gpt":0.33254499300731283,"score_spread":0.3048534511650142,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2797283743","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9974402,0.00006930546,0.00060772616,0.00007502995,0.000008099217,0.00011223764,0.00008544788,0.000009639859,0.0015922327],"genre_scores_gemma":[0.99439114,0.00018146244,0.0027325694,0.00010382895,0.000016042319,0.00016763352,0.00034993334,0.0000312514,0.0020262073],"study_design_codex":"qualitative","study_design_gemma":"observational","domain_scores_codex":[0.9954072,0.0025902141,0.00044776555,0.00056268374,0.00081366545,0.00017850519],"domain_scores_gemma":[0.94640285,0.03635377,0.004849814,0.0043791044,0.0071357265,0.0008787126],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0064248694,0.000574008,0.0005473303,0.00085571676,0.0019998527,0.0011484622,0.0008270871,0.00080164237,0.0043942155],"category_scores_gemma":[0.04982287,0.0004120238,0.00026294083,0.0020489877,0.001491677,0.0023473452,0.001145807,0.0011257071,0.0009341366],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0026748367,0.008352659,0.21703297,0.0029282486,0.00019520456,0.010207867,0.50803465,0.0010032859,0.03407059,0.0059762844,0.006184629,0.2033388],"study_design_scores_gemma":[0.00078249275,0.008994189,0.5974003,0.0004310044,0.00030166993,0.007666145,0.2744787,0.009547143,0.04548018,0.0032019129,0.05143302,0.00028318545],"about_ca_topic_score_codex":0.004531315,"about_ca_topic_score_gemma":0.008044207,"teacher_disagreement_score":0.0064248694,"about_ca_system_score_codex":0.0012421003,"about_ca_system_score_gemma":0.0018780723,"threshold_uncertainty_score":0.033978403},"labels":[],"label_agreement":null},{"id":"W2798569372","doi":"10.18653/v1/p18-1108","title":"Straight to the Tree: Constituency Parsing with Neural Syntactic Distance","year":2018,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":98,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; Microsoft (Canada); University of Waterloo; Canadian Institute for Advanced Research","funders":"Compute Canada","keywords":"Parsing; Computer science; Tree (set theory); Natural language processing; Artificial intelligence; Linguistics; Computational linguistics; Volume (thermodynamics); Mathematics; Philosophy; Physics; Combinatorics","score_opus":0.010567476193544536,"score_gpt":0.261205585108001,"score_spread":0.2506381089144565,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2798569372","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.038163386,0.001477973,0.90743995,0.0011093755,0.0006353828,0.0002480716,0.003393048,0.031161033,0.016371846],"genre_scores_gemma":[0.29971924,0.00095890224,0.6656273,0.00052554655,0.00027439793,0.00033952534,0.014506378,0.005180927,0.01286774],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99859387,0.00036016956,0.00010889407,0.0005221405,0.00025398194,0.00016082115],"domain_scores_gemma":[0.9982128,0.00085834484,0.0000744555,0.0004919559,0.00029678995,0.00006568],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015228629,0.0014856968,0.0019271941,0.0022339118,0.0017422412,0.002980972,0.0035872208,0.0019749082,0.014377145],"category_scores_gemma":[0.004810095,0.0012668999,0.0018232048,0.0032206576,0.001011538,0.007893971,0.0047965692,0.0035204834,0.0070757386],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00070626027,0.0003013101,0.0016510045,0.00043331092,0.000198582,0.00038386913,0.0006627163,0.02958136,0.008165505,0.0889369,0.05933148,0.80964774],"study_design_scores_gemma":[0.00014593851,0.00008736938,0.0007898531,0.00007811756,0.00017143949,0.00018118053,0.00035322798,0.68618435,0.010988066,0.27697653,0.023967968,0.000076028315],"about_ca_topic_score_codex":0.007271175,"about_ca_topic_score_gemma":0.013072674,"teacher_disagreement_score":0.014377145,"about_ca_system_score_codex":0.0011199013,"about_ca_system_score_gemma":0.0020603214,"threshold_uncertainty_score":0.04809636},"labels":[],"label_agreement":null},{"id":"W2798673653","doi":"10.52034/lanstts.v8i.245","title":"Scaling up a Hybrid MT System: From low to full resources","year":2021,"lang":"en","type":"article","venue":"Linguistica Antverpiensia New Series – Themes in Translation Studies","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Machine translation; Computer science; Scaling; Translation (biology); Natural language processing; Artificial intelligence; Hybrid system; Quality (philosophy); Machine learning; Chemistry; Mathematics","score_opus":0.02673290430719427,"score_gpt":0.29726307417715864,"score_spread":0.27053016986996437,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2798673653","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.068214476,0.00083845685,0.89336056,0.0009982275,0.00031827466,0.0005413053,0.0005270688,0.019017149,0.01618459],"genre_scores_gemma":[0.27066854,0.00036861398,0.712992,0.00063134125,0.00025592837,0.0004979145,0.0024391322,0.0031346867,0.009011794],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99596775,0.0015291702,0.00036955593,0.0009277314,0.0010187615,0.00018691747],"domain_scores_gemma":[0.9921772,0.0025862944,0.00020790804,0.0028227062,0.0018259295,0.00037990275],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004031825,0.0012696105,0.0014409518,0.0016784758,0.0013684976,0.0035873163,0.0020139702,0.0017887683,0.010129216],"category_scores_gemma":[0.010787406,0.00082548324,0.00086347084,0.0019219458,0.001048243,0.0061415457,0.0060353153,0.002407299,0.01277055],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013097243,0.0006363369,0.0046210494,0.001070149,0.00038488273,0.0011839806,0.002471538,0.022111228,0.16928662,0.019682415,0.017589357,0.7596528],"study_design_scores_gemma":[0.0007024173,0.0018614301,0.006215028,0.00046073826,0.0008484001,0.0040293317,0.0019392043,0.44734785,0.1836937,0.08994445,0.2624949,0.0004625426],"about_ca_topic_score_codex":0.0014951313,"about_ca_topic_score_gemma":0.0014197668,"teacher_disagreement_score":0.010129216,"about_ca_system_score_codex":0.00056922494,"about_ca_system_score_gemma":0.0012634528,"threshold_uncertainty_score":0.0338856},"labels":[],"label_agreement":null},{"id":"W2798727045","doi":"10.18653/v1/p18-2055","title":"Leveraging distributed representations and lexico-syntactic fixedness for token-level prediction of the idiomaticity of English verb-noun combinations","year":2018,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of New Brunswick","funders":"","keywords":"Computer science; Literal (mathematical logic); Verb; Natural language processing; Variety (cybernetics); Artificial intelligence; Noun; Linguistics; Noun phrase; Programming language","score_opus":0.03664321681416376,"score_gpt":0.2890530897824642,"score_spread":0.2524098729683004,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2798727045","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5908427,0.00049027643,0.40332037,0.0005561312,0.00010003743,0.00007417685,0.00061228103,0.0010035328,0.0030004005],"genre_scores_gemma":[0.97205997,0.000093494375,0.026436709,0.00004664007,0.000021542972,0.000035221015,0.0005152967,0.00004995291,0.00074122194],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994343,0.00020045636,0.00003796073,0.00021101919,0.000053144377,0.000063033636],"domain_scores_gemma":[0.99721754,0.0016127898,0.0004399405,0.0003988443,0.00023370834,0.00009717118],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011299106,0.000660074,0.000578429,0.0012245225,0.00039624074,0.0013742789,0.00085209316,0.00070072885,0.0014628348],"category_scores_gemma":[0.00627387,0.00029567338,0.00079291186,0.0012295312,0.0008195083,0.0037326673,0.0011575591,0.0017745425,0.00054227957],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014283768,0.00075913995,0.067723,0.00033937403,0.00057072524,0.000649138,0.00152495,0.27220652,0.024969215,0.02984661,0.0040977336,0.59588516],"study_design_scores_gemma":[0.00002075017,0.0001067242,0.0054158177,0.0000221429,0.00005847178,0.000118248456,0.0001622552,0.9597237,0.0024323617,0.031404093,0.0005048352,0.000030580206],"about_ca_topic_score_codex":0.003529261,"about_ca_topic_score_gemma":0.006431557,"teacher_disagreement_score":0.003529261,"about_ca_system_score_codex":0.0007857636,"about_ca_system_score_gemma":0.000567359,"threshold_uncertainty_score":0.0070174336},"labels":[],"label_agreement":null},{"id":"W2799734184","doi":"10.1080/23273798.2018.1465187","title":"The role of distributional factors in learning and generalising affixal plural inflection: An artificial language study","year":2018,"lang":"en","type":"article","venue":"Language Cognition and Neuroscience","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"United States-Israel Binational Science Foundation","keywords":"Inflection; Plural; Natural language processing; Artificial intelligence; Computer science; Language acquisition; Linguistics; Inflection point; Cognitive science; Psychology; Mathematics; Philosophy","score_opus":0.016494002581445132,"score_gpt":0.3059959006292845,"score_spread":0.2895018980478394,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2799734184","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99512947,0.00006809414,0.004451334,0.000022071034,0.000003246474,0.000006780534,0.000012354849,0.000014234958,0.00029252298],"genre_scores_gemma":[0.9890933,0.00014341131,0.010475369,0.000014895068,0.000003845821,0.00001806807,0.000029130828,0.000008170999,0.00021382654],"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9996462,0.00012924836,0.000041556275,0.000108287655,0.000056533187,0.00001813829],"domain_scores_gemma":[0.99680173,0.0022427149,0.00039922362,0.0003099195,0.00014169123,0.00010477599],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006848338,0.00025428535,0.00024907195,0.00025969028,0.00015885812,0.00070688105,0.0003617797,0.00029531025,0.0005220581],"category_scores_gemma":[0.00437871,0.00015826107,0.00029010858,0.00020402351,0.0011095379,0.0010971726,0.00068998465,0.00047006767,0.00007836492],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006994774,0.00060576963,0.03226099,0.0005815935,0.00007371959,0.00038980928,0.0037851166,0.0072121187,0.8691463,0.0058402014,0.000109220775,0.079295635],"study_design_scores_gemma":[0.00021940663,0.007712865,0.25046524,0.00010305648,0.00035871076,0.0024030688,0.0029059437,0.159159,0.53316027,0.035191536,0.008080985,0.00023996155],"about_ca_topic_score_codex":0.00026729034,"about_ca_topic_score_gemma":0.00042023096,"teacher_disagreement_score":0.00070688105,"about_ca_system_score_codex":0.00025328505,"about_ca_system_score_gemma":0.000177932,"threshold_uncertainty_score":0.003621757},"labels":[{"model":"gemma","categories":[],"domain":null,"study_design":"bench_or_experimental","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"low"},{"model":"gpt","categories":[],"domain":null,"study_design":"bench_or_experimental","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"high"}],"label_agreement":"agree"},{"id":"W2799804203","doi":"10.5539/ijel.v8n4p282","title":"Academic Vocabulary Use in Doctoral Theses: A Corpus-Based Lexical Analysis of Academic Word List (AWL) in Major Scientific Disciplinary Groups","year":2018,"lang":"en","type":"article","venue":"International Journal of English Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Vocabulary; Word (group theory); Discipline; Word list; Lexical density; Concordance; Computer science; Corpus linguistics; Linguistics; Natural language processing; Artificial intelligence; Lexical item; Sociology; Social science; Medicine","score_opus":0.03791565145728037,"score_gpt":0.3492451197885814,"score_spread":0.31132946833130104,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2799804203","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.97804606,0.0018513266,0.0034181348,0.0003177572,0.00006792736,0.0003968936,0.01038827,0.000076778095,0.005436763],"genre_scores_gemma":[0.95419466,0.0016598407,0.016191624,0.0001404252,0.000059186743,0.0014744749,0.023889571,0.00010495859,0.002285269],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9981238,0.00050593953,0.00038475092,0.00041279523,0.00048520352,0.000087354376],"domain_scores_gemma":[0.9897781,0.0070604887,0.0012886615,0.00035492124,0.0013125257,0.00020527715],"candidate_categories":["bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.00222434,0.0002846699,0.00047014968,0.010519035,0.0012773168,0.001999162,0.00038999718,0.000468873,0.002442347],"category_scores_gemma":[0.014134117,0.00019418536,0.00030936097,0.010782995,0.0012256246,0.0019994085,0.0026749542,0.0006759788,0.0005920612],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00077579444,0.00047766877,0.27974913,0.0126836365,0.00019611715,0.0032559815,0.29445824,0.00086215086,0.0345971,0.008001297,0.024423368,0.34051955],"study_design_scores_gemma":[0.00007911289,0.00024108417,0.6767305,0.0022231063,0.0002619603,0.0023863486,0.18328926,0.004086225,0.007236813,0.0028106917,0.12050281,0.00015205421],"about_ca_topic_score_codex":0.006048999,"about_ca_topic_score_gemma":0.0110867685,"teacher_disagreement_score":0.989481,"about_ca_system_score_codex":0.0009400402,"about_ca_system_score_gemma":0.0015853784,"threshold_uncertainty_score":0.012027621},"labels":[],"label_agreement":null},{"id":"W2800994542","doi":"10.5167/uzh-8817","title":"The automatic translation of film subtitles: a machine translation success story?","year":2008,"lang":"en","type":"book-chapter","venue":"Zurich Open Repository and Archive (University of Zurich)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Machine translation; Artificial intelligence; Natural language processing; Linguistics","score_opus":0.014589106777421807,"score_gpt":0.21445195386709914,"score_spread":0.19986284708967733,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2800994542","genre_codex":"commentary","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.053526133,0.10571843,0.054467704,0.6273178,0.009320796,0.000224186,0.0014566208,0.0022593497,0.145709],"genre_scores_gemma":[0.596633,0.07837634,0.09542865,0.06354315,0.025738165,0.00042998308,0.004405362,0.005479964,0.12996542],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9885318,0.005478783,0.0007202383,0.0010217841,0.003788596,0.00045873568],"domain_scores_gemma":[0.9558598,0.030017732,0.0018446486,0.0036068554,0.007896393,0.0007745092],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011319207,0.0010506805,0.0008086983,0.0032123865,0.0036608863,0.010339217,0.0016387874,0.0050099585,0.0071155243],"category_scores_gemma":[0.042376384,0.0006500568,0.0005346414,0.0034285593,0.007942287,0.020503977,0.0036239626,0.0062598833,0.007327036],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00027509776,0.00016353052,0.0033617786,0.0019263093,0.000082081024,0.0013467347,0.020603165,0.0011018934,0.0047254483,0.14040478,0.29380897,0.5322002],"study_design_scores_gemma":[0.000061805746,0.0001888862,0.005372501,0.0009274355,0.00007221036,0.002759832,0.014673034,0.0060643475,0.012901611,0.053314447,0.9035245,0.0001393829],"about_ca_topic_score_codex":0.0068119983,"about_ca_topic_score_gemma":0.004705558,"teacher_disagreement_score":0.011319207,"about_ca_system_score_codex":0.0029281809,"about_ca_system_score_gemma":0.0014176928,"threshold_uncertainty_score":0.059862375},"labels":[],"label_agreement":null},{"id":"W2802325800","doi":"10.1108/lht-12-2017-0271","title":"Corpus linguistics is not just for linguists","year":2018,"lang":"en","type":"article","venue":"Library Hi Tech","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Corpus linguistics; Computer science; Applied linguistics; Computational linguistics; Field (mathematics); Quantitative linguistics; Text linguistics; Text corpus; Linguistics; Natural language processing; Artificial intelligence; Value (mathematics); Data science","score_opus":0.025081537200789174,"score_gpt":0.2969873667420595,"score_spread":0.27190582954127035,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2802325800","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0073221624,0.042392615,0.23964302,0.5508641,0.02719706,0.0010198058,0.0017034955,0.0017943045,0.12806338],"genre_scores_gemma":[0.27366844,0.052946266,0.42130855,0.13489927,0.03413493,0.0058461213,0.002793302,0.00570688,0.068696246],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9331446,0.05097947,0.0032927275,0.0041705053,0.0076926667,0.00072012463],"domain_scores_gemma":[0.7461881,0.18955708,0.0075402535,0.030345665,0.022124736,0.0042442014],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.055808887,0.0009768044,0.0020505963,0.009259458,0.010483135,0.021329178,0.0034787285,0.004539399,0.017504169],"category_scores_gemma":[0.14357391,0.0011312417,0.0008999488,0.0094624935,0.027372895,0.034695223,0.010536436,0.014097467,0.0062520187],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005491614,0.000041528616,0.0010437808,0.0013568493,0.000073922485,0.0002051007,0.018824644,0.00049928576,0.0005707144,0.75604904,0.13446097,0.086819254],"study_design_scores_gemma":[0.000016898279,0.000024134051,0.00026366086,0.0028415336,0.000024213565,0.0002726071,0.0073417732,0.0006056573,0.00034966032,0.23612307,0.75209665,0.00004006433],"about_ca_topic_score_codex":0.0035302774,"about_ca_topic_score_gemma":0.0041565835,"teacher_disagreement_score":0.055808887,"about_ca_system_score_codex":0.0066931206,"about_ca_system_score_gemma":0.012879548,"threshold_uncertainty_score":0.29514915},"labels":[],"label_agreement":null},{"id":"W2803285567","doi":"10.5539/ijel.v8n5p87","title":"Deixis Role as an Index of Style: A Comparative Corpus Stylistics Analysis of Self, Pakistani and Other Translators","year":2018,"lang":"en","type":"article","venue":"International Journal of English Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Deixis; Style (visual arts); Linguistics; Context (archaeology); Computer science; Coherence (philosophical gambling strategy); Psychology; Literature; History; Art; Philosophy; Mathematics","score_opus":0.013940346177505008,"score_gpt":0.32436400047765257,"score_spread":0.31042365430014757,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2803285567","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98305446,0.0003811566,0.0013776749,0.00021000383,0.000027598835,0.00008740798,0.0006472572,0.000017471912,0.014197041],"genre_scores_gemma":[0.9934562,0.00036158605,0.0016835764,0.000032627624,0.00002551968,0.00010969678,0.000594824,0.000042223928,0.0036937515],"study_design_codex":"qualitative","study_design_gemma":"observational","domain_scores_codex":[0.9990176,0.0003485231,0.000090553905,0.00019166971,0.00027614812,0.00007548678],"domain_scores_gemma":[0.9937262,0.004175159,0.0007029112,0.00030319378,0.00093568995,0.00015679447],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017036531,0.00019411762,0.00028103875,0.004087326,0.0021533505,0.00199085,0.00024352895,0.0002420479,0.0031409676],"category_scores_gemma":[0.0057824175,0.00014950102,0.00015138164,0.004206558,0.0017041279,0.0016216672,0.0011159903,0.00046929266,0.000368388],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00039189492,0.00007420957,0.087277986,0.00065827696,0.000036886624,0.002825931,0.77725756,0.00016020023,0.01567166,0.015062249,0.005415575,0.095167615],"study_design_scores_gemma":[0.000021718033,0.00015515214,0.46109077,0.0002467219,0.00004810087,0.002724865,0.43371144,0.0017887603,0.0064459853,0.002065345,0.09164243,0.000058656136],"about_ca_topic_score_codex":0.004704428,"about_ca_topic_score_gemma":0.009575453,"teacher_disagreement_score":0.004704428,"about_ca_system_score_codex":0.001377827,"about_ca_system_score_gemma":0.0007435927,"threshold_uncertainty_score":0.010507524},"labels":[],"label_agreement":null},{"id":"W2803450908","doi":"10.1515/cllt-2016-0078","title":"Unifying dimensions in coherence relations: How various annotation frameworks are related","year":2018,"lang":"en","type":"article","venue":"Corpus Linguistics and Linguistic Theory","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":71,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Universiteit Utrecht; Schweizerischer Nationalfonds zur Förderung der Wissenschaftlichen Forschung; Deutsche Forschungsgemeinschaft","keywords":"Treebank; Computer science; Annotation; Coherence (philosophical gambling strategy); Dimension (graph theory); Set (abstract data type); Representation (politics); Rhetorical question; Natural language processing; Artificial intelligence; Interface (matter); Linguistics; Programming language; Mathematics","score_opus":0.010668414240520572,"score_gpt":0.2592416203831324,"score_spread":0.24857320614261186,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2803450908","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10954026,0.0011838954,0.8587389,0.0037275434,0.00019466033,0.0002467319,0.00067681697,0.001634307,0.024056865],"genre_scores_gemma":[0.68292624,0.00047751094,0.31292242,0.00029152117,0.000056772387,0.00044008225,0.0007274777,0.00062150037,0.0015364907],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.96221215,0.025893295,0.0022718112,0.003941842,0.0047689537,0.0009119145],"domain_scores_gemma":[0.8836663,0.08192499,0.006693074,0.015607001,0.010945959,0.001162666],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.035313495,0.00074832456,0.0008822095,0.010172279,0.0034974925,0.010866551,0.0018892824,0.0015991734,0.003390228],"category_scores_gemma":[0.13137606,0.001046805,0.00079426856,0.007739642,0.008524878,0.024497647,0.010828726,0.003397983,0.0006181897],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003875284,0.00008431241,0.0123370625,0.0006216745,0.00008803532,0.00015281513,0.052502804,0.0029234579,0.008466743,0.66977847,0.0033687034,0.2492885],"study_design_scores_gemma":[0.00007565854,0.00012675715,0.014433395,0.001081267,0.00017769155,0.00033105342,0.03414431,0.050652973,0.014418962,0.7954169,0.088849775,0.0002913687],"about_ca_topic_score_codex":0.0068371138,"about_ca_topic_score_gemma":0.006671171,"teacher_disagreement_score":0.035313495,"about_ca_system_score_codex":0.0034284343,"about_ca_system_score_gemma":0.0032387988,"threshold_uncertainty_score":0.18675786},"labels":[],"label_agreement":null},{"id":"W2803871820","doi":"10.1007/978-3-319-91947-8_48","title":"Cross-Language Text Summarization Using Sentence and Multi-Sentence Compression","year":2018,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Agence Nationale de la Recherche","keywords":"Automatic summarization; Computer science; Natural language processing; Sentence; Artificial intelligence; Multi-document summarization","score_opus":0.02272415501754649,"score_gpt":0.3068379027972656,"score_spread":0.2841137477797191,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2803871820","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04233222,0.0034669463,0.9160141,0.0007487051,0.0016290818,0.00067778176,0.0053498894,0.022598729,0.0071825357],"genre_scores_gemma":[0.132755,0.0015449751,0.8160201,0.0003691984,0.0013965515,0.00063204695,0.027419128,0.0020023116,0.017860824],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989478,0.00022706525,0.0001494789,0.00026922807,0.00031525042,0.00009131111],"domain_scores_gemma":[0.99676466,0.000917882,0.00021973859,0.00043688007,0.0015632875,0.000097663455],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009849619,0.0024367373,0.0018523751,0.0034299234,0.00091410964,0.0016951414,0.0011012868,0.0010955262,0.010418712],"category_scores_gemma":[0.0032101383,0.0005919529,0.0016476105,0.003189695,0.0003116502,0.002138539,0.0015107142,0.0012568946,0.008785684],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00058965373,0.00023186393,0.0005731585,0.00077413896,0.00026569772,0.0005551956,0.00026078086,0.0055082883,0.11837783,0.0016451911,0.03704191,0.8341763],"study_design_scores_gemma":[0.00025719497,0.0011887507,0.00964075,0.00018909648,0.0015267327,0.0017475878,0.0010563694,0.5221553,0.35729027,0.011672153,0.09300861,0.00026724392],"about_ca_topic_score_codex":0.0012509896,"about_ca_topic_score_gemma":0.0024221186,"teacher_disagreement_score":0.010418712,"about_ca_system_score_codex":0.00037492142,"about_ca_system_score_gemma":0.00075221364,"threshold_uncertainty_score":0.034854054},"labels":[],"label_agreement":null},{"id":"W2804967932","doi":"10.18653/v1/n18-1087","title":"Noise-Robust Morphological Disambiguation for Dialectal Arabic","year":2018,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"York University; New York University Abu Dhabi","keywords":"Arabic; Computer science; Natural language processing; Computational linguistics; Artificial intelligence; Linguistics; Speech recognition; Philosophy","score_opus":0.03194904325261866,"score_gpt":0.29231115738155505,"score_spread":0.2603621141289364,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2804967932","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19667976,0.013243273,0.7382685,0.001431075,0.0027408437,0.00018653969,0.0028944511,0.029165426,0.015390217],"genre_scores_gemma":[0.5160023,0.0021031948,0.46184865,0.0005032481,0.0004284972,0.00010104904,0.0063052624,0.0021678507,0.010540031],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99886084,0.0002105123,0.0001433997,0.0003546957,0.00029975665,0.00013088806],"domain_scores_gemma":[0.9984016,0.00033783988,0.000096154785,0.00033687952,0.00075631274,0.00007119348],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009485982,0.0014455059,0.0013030085,0.002697224,0.0016826774,0.0015956828,0.0012195709,0.0010249404,0.0044406955],"category_scores_gemma":[0.0030942468,0.00045114008,0.0007022182,0.0016199696,0.00064643094,0.0016427423,0.0025217107,0.0010112094,0.0073042433],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011184452,0.00010060871,0.0016478647,0.0005735216,0.000105400635,0.0009912598,0.0005323968,0.006123424,0.13422862,0.005730178,0.023914222,0.8249341],"study_design_scores_gemma":[0.00021079594,0.00052126945,0.009404371,0.00026593753,0.00035517986,0.0048127086,0.0030253076,0.36842924,0.4211413,0.053370062,0.13819401,0.00026983477],"about_ca_topic_score_codex":0.0016787107,"about_ca_topic_score_gemma":0.003546923,"teacher_disagreement_score":0.0044406955,"about_ca_system_score_codex":0.00035716264,"about_ca_system_score_gemma":0.000951023,"threshold_uncertainty_score":0.014855623},"labels":[],"label_agreement":null},{"id":"W2805088214","doi":"10.63317/57xwzvstuxxf","title":"Retrieving Information from the French Lexical Network in RDF/OWL Format","year":2018,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; Université du Québec à Montréal; Université du Québec","funders":"","keywords":"RDF; Computer science; Information retrieval; Natural language processing; Web Ontology Language; RDF Schema; Simple Knowledge Organization System; SPARQL; World Wide Web; Artificial intelligence; Semantic Web","score_opus":0.009579018819570644,"score_gpt":0.24659368900266354,"score_spread":0.23701467018309288,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2805088214","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17787941,0.0057881987,0.46912616,0.005627492,0.00061401684,0.00086692994,0.17686327,0.03503738,0.12819716],"genre_scores_gemma":[0.4805527,0.0061932807,0.26526487,0.00085496367,0.00023655669,0.00040943729,0.21814568,0.0032478191,0.025094572],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99937004,0.00013896992,0.00006880152,0.00014168369,0.0002047746,0.000075706586],"domain_scores_gemma":[0.9991829,0.00034861884,0.000046266,0.00011659335,0.00026193733,0.000043757675],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005878592,0.0012382859,0.00082552736,0.00823902,0.0012625819,0.0029253713,0.00070351374,0.00080178556,0.013061882],"category_scores_gemma":[0.003168451,0.00036090318,0.00095567585,0.0050516366,0.0004977612,0.0041547893,0.0013493809,0.000712736,0.004386229],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00079717854,0.00043947613,0.013258178,0.0027693342,0.00034536578,0.0060524,0.0026310782,0.015665038,0.044528577,0.13287409,0.24006814,0.54057115],"study_design_scores_gemma":[0.00013198455,0.000121816185,0.014222828,0.001217518,0.000617651,0.002825359,0.0044669793,0.094789036,0.0486843,0.099418946,0.73330706,0.00019648655],"about_ca_topic_score_codex":0.071963005,"about_ca_topic_score_gemma":0.06219511,"teacher_disagreement_score":0.071963005,"about_ca_system_score_codex":0.0018876627,"about_ca_system_score_gemma":0.0023895395,"threshold_uncertainty_score":0.14308828},"labels":[],"label_agreement":null},{"id":"W2805089673","doi":"10.63317/3wos9hprfv6b","title":"You Tweet What You Speak: A City-Level Dataset of Arabic Dialects","year":2018,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":62,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Arabic; Computer science; Natural language processing; Artificial intelligence; Linguistics","score_opus":0.03993097122083863,"score_gpt":0.30376725739816235,"score_spread":0.26383628617732374,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2805089673","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.37089285,0.0007715513,0.0015946092,0.0005858867,0.0002527098,0.00014026546,0.6177856,0.002045173,0.005931386],"genre_scores_gemma":[0.18835282,0.00027770502,0.0042519094,0.00019778643,0.00010170711,0.0002438538,0.7999272,0.0001626438,0.00648441],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9997216,0.00005135006,0.000030488703,0.00007152635,0.000067208246,0.000057778212],"domain_scores_gemma":[0.9992551,0.00016509987,0.00006618417,0.00011554862,0.00022402439,0.00017394322],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00021838922,0.0007386524,0.00044684752,0.0022668573,0.0007440056,0.00055701216,0.00047145336,0.0008882022,0.0050142272],"category_scores_gemma":[0.0014098978,0.00015801437,0.00045893347,0.0023611416,0.0002332637,0.00054198474,0.0010044244,0.000576781,0.009686495],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0027446654,0.0010857556,0.19822936,0.0022167873,0.0006336486,0.002592729,0.0043290313,0.0040511973,0.03032833,0.0014808973,0.63695794,0.115349695],"study_design_scores_gemma":[0.00028716744,0.0003902048,0.5926418,0.00020057157,0.00025429146,0.0021074878,0.0082125515,0.01435788,0.008591783,0.0010251054,0.37169877,0.00023253553],"about_ca_topic_score_codex":0.023154292,"about_ca_topic_score_gemma":0.048661634,"teacher_disagreement_score":0.023154292,"about_ca_system_score_codex":0.00039142728,"about_ca_system_score_gemma":0.00061160635,"threshold_uncertainty_score":0.046039045},"labels":[],"label_agreement":null},{"id":"W2805216538","doi":"","title":"SINAI at TAC-KBP BeST 2017: evaluating the impact of modal verbs in the classification of beliefs.","year":2017,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Modal; Modal verb; Natural language processing; Computer science; Artificial intelligence; Linguistics; Philosophy; Verb; Chemistry","score_opus":0.03690105924165262,"score_gpt":0.3739920860478443,"score_spread":0.3370910268061917,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2805216538","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.57291883,0.0064668143,0.037316926,0.0096030645,0.0056010894,0.0027329128,0.21250325,0.050108954,0.102748215],"genre_scores_gemma":[0.46009377,0.0008230139,0.05898556,0.001469614,0.00049339,0.0015222529,0.44563884,0.003408116,0.027565423],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99146163,0.004878974,0.00048725083,0.0009270951,0.0018542035,0.0003908159],"domain_scores_gemma":[0.96205246,0.02339567,0.00077815773,0.0048864875,0.006543774,0.002343477],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008985077,0.0025658647,0.0015459785,0.0034950569,0.0025176264,0.004505244,0.0039983653,0.0036060866,0.021903556],"category_scores_gemma":[0.054675873,0.0010110891,0.0013804986,0.0028150172,0.001508991,0.008416718,0.004599451,0.005102819,0.01664141],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.009124908,0.005876677,0.019683456,0.0024109704,0.0009017107,0.00084855396,0.0019799203,0.017302372,0.0033822472,0.0045603504,0.74851865,0.18541023],"study_design_scores_gemma":[0.007615491,0.0036131244,0.065998845,0.001251936,0.0017428184,0.0013094277,0.01031392,0.5252092,0.026059376,0.031072246,0.32509118,0.00072232814],"about_ca_topic_score_codex":0.06738105,"about_ca_topic_score_gemma":0.089241154,"teacher_disagreement_score":0.06738105,"about_ca_system_score_codex":0.0025903385,"about_ca_system_score_gemma":0.0030928026,"threshold_uncertainty_score":0.13397771},"labels":[],"label_agreement":null},{"id":"W2805224294","doi":"","title":"SRCB Entity Discovery and Linking (EDL) and Event Nugget Systems for TAC 2017.","year":2017,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Event (particle physics); Physics","score_opus":0.010637355573387833,"score_gpt":0.2876668333796569,"score_spread":0.2770294778062691,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2805224294","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.036364097,0.0023382548,0.4929368,0.005176669,0.0016403089,0.001027952,0.08196357,0.33227956,0.04627284],"genre_scores_gemma":[0.15411289,0.0008863658,0.53899497,0.0008616864,0.00033089845,0.00068057014,0.26732266,0.009590363,0.02721962],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9955449,0.0014699544,0.0003550393,0.0007653979,0.0014824204,0.000382225],"domain_scores_gemma":[0.99190044,0.0022625097,0.0004604012,0.0033822942,0.001528479,0.00046583515],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006815654,0.0010321637,0.0008486439,0.0042724577,0.001749135,0.0039533055,0.0030348746,0.0016789832,0.0239556],"category_scores_gemma":[0.01763331,0.0007736993,0.0013782437,0.0033979265,0.0009307664,0.00908754,0.005059262,0.002678009,0.015447561],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015235228,0.000514223,0.0065742256,0.00092993735,0.00029593988,0.0006357975,0.00079812383,0.014311712,0.0059496295,0.07493033,0.5979209,0.29561567],"study_design_scores_gemma":[0.00031779413,0.00026133118,0.0039337287,0.00026252645,0.00016243654,0.000580471,0.0005459876,0.29880753,0.019779427,0.11825188,0.5569409,0.00015603176],"about_ca_topic_score_codex":0.01813657,"about_ca_topic_score_gemma":0.036937714,"teacher_disagreement_score":0.0239556,"about_ca_system_score_codex":0.0016347037,"about_ca_system_score_gemma":0.0043326598,"threshold_uncertainty_score":0.08013952},"labels":[],"label_agreement":null},{"id":"W2805259170","doi":"","title":"Stanford at TAC KBP 2016: Sealing Pipeline Leaks and Understanding Chinese.","year":2016,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Pipeline (software); Computer science; Petroleum engineering; Engineering; Programming language","score_opus":0.012161183004247166,"score_gpt":0.2670602868645314,"score_spread":0.25489910386028425,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2805259170","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12233805,0.008579005,0.36239603,0.03432687,0.0043295794,0.0007483465,0.23781464,0.096451305,0.13301621],"genre_scores_gemma":[0.32916954,0.003347529,0.2923169,0.0014879603,0.0007001044,0.00059304835,0.2872707,0.009346471,0.075767756],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9986313,0.0005068986,0.000102032405,0.0002784211,0.0003924824,0.000088856774],"domain_scores_gemma":[0.99688077,0.001193223,0.000105950334,0.00069155864,0.00091212464,0.00021643692],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025692924,0.0013563696,0.00071444054,0.0025065348,0.0027730998,0.0028413017,0.0015882509,0.0014022699,0.032367725],"category_scores_gemma":[0.010981306,0.0007886641,0.00058424385,0.0024887542,0.0010587006,0.009961801,0.0031009847,0.0020451506,0.011799736],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002625092,0.00012744796,0.0035873926,0.0005221515,0.000050896764,0.0004951468,0.0031412905,0.0030739924,0.0035778314,0.021596981,0.83034784,0.13321654],"study_design_scores_gemma":[0.00018681335,0.0000864058,0.010794445,0.00044220217,0.00010601502,0.00047311356,0.0040100375,0.055485025,0.016541524,0.090031594,0.82166034,0.00018244912],"about_ca_topic_score_codex":0.06762685,"about_ca_topic_score_gemma":0.10035875,"teacher_disagreement_score":0.06762685,"about_ca_system_score_codex":0.0017092769,"about_ca_system_score_gemma":0.0036905748,"threshold_uncertainty_score":0.13446641},"labels":[],"label_agreement":null},{"id":"W2805287993","doi":"10.18653/v1/w18-1704","title":"Multi-Sentence Compression with Word Vertex-Labeled Graphs and Integer Linear Programming","year":2018,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Sentence; Integer programming; Vertex (graph theory); Graph; Word (group theory); Combinatorics; Integer (computer science); Data compression; Natural language processing; Artificial intelligence; Discrete mathematics; Programming language; Theoretical computer science; Algorithm; Mathematics","score_opus":0.022888285586641563,"score_gpt":0.2949515939252628,"score_spread":0.27206330833862125,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2805287993","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025240865,0.0015401222,0.9549929,0.001402185,0.00045388372,0.00023781728,0.0020473183,0.009566061,0.0045187348],"genre_scores_gemma":[0.17642796,0.0005604925,0.80781275,0.000516843,0.00037749266,0.00039493496,0.0068221814,0.0015491412,0.0055382233],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99838185,0.00061047607,0.00012208932,0.0003994136,0.00034495612,0.00014122928],"domain_scores_gemma":[0.9958835,0.0027330879,0.0002074022,0.0006624141,0.00042702057,0.000086571876],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013181731,0.000999099,0.0013549409,0.00234301,0.00075134146,0.002072683,0.0017266448,0.0010933981,0.007125287],"category_scores_gemma":[0.007350111,0.00058455765,0.0014121386,0.0030311157,0.0006425502,0.0038996025,0.0018216546,0.002149877,0.0030753186],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006719765,0.00047423932,0.0009825133,0.00056471216,0.00016401672,0.00038019917,0.00033611996,0.14932485,0.0078477245,0.05890754,0.04374033,0.73660576],"study_design_scores_gemma":[0.00006137648,0.00007205886,0.00019648716,0.00005107648,0.000045995563,0.00008621225,0.00013663719,0.86684084,0.0046371124,0.12239564,0.0054529244,0.000023681972],"about_ca_topic_score_codex":0.0031666006,"about_ca_topic_score_gemma":0.0068018017,"teacher_disagreement_score":0.007125287,"about_ca_system_score_codex":0.0010739644,"about_ca_system_score_gemma":0.0015223462,"threshold_uncertainty_score":0.023836493},"labels":[],"label_agreement":null},{"id":"W2805323283","doi":"10.63317/4rme2bjhe3u5","title":"Attention for Implicit Discourse Relation Recognition","year":2018,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Relation (database); Computer science; Natural language processing; Artificial intelligence; Linguistics; Speech recognition; Data mining; Philosophy","score_opus":0.023412158222103015,"score_gpt":0.32272930178462245,"score_spread":0.2993171435625194,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2805323283","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07003242,0.0027219865,0.89899915,0.0010266352,0.0008180278,0.00020056443,0.0013550028,0.012230958,0.0126151545],"genre_scores_gemma":[0.7551347,0.0010423709,0.21398665,0.00048856315,0.00047975907,0.0002428981,0.00363805,0.0008841979,0.0241028],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9983487,0.000477658,0.00010987554,0.00055576273,0.0002961297,0.00021187628],"domain_scores_gemma":[0.9954341,0.0025863124,0.00019012373,0.0009395606,0.0006813763,0.00016855115],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017295559,0.0010116623,0.0011788432,0.0015869499,0.00109762,0.0023474311,0.0021003406,0.0013797719,0.012065778],"category_scores_gemma":[0.008310511,0.0006065796,0.000753111,0.0013065184,0.00079094,0.0063289604,0.0032081965,0.0030042087,0.003786664],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007437914,0.00023770722,0.001485826,0.00046080223,0.00007268664,0.00022063003,0.00079327455,0.004761459,0.040475346,0.032602295,0.0160205,0.9021258],"study_design_scores_gemma":[0.00010048617,0.00034852108,0.004011698,0.00021150922,0.00026642048,0.00035168065,0.00062000356,0.6663557,0.106053405,0.17857295,0.042998377,0.0001092784],"about_ca_topic_score_codex":0.0057212557,"about_ca_topic_score_gemma":0.008200912,"teacher_disagreement_score":0.012065778,"about_ca_system_score_codex":0.0010553956,"about_ca_system_score_gemma":0.0015287487,"threshold_uncertainty_score":0.040364027},"labels":[],"label_agreement":null},{"id":"W2805370698","doi":"","title":"BBN's 2017 KBP EAL Submission.","year":2017,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science","score_opus":0.011552152296750886,"score_gpt":0.2924317186969415,"score_spread":0.28087956640019063,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2805370698","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0026367672,0.005285636,0.05019624,0.053124998,0.20446531,0.0006248586,0.30290082,0.036259174,0.3445063],"genre_scores_gemma":[0.006638772,0.0025275433,0.01600403,0.007098355,0.017893933,0.00044854122,0.43445525,0.02348112,0.49145246],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9926555,0.0015229833,0.00059856864,0.0008321843,0.0036761472,0.00071459735],"domain_scores_gemma":[0.97294,0.005687369,0.00053868885,0.0037958233,0.012378547,0.0046595037],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0052535087,0.0023878715,0.0030882733,0.004177409,0.0029492134,0.013952372,0.004000242,0.0043321783,0.58354926],"category_scores_gemma":[0.039923515,0.0011762717,0.0020876704,0.0040818774,0.0010789705,0.009438901,0.009788621,0.004568336,0.6327764],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00004474007,0.000014236228,0.00002265903,0.000076719,0.0000056794593,0.000021400476,0.000008274314,0.000050719143,0.00013364562,0.0010195668,0.99390644,0.004695875],"study_design_scores_gemma":[0.000050559953,0.000020237052,0.00023692714,0.00014742208,0.000010356539,0.0000945436,0.00006546495,0.0007119847,0.00059876865,0.008090639,0.98994154,0.000031495107],"about_ca_topic_score_codex":0.0068704155,"about_ca_topic_score_gemma":0.008075605,"teacher_disagreement_score":0.58354926,"about_ca_system_score_codex":0.0039466624,"about_ca_system_score_gemma":0.004574409,"threshold_uncertainty_score":0.59401643},"labels":[],"label_agreement":null},{"id":"W2805386180","doi":"","title":"A Hybrid Model for Trilingual Entity Detection and Linking Tasks at TAC KBP 2017.","year":2017,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence","score_opus":0.01546274021574759,"score_gpt":0.2934644213378367,"score_spread":0.27800168112208906,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2805386180","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.040155217,0.0008975453,0.92639035,0.0016073565,0.00034841706,0.0003905139,0.0070851445,0.015334359,0.007791188],"genre_scores_gemma":[0.41900358,0.0006758817,0.5383712,0.00065424445,0.00016304803,0.0009061814,0.018757258,0.001009442,0.020459164],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991597,0.00019332654,0.00007477009,0.00024523094,0.00023437574,0.00009253572],"domain_scores_gemma":[0.99823403,0.0005997863,0.00007517938,0.0003729638,0.0005801426,0.00013789993],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018238758,0.00072177325,0.00076766504,0.0021395423,0.0008651134,0.0025151167,0.0031400542,0.0014888275,0.006542891],"category_scores_gemma":[0.004493215,0.0005286111,0.0010644671,0.0019194885,0.00038487997,0.004833046,0.0017605391,0.0016227353,0.0044858223],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012371426,0.00100118,0.008723828,0.00057918334,0.00053181325,0.0006102658,0.0007873227,0.2882629,0.010600386,0.04629542,0.090996474,0.55037403],"study_design_scores_gemma":[0.00002566496,0.00004430716,0.0005341001,0.000023283948,0.0000664454,0.000070366776,0.00007588245,0.9701073,0.001750796,0.0190778,0.0082019,0.000022153208],"about_ca_topic_score_codex":0.026065435,"about_ca_topic_score_gemma":0.053544987,"teacher_disagreement_score":0.026065435,"about_ca_system_score_codex":0.0012279775,"about_ca_system_score_gemma":0.002757519,"threshold_uncertainty_score":0.05182743},"labels":[],"label_agreement":null},{"id":"W2805545281","doi":"","title":"RPI BLENDER TAC-KBP2017 13 Languages EDL System.","year":2017,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Programming language","score_opus":0.00811745414569667,"score_gpt":0.281512245215939,"score_spread":0.27339479107024234,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2805545281","genre_codex":"software","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0056756753,0.0006208296,0.17242235,0.0005503648,0.0004994851,0.00052070606,0.082885586,0.64002067,0.09680432],"genre_scores_gemma":[0.057858825,0.00073968864,0.26598072,0.0012076877,0.00022387216,0.0011445755,0.4619362,0.105385005,0.10552343],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9982944,0.00030444932,0.00024904893,0.0003508236,0.0006103448,0.00019087232],"domain_scores_gemma":[0.9975793,0.00042140973,0.00012986884,0.0010064993,0.0007261042,0.00013679755],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020386134,0.0015519328,0.0013035466,0.0023782991,0.00083565025,0.003323982,0.003907768,0.0014492515,0.08590802],"category_scores_gemma":[0.004781206,0.0011746521,0.0012000204,0.0017980394,0.0006141727,0.0063071814,0.004080114,0.002888872,0.097161144],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011309395,0.0003502214,0.0011958396,0.0018385123,0.000116968804,0.00036780591,0.0005641081,0.0021341275,0.013387016,0.04363341,0.7507768,0.18450437],"study_design_scores_gemma":[0.00024798146,0.00011562615,0.000795938,0.00029070905,0.000082847735,0.00038696645,0.0001668356,0.016170433,0.019055428,0.021286516,0.94129276,0.00010806638],"about_ca_topic_score_codex":0.005720333,"about_ca_topic_score_gemma":0.007107972,"teacher_disagreement_score":0.08590802,"about_ca_system_score_codex":0.001101904,"about_ca_system_score_gemma":0.0019421598,"threshold_uncertainty_score":0.28739095},"labels":[],"label_agreement":null},{"id":"W2805552138","doi":"10.1007/978-3-319-92058-0_80","title":"Identifying Similar Sentences by Using N-Grams of Characters","year":2018,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Trois-Rivières","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Syntax; Similarity (geometry); Grammar; Range (aeronautics); Categorization; Linguistics; Image (mathematics)","score_opus":0.025870512414592144,"score_gpt":0.28530009870324685,"score_spread":0.2594295862886547,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2805552138","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.61174643,0.005052882,0.30232188,0.0013141873,0.0018173555,0.00164843,0.027666682,0.017727502,0.030704705],"genre_scores_gemma":[0.5615494,0.001431423,0.394167,0.00046726724,0.00075388874,0.00048686366,0.028313337,0.0010814107,0.011749388],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9991124,0.00013838056,0.00012705794,0.00030718232,0.00024892282,0.00006601361],"domain_scores_gemma":[0.99769396,0.0010330839,0.00032001798,0.00021461236,0.0005999971,0.00013820505],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000558821,0.0013972151,0.00090058486,0.0055652284,0.0011525331,0.0013771738,0.0006131907,0.0010444113,0.007234911],"category_scores_gemma":[0.0031553062,0.00032059156,0.0009557578,0.003405521,0.00035973577,0.0020874925,0.0009197899,0.0007869487,0.0069092116],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017260923,0.00052095467,0.018649984,0.0014918123,0.00026810617,0.0023590948,0.0012406299,0.0018451604,0.2684498,0.0034360886,0.031022675,0.66898966],"study_design_scores_gemma":[0.00031083863,0.0030565073,0.117862776,0.0009872245,0.0019481909,0.013212239,0.005791487,0.37132606,0.27619764,0.03584666,0.1730192,0.00044121273],"about_ca_topic_score_codex":0.0017764195,"about_ca_topic_score_gemma":0.003966196,"teacher_disagreement_score":0.007234911,"about_ca_system_score_codex":0.0003913107,"about_ca_system_score_gemma":0.000848417,"threshold_uncertainty_score":0.024203181},"labels":[],"label_agreement":null},{"id":"W2805629278","doi":"10.63317/29r9yggbjv5c","title":"BULBasaa: A Bilingual Basaa-French Speech Corpus for the Evaluation of Language Documentation Tools","year":2018,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canada Research Chairs; University of Toronto","funders":"","keywords":"Documentation; Computer science; Natural language processing; Artificial intelligence; Linguistics; Speech recognition; Programming language","score_opus":0.046221308491140936,"score_gpt":0.3719708817561194,"score_spread":0.3257495732649785,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2805629278","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4866566,0.004841882,0.070464976,0.0017425905,0.0011299236,0.0027100819,0.3341015,0.03500429,0.063348174],"genre_scores_gemma":[0.38094515,0.0011497574,0.070888735,0.00038654282,0.00022887976,0.0030341528,0.5256234,0.0032209186,0.01452251],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99697495,0.0014619005,0.0002684843,0.00044056433,0.00064784335,0.00020622544],"domain_scores_gemma":[0.9937471,0.0023830545,0.00025955585,0.0006864758,0.002440727,0.00048309783],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023289598,0.0015490038,0.0007879414,0.005800365,0.0019829168,0.001855067,0.0013671786,0.0015622802,0.015578104],"category_scores_gemma":[0.0075688297,0.00043305958,0.00052132463,0.002591173,0.0008302602,0.0016480837,0.0019857555,0.0010678163,0.006787534],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0042775474,0.0029909178,0.019869436,0.008007859,0.00060176937,0.0029336582,0.007916044,0.008118283,0.13149147,0.017085437,0.36248088,0.43422666],"study_design_scores_gemma":[0.002052592,0.0016308456,0.09798576,0.0012219811,0.0006434385,0.00454625,0.009256911,0.045622338,0.0845112,0.0042747697,0.7478158,0.0004381889],"about_ca_topic_score_codex":0.03292585,"about_ca_topic_score_gemma":0.025938194,"teacher_disagreement_score":0.03292585,"about_ca_system_score_codex":0.0012977992,"about_ca_system_score_gemma":0.0040186653,"threshold_uncertainty_score":0.06546837},"labels":[],"label_agreement":null},{"id":"W2805639923","doi":"","title":"The USTC NELSLIP Systems for Trilingual Entity Detection and Linking Tasks at TAC KBP 2016.","year":2016,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence","score_opus":0.008215481376416218,"score_gpt":0.2586632770333835,"score_spread":0.2504477956569673,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2805639923","genre_codex":"software","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07440213,0.0019851734,0.24295063,0.0025403192,0.001991369,0.001960322,0.1195651,0.5012857,0.05331924],"genre_scores_gemma":[0.17253798,0.00062578835,0.40980643,0.0008343693,0.00021419398,0.0019119579,0.36061445,0.016101755,0.037353057],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9970594,0.0008239534,0.00031503278,0.00068351015,0.0008692906,0.0002488068],"domain_scores_gemma":[0.9945891,0.0012175038,0.00018280314,0.001694555,0.0018756853,0.0004403508],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0043959804,0.0014637567,0.00085757504,0.0030334797,0.0013349329,0.0024210613,0.0026550614,0.0015180078,0.029888116],"category_scores_gemma":[0.013644439,0.0008981572,0.00075371115,0.0016690127,0.0006300089,0.0066022077,0.00392608,0.0021891657,0.023738077],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002045055,0.00056827423,0.0032628365,0.0008945569,0.00017757411,0.00047730087,0.0011020672,0.0054195253,0.015143117,0.006532989,0.67317194,0.2912049],"study_design_scores_gemma":[0.0012720033,0.0007258769,0.0068430444,0.0004720057,0.00025185716,0.00078641553,0.0012523144,0.26327702,0.07973676,0.018576859,0.6264721,0.0003337345],"about_ca_topic_score_codex":0.0288213,"about_ca_topic_score_gemma":0.042040896,"teacher_disagreement_score":0.029888116,"about_ca_system_score_codex":0.0012312832,"about_ca_system_score_gemma":0.0035306115,"threshold_uncertainty_score":0.09998572},"labels":[],"label_agreement":null},{"id":"W2805660998","doi":"10.63317/44bfrjvhx2no","title":"Lexical Profiling of Environmental Corpora","year":2018,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Profiling (computer programming); Computer science; Natural language processing; Artificial intelligence; Programming language","score_opus":0.012543056797120056,"score_gpt":0.2557455875849487,"score_spread":0.24320253078782864,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2805660998","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.63735265,0.0033226707,0.22642194,0.001019272,0.0005253026,0.00051866256,0.042705063,0.007439442,0.08069505],"genre_scores_gemma":[0.81685126,0.001372697,0.10777552,0.00016656469,0.00017203616,0.0003066174,0.061910585,0.0014067849,0.010037902],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99842083,0.00042750532,0.00020662695,0.00036233495,0.00044033903,0.00014231211],"domain_scores_gemma":[0.9963936,0.0016947825,0.00027745328,0.00042667758,0.0010904034,0.00011707936],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00080031046,0.00044036927,0.00043799987,0.0072073825,0.0013255305,0.0024062525,0.0005599524,0.0004360105,0.006569961],"category_scores_gemma":[0.0059721665,0.00032617073,0.00045710694,0.0062396294,0.0004279267,0.002075477,0.0013951594,0.00059061614,0.00311861],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007788531,0.00045623176,0.05614619,0.0024402763,0.00024586992,0.0031247186,0.0044415146,0.0101417,0.14921804,0.028125772,0.03732418,0.70755666],"study_design_scores_gemma":[0.00011753545,0.0003651822,0.20009689,0.0010821086,0.00061882,0.0047219438,0.012855419,0.15708071,0.14869733,0.035716247,0.43839324,0.00025451678],"about_ca_topic_score_codex":0.0029372787,"about_ca_topic_score_gemma":0.009301638,"teacher_disagreement_score":0.0072073825,"about_ca_system_score_codex":0.00041290556,"about_ca_system_score_gemma":0.0011121136,"threshold_uncertainty_score":0.021978676},"labels":[],"label_agreement":null},{"id":"W2805723491","doi":"","title":"The TAI System for Trilingual Entity Discovery and Linking Track in TAC KBP 2017.","year":2017,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Track (disk drive); Computer science; Natural language processing; Artificial intelligence; Operating system","score_opus":0.010831340757054296,"score_gpt":0.29013997905092964,"score_spread":0.27930863829387537,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2805723491","genre_codex":"software","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019933226,0.001405117,0.32366702,0.0009779999,0.0006628911,0.00079779106,0.1783353,0.45214158,0.022078993],"genre_scores_gemma":[0.06751652,0.0008374475,0.34818277,0.000370221,0.00013843743,0.0009180333,0.5523835,0.011710357,0.017942727],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99811625,0.00037768108,0.00026949082,0.00049458916,0.0005913825,0.0001506188],"domain_scores_gemma":[0.99541265,0.0011120657,0.00037505475,0.0016545273,0.0011192454,0.00032650394],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003350661,0.0014838199,0.0011032562,0.0073325983,0.0016239733,0.0044612144,0.002559558,0.0014784955,0.022179719],"category_scores_gemma":[0.013940588,0.00086499634,0.0009892291,0.0059610433,0.00049547246,0.008683568,0.004378338,0.0020183395,0.026049457],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00094132306,0.00032471327,0.0062018083,0.0013784182,0.00045282813,0.00068511517,0.0013494444,0.005482592,0.0076996894,0.01648885,0.6958519,0.26314315],"study_design_scores_gemma":[0.00042028897,0.00027385604,0.0064485837,0.0005683934,0.00050466286,0.0008869382,0.0015940989,0.18442553,0.037721485,0.053308185,0.71354586,0.00030207017],"about_ca_topic_score_codex":0.023100706,"about_ca_topic_score_gemma":0.027686395,"teacher_disagreement_score":0.023100706,"about_ca_system_score_codex":0.0008948765,"about_ca_system_score_gemma":0.0040625813,"threshold_uncertainty_score":0.074198544},"labels":[],"label_agreement":null},{"id":"W2805957208","doi":"10.63317/5htxpej9286e","title":"GenDR: A Generic Deep Realizer with Complex Lexicalization","year":2018,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Lexicalization; Computer science; Artificial intelligence","score_opus":0.027059232354739483,"score_gpt":0.2831159990940788,"score_spread":0.2560567667393393,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2805957208","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017781377,0.00051556627,0.8609285,0.00023766096,0.00026486814,0.0001145887,0.0019432484,0.10688706,0.011327207],"genre_scores_gemma":[0.32137594,0.00041981903,0.62527484,0.0006875907,0.00013939005,0.00026728556,0.006865113,0.01699857,0.02797156],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993063,0.00008422804,0.00006151775,0.00026346243,0.000184331,0.00010018615],"domain_scores_gemma":[0.9995802,0.00010623074,0.000022808925,0.0001952396,0.000068979534,0.000026523569],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005338009,0.0013277619,0.0010156807,0.00072634424,0.00037882492,0.0014189362,0.0019771256,0.0012180199,0.0172056],"category_scores_gemma":[0.0011153882,0.0011766097,0.0011268654,0.0005881447,0.0007926777,0.0031562515,0.0026559706,0.002065937,0.00880819],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001168946,0.00022603145,0.0013729127,0.001643988,0.00024631768,0.0008315372,0.0005227282,0.013806007,0.111508705,0.121463664,0.06782532,0.6793839],"study_design_scores_gemma":[0.00051157735,0.00064750714,0.0016645819,0.0002888315,0.0005076284,0.0015402702,0.00028174804,0.2773559,0.2329283,0.25682738,0.22711065,0.0003356049],"about_ca_topic_score_codex":0.0015531207,"about_ca_topic_score_gemma":0.003997676,"teacher_disagreement_score":0.0172056,"about_ca_system_score_codex":0.0005133335,"about_ca_system_score_gemma":0.00082224526,"threshold_uncertainty_score":0.057558477},"labels":[],"label_agreement":null},{"id":"W2805996706","doi":"","title":"Extracting Multilingual Relations under Limited Resources: TAC 2016 Cold-Start KB construction and Slot-Filling using Compositional Universal Schema.","year":2016,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Schema (genetic algorithms); Computer science; Database; Information retrieval","score_opus":0.012828407632205014,"score_gpt":0.2598121704420818,"score_spread":0.2469837628098768,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2805996706","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.035738453,0.0009768616,0.9059649,0.0007105637,0.0002922605,0.00050300907,0.019124618,0.022339536,0.014349701],"genre_scores_gemma":[0.1429306,0.0005419676,0.7914631,0.00022659244,0.000081637096,0.0004082575,0.054943733,0.0034412302,0.00596296],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99771994,0.00055751344,0.00035593545,0.0006573747,0.0005131995,0.0001959056],"domain_scores_gemma":[0.9958501,0.0018096168,0.00019054339,0.00092748244,0.0010902836,0.00013201841],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019208712,0.0012008373,0.0011352445,0.004434603,0.0015892242,0.0031085147,0.0020088283,0.00127507,0.013148529],"category_scores_gemma":[0.010250459,0.0011455526,0.0016259126,0.0050560916,0.00093773357,0.008645257,0.0050839353,0.0018903805,0.008038448],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00076873595,0.00039074253,0.005782491,0.0038201984,0.00032442014,0.0037925937,0.00881667,0.009521149,0.059798192,0.13716325,0.11977838,0.6500432],"study_design_scores_gemma":[0.00024266266,0.0002590109,0.004333647,0.0013842537,0.0007137471,0.0027178782,0.009596053,0.22198823,0.110295214,0.20553795,0.44262937,0.0003019925],"about_ca_topic_score_codex":0.006696599,"about_ca_topic_score_gemma":0.012754702,"teacher_disagreement_score":0.013148529,"about_ca_system_score_codex":0.00089485897,"about_ca_system_score_gemma":0.004578762,"threshold_uncertainty_score":0.04398614},"labels":[],"label_agreement":null},{"id":"W2806065745","doi":"10.63317/56fzpsjj9h76","title":"Developing the Bangla RST Discourse Treebank","year":2018,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Treebank; Bengali; Computer science; Natural language processing; Artificial intelligence; Linguistics; Parsing","score_opus":0.021456067771928364,"score_gpt":0.3154380138204389,"score_spread":0.2939819460485105,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2806065745","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.101750195,0.002096289,0.5920318,0.0046672006,0.0019111591,0.001545941,0.16187397,0.047678076,0.086445384],"genre_scores_gemma":[0.20596959,0.0012628839,0.529582,0.0006495015,0.0003020013,0.0008288085,0.214204,0.004230046,0.04297119],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9989967,0.00021535707,0.00015117433,0.00033679852,0.0001989656,0.00010108359],"domain_scores_gemma":[0.997288,0.0010353032,0.00014859629,0.0004086225,0.0009947635,0.00012479456],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014145278,0.0007137759,0.00080750743,0.002695098,0.0015921133,0.0033130303,0.0010908652,0.0010019997,0.03026469],"category_scores_gemma":[0.0041174446,0.0013392027,0.0007051787,0.002374172,0.00046129004,0.0047404077,0.00211888,0.0021577156,0.021997636],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008181702,0.00035054478,0.0065432987,0.0024990041,0.0001338794,0.0016327527,0.004210772,0.009109422,0.108434275,0.08107647,0.28781807,0.49737325],"study_design_scores_gemma":[0.00015943547,0.00024392032,0.010422041,0.0005435034,0.00021796714,0.0010793963,0.0034014992,0.102155305,0.0837304,0.025113441,0.77272946,0.00020359442],"about_ca_topic_score_codex":0.0085502025,"about_ca_topic_score_gemma":0.010893271,"teacher_disagreement_score":0.03026469,"about_ca_system_score_codex":0.0017867762,"about_ca_system_score_gemma":0.0049573523,"threshold_uncertainty_score":0.10124552},"labels":[],"label_agreement":null},{"id":"W2806204962","doi":"10.1075/term.00010.dro","title":"Computational terminology and filtering of terminological information","year":2018,"lang":"en","type":"article","venue":"Terminology International Journal of Theoretical and Applied Issues in Specialized Communication","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":34,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Terminology; Computer science; Natural language processing; Linguistics; Artificial intelligence; Information retrieval; Philosophy","score_opus":0.01201346836900566,"score_gpt":0.30969218716097335,"score_spread":0.29767871879196767,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2806204962","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017735085,0.0014788137,0.9528229,0.002604356,0.00042400433,0.00021722128,0.0012415525,0.0011612469,0.022314822],"genre_scores_gemma":[0.23455195,0.0019548582,0.7451493,0.00090794713,0.00091366697,0.0005140151,0.007363159,0.0008455762,0.0077995285],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.97980154,0.008087013,0.0033931436,0.002682758,0.005208696,0.00082687504],"domain_scores_gemma":[0.94776356,0.027367422,0.0029084263,0.010107727,0.011212624,0.00064036075],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013025302,0.0007409976,0.0014340224,0.013170921,0.0036693208,0.010387682,0.0027963147,0.002117084,0.0075520356],"category_scores_gemma":[0.07174361,0.00076656777,0.0023192412,0.01254365,0.0046712733,0.015665459,0.0047889724,0.002634642,0.003475224],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009380424,0.000032343858,0.001096209,0.00035667885,0.00005406194,0.00022250174,0.002171259,0.0021935245,0.0022079651,0.8824617,0.013307031,0.095803],"study_design_scores_gemma":[0.000023529377,0.000032210817,0.0007325679,0.0003504612,0.00008807967,0.00048363043,0.0008861076,0.022416305,0.0038770735,0.9083076,0.0627425,0.0000599094],"about_ca_topic_score_codex":0.0031277463,"about_ca_topic_score_gemma":0.0020432675,"teacher_disagreement_score":0.013170921,"about_ca_system_score_codex":0.003267013,"about_ca_system_score_gemma":0.0035618949,"threshold_uncertainty_score":0.06888521},"labels":[],"label_agreement":null},{"id":"W2806272805","doi":"","title":"Overview of TAC-KBP2017 13 Languages Entity Discovery and Linking.","year":2017,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":39,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Natural language processing","score_opus":0.014325801490662333,"score_gpt":0.30995772896241264,"score_spread":0.2956319274717503,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2806272805","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.003963454,0.009068254,0.7110859,0.0036517945,0.0012194411,0.0016200066,0.04319349,0.16110173,0.06509593],"genre_scores_gemma":[0.021208797,0.005867377,0.64014834,0.0023464481,0.0005203904,0.0020260953,0.28154546,0.022125501,0.02421159],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99184203,0.0023929914,0.0011023008,0.001163684,0.0029197636,0.00057928957],"domain_scores_gemma":[0.9873767,0.0033501429,0.00043653013,0.0038927472,0.004183685,0.0007601504],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009951638,0.0023371442,0.002163985,0.011090282,0.0026523303,0.008837692,0.008254721,0.0025667127,0.030363675],"category_scores_gemma":[0.022505928,0.0022651765,0.002425566,0.010833411,0.0014041448,0.0151333725,0.00804968,0.005882421,0.035253804],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00040260013,0.00042932158,0.001024272,0.0021839563,0.00027396172,0.00036523963,0.0007411453,0.00579262,0.004640871,0.09668998,0.61463743,0.2728186],"study_design_scores_gemma":[0.00007226017,0.00007813023,0.0004741901,0.0006310297,0.0000980068,0.0004942303,0.00021309778,0.031396113,0.005814771,0.064371236,0.89624965,0.00010726538],"about_ca_topic_score_codex":0.022988826,"about_ca_topic_score_gemma":0.027516551,"teacher_disagreement_score":0.030363675,"about_ca_system_score_codex":0.0030319935,"about_ca_system_score_gemma":0.009071663,"threshold_uncertainty_score":0.10157663},"labels":[],"label_agreement":null},{"id":"W2806319975","doi":"10.63317/2zjrem3tbunr","title":"Transforming Wikipedia into a Large-Scale Fine-Grained Entity Type Corpus","year":2018,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Natural language processing; Scale (ratio); Information retrieval; Artificial intelligence; Type (biology); World Wide Web; Geology; Geography; Cartography","score_opus":0.009159296413010809,"score_gpt":0.26874343850946825,"score_spread":0.2595841420964574,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2806319975","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12311076,0.001347396,0.64195305,0.0021546434,0.0021837235,0.0009317891,0.15854484,0.046191424,0.023582304],"genre_scores_gemma":[0.19406703,0.00069254753,0.5841897,0.00036254586,0.00025975215,0.00050073647,0.20877503,0.0031969736,0.007955692],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988034,0.00020015279,0.00015065608,0.0004427159,0.00031909588,0.00008401739],"domain_scores_gemma":[0.99491465,0.0020801774,0.00020984026,0.0010617088,0.0015869989,0.00014657903],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009555981,0.00072709116,0.0007474394,0.004966089,0.0011857734,0.0020307146,0.0009559587,0.00083101174,0.0074137542],"category_scores_gemma":[0.0076157087,0.00074152637,0.0010233106,0.004921018,0.0005276738,0.003108339,0.0023898487,0.0020344316,0.006460927],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00091833144,0.00085745944,0.012403009,0.0029901974,0.00046031072,0.0042532436,0.0025405323,0.028360585,0.14998722,0.0381656,0.23741657,0.521647],"study_design_scores_gemma":[0.00016298842,0.0002297142,0.02334255,0.00035077208,0.00045095148,0.0029930864,0.002544877,0.26518714,0.13060872,0.07397275,0.4998486,0.00030783322],"about_ca_topic_score_codex":0.00804121,"about_ca_topic_score_gemma":0.018485416,"teacher_disagreement_score":0.00804121,"about_ca_system_score_codex":0.00069651776,"about_ca_system_score_gemma":0.0021626872,"threshold_uncertainty_score":0.024801433},"labels":[],"label_agreement":null},{"id":"W2806369456","doi":"","title":"Overview of TAC-KBP2016 Tri-lingual EDL and Its Impact on End-to-End KBP.","year":2016,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"End-to-end principle; Computer science; Dead end; Artificial intelligence; Mathematics; Geometry","score_opus":0.015409448958283294,"score_gpt":0.3218084272056579,"score_spread":0.3063989782473746,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2806369456","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011550717,0.014499953,0.66544765,0.006052178,0.0019734374,0.0014403287,0.049279835,0.18037604,0.06937979],"genre_scores_gemma":[0.042162772,0.008178441,0.68081546,0.002260536,0.0004439893,0.0011626564,0.20770645,0.023056217,0.034213416],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99277747,0.00195841,0.0009956881,0.0010031468,0.0028910527,0.00037433556],"domain_scores_gemma":[0.9836173,0.0044197896,0.000396975,0.004154462,0.006507308,0.0009041906],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009590578,0.001520691,0.0014264752,0.0050796126,0.0011773659,0.006586236,0.005258295,0.001898684,0.02733996],"category_scores_gemma":[0.023304043,0.0016190528,0.001167564,0.005281301,0.0011087127,0.0119590415,0.0055435756,0.00552894,0.025280144],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008171156,0.0006976414,0.0012567441,0.0030383859,0.0002228394,0.00048563664,0.00084315974,0.0074293795,0.014619839,0.040947292,0.43535003,0.4942919],"study_design_scores_gemma":[0.00013222487,0.00017942958,0.0008738068,0.00064077275,0.00010356223,0.0005118626,0.0003111108,0.056858625,0.016109984,0.021202683,0.9029191,0.0001568853],"about_ca_topic_score_codex":0.017969448,"about_ca_topic_score_gemma":0.018839471,"teacher_disagreement_score":0.02733996,"about_ca_system_score_codex":0.0028150533,"about_ca_system_score_gemma":0.005481104,"threshold_uncertainty_score":0.0914613},"labels":[],"label_agreement":null},{"id":"W2806418693","doi":"","title":"MSIIPL THU’s Slot-Filling Method for TAC-KBP 2015.","year":2015,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Mathematics; Computer science","score_opus":0.018584716169399548,"score_gpt":0.3284024280685134,"score_spread":0.30981771189911383,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2806418693","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005307865,0.00067404297,0.84052783,0.0008140371,0.0006938984,0.00033207456,0.011884683,0.109177686,0.030587869],"genre_scores_gemma":[0.09012358,0.00037175603,0.84347004,0.0004984602,0.00018882984,0.00057181314,0.022412244,0.014243488,0.028119767],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9978708,0.00055536395,0.00032485792,0.00046020217,0.0005915899,0.0001972158],"domain_scores_gemma":[0.99763227,0.0008311001,0.00009242052,0.0007838741,0.0005797365,0.00008058371],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018479835,0.0010456817,0.00088873046,0.0031247162,0.0014591911,0.0041529136,0.0024162554,0.0015367409,0.059310045],"category_scores_gemma":[0.009723764,0.0011046849,0.0015654169,0.002899217,0.0008868382,0.00591508,0.0037912973,0.0019609176,0.029830053],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00047348725,0.000102340906,0.00096431706,0.0012600687,0.00010892003,0.00037130076,0.0016053407,0.0027266194,0.0082787,0.0918799,0.24538578,0.64684325],"study_design_scores_gemma":[0.000115120776,0.00010056,0.00079961075,0.00042683684,0.00009257485,0.0006379192,0.000978103,0.073836304,0.033056233,0.13310988,0.75669837,0.00014858006],"about_ca_topic_score_codex":0.004597426,"about_ca_topic_score_gemma":0.007420653,"teacher_disagreement_score":0.059310045,"about_ca_system_score_codex":0.0010707541,"about_ca_system_score_gemma":0.002624301,"threshold_uncertainty_score":0.19841188},"labels":[],"label_agreement":null},{"id":"W2806617565","doi":"","title":"The IBM Systems for Trilingual Entity Discovery and Linking at TAC 2015.","year":2015,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"IBM; Computer science; Natural language processing","score_opus":0.0126383957649103,"score_gpt":0.2859319696838521,"score_spread":0.27329357391894177,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2806617565","genre_codex":"software","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0064517055,0.0027178866,0.28618887,0.0031993103,0.0023782807,0.00096501777,0.10556367,0.5505619,0.04197335],"genre_scores_gemma":[0.054794174,0.0014498003,0.49291077,0.0009443483,0.0008610977,0.0011978981,0.3407221,0.049550574,0.05756923],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99445313,0.0016913844,0.00037353009,0.0012230936,0.001766034,0.0004928891],"domain_scores_gemma":[0.9900435,0.0019497431,0.00038317227,0.0039547235,0.0024070449,0.0012616968],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011080209,0.0028757313,0.0025520665,0.005311598,0.0021822646,0.007601926,0.004554075,0.001758869,0.13033643],"category_scores_gemma":[0.0229875,0.002437695,0.0018868687,0.0056163124,0.0011004474,0.011157306,0.0063089966,0.004102303,0.12050016],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00070920633,0.0001516558,0.00069250405,0.00029301643,0.0001834276,0.00014704677,0.0002818766,0.0011415264,0.0018826033,0.010707582,0.9108688,0.072940744],"study_design_scores_gemma":[0.00092361175,0.00023540106,0.0024831176,0.00021944116,0.00017751,0.00031979388,0.00036301126,0.0626807,0.010680785,0.079538524,0.84215724,0.00022083348],"about_ca_topic_score_codex":0.015163506,"about_ca_topic_score_gemma":0.016298058,"teacher_disagreement_score":0.13033643,"about_ca_system_score_codex":0.0016911237,"about_ca_system_score_gemma":0.0047393236,"threshold_uncertainty_score":0.43601882},"labels":[],"label_agreement":null},{"id":"W2806710540","doi":"10.18653/v1/n18-4004","title":"A Generalized Knowledge Hunting Framework for the Winograd Schema Challenge","year":2018,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Schema (genetic algorithms); Computer science; Theoretical computer science; Artificial intelligence; Knowledge management; Algebra over a field; Mathematics; Machine learning; Pure mathematics","score_opus":0.03997580010602558,"score_gpt":0.3405327431500678,"score_spread":0.30055694304404224,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2806710540","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0133682275,0.0041738003,0.9339215,0.008200857,0.00048323124,0.0007683037,0.007930246,0.012137704,0.01901612],"genre_scores_gemma":[0.10893586,0.0016486538,0.85454243,0.0017865445,0.00025824292,0.00041933559,0.021494314,0.0015543875,0.009360127],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9953465,0.0015988678,0.00041092606,0.0009414376,0.001391596,0.00031076744],"domain_scores_gemma":[0.99469817,0.0017178032,0.00018849947,0.002230003,0.0007753016,0.00039019357],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0057463716,0.0009360344,0.0015739504,0.0036495952,0.0025856048,0.006690883,0.0068131504,0.0030854868,0.015113563],"category_scores_gemma":[0.015768768,0.0009071537,0.0024265354,0.0049775415,0.002017436,0.014642043,0.012100358,0.004914144,0.005541607],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005188188,0.0005055636,0.0022585024,0.00086023717,0.00033083028,0.0005495577,0.0008549986,0.021311233,0.0019885895,0.40170538,0.18232071,0.38679558],"study_design_scores_gemma":[0.00014291721,0.00007202974,0.00035063757,0.0002687092,0.00009753145,0.00042939422,0.0007211903,0.19602062,0.0023067384,0.66459835,0.13493796,0.000053996322],"about_ca_topic_score_codex":0.00982092,"about_ca_topic_score_gemma":0.02338065,"teacher_disagreement_score":0.015113563,"about_ca_system_score_codex":0.0015244227,"about_ca_system_score_gemma":0.003704047,"threshold_uncertainty_score":0.05055988},"labels":[],"label_agreement":null},{"id":"W2806769790","doi":"","title":"The Open Knowledge System for TAC KBP 2017.","year":2017,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science","score_opus":0.018678234132275616,"score_gpt":0.3278012789018057,"score_spread":0.3091230447695301,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2806769790","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010861229,0.0020997785,0.4768423,0.003288406,0.0017462627,0.0009517605,0.09855122,0.2971877,0.108471245],"genre_scores_gemma":[0.105710536,0.0020528592,0.3845918,0.0008075604,0.00065322156,0.0013920357,0.39238742,0.033540986,0.07886362],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9972307,0.0006935466,0.00036384794,0.00041546597,0.0010734856,0.0002228825],"domain_scores_gemma":[0.994372,0.0016294739,0.0002999583,0.001974161,0.0012342562,0.00049014925],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033709672,0.0010645214,0.0010106985,0.004261356,0.0014433346,0.0067346636,0.0028756917,0.0016286334,0.05100214],"category_scores_gemma":[0.020576173,0.0008831826,0.00093709544,0.003377948,0.0009558754,0.011014314,0.0063970527,0.002638539,0.04262365],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005508793,0.00022372238,0.0012328895,0.00079323514,0.00008482212,0.00045565193,0.0005587299,0.0031266636,0.0025351564,0.07052667,0.6652587,0.25465295],"study_design_scores_gemma":[0.00020849935,0.00006942475,0.0011438077,0.00047297293,0.00007151871,0.00041293734,0.0004053146,0.03757044,0.007625447,0.10760976,0.8442957,0.00011418176],"about_ca_topic_score_codex":0.0097516,"about_ca_topic_score_gemma":0.010628733,"teacher_disagreement_score":0.05100214,"about_ca_system_score_codex":0.0012204279,"about_ca_system_score_gemma":0.0048471736,"threshold_uncertainty_score":0.17061919},"labels":[],"label_agreement":null},{"id":"W2806846945","doi":"10.63317/4pm5rtxujcff","title":"Automatic Enrichment of Terminological Resources: the IATE RDF Example","year":2018,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Horizon 2020 Framework Programme; Universidad Politécnica de Madrid; Atomic Energy of Canada Limited; Science Foundation Ireland","keywords":"RDF; Library science; Computer science; Foundation (evidence); Political science; World Wide Web; Semantic Web","score_opus":0.02630978432693634,"score_gpt":0.2777227911253738,"score_spread":0.2514130067984375,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2806846945","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.47975263,0.0014061626,0.36627132,0.004558725,0.00053698185,0.0005841729,0.04585739,0.015416078,0.08561656],"genre_scores_gemma":[0.49715453,0.0006287274,0.4548385,0.00043167933,0.000105605825,0.00026261245,0.031672627,0.0014278837,0.0134778125],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99952674,0.0001476341,0.00004254823,0.00007130185,0.00017354869,0.00003816047],"domain_scores_gemma":[0.9990308,0.00041743115,0.00005073813,0.0002094818,0.00025399617,0.00003749403],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000691081,0.00034782712,0.00028165517,0.0016559787,0.0009324391,0.00067252363,0.0004882826,0.00074896275,0.0035522308],"category_scores_gemma":[0.0019910196,0.00021243942,0.0004986283,0.0014857952,0.00059894926,0.0019486888,0.0009769045,0.0007369294,0.0012581599],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017262391,0.0010425858,0.025837578,0.0032583808,0.0002112348,0.017376456,0.005605488,0.033382073,0.12415967,0.20424852,0.16171269,0.42143914],"study_design_scores_gemma":[0.00016885475,0.00012376579,0.015630504,0.00038704244,0.00012047026,0.006456933,0.0020344933,0.15417568,0.10762038,0.037744302,0.67539626,0.00014135412],"about_ca_topic_score_codex":0.004777713,"about_ca_topic_score_gemma":0.013184248,"teacher_disagreement_score":0.004777713,"about_ca_system_score_codex":0.00051798933,"about_ca_system_score_gemma":0.00054221443,"threshold_uncertainty_score":0.011883378},"labels":[],"label_agreement":null},{"id":"W2806889342","doi":"","title":"The YorkNRM Systems for Trilingual EDL Tasks at TAC KBP 2016.","year":2016,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Natural language processing","score_opus":0.008514998930767157,"score_gpt":0.26696746321558334,"score_spread":0.25845246428481616,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2806889342","genre_codex":"software","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.051687036,0.002777251,0.2237375,0.0027675752,0.0016091464,0.0013708614,0.10027417,0.55265355,0.063122965],"genre_scores_gemma":[0.18611231,0.0010367915,0.43741152,0.0008451061,0.00019874242,0.001694595,0.2872805,0.029174024,0.056246385],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99727756,0.0008684007,0.0003128879,0.0006445456,0.0006780981,0.00021849638],"domain_scores_gemma":[0.99601656,0.0011456143,0.00013114832,0.0012536251,0.0011877764,0.00026511593],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0042263125,0.0017174092,0.0009965607,0.0027427727,0.0011876961,0.0026005448,0.0036586884,0.001532512,0.057215102],"category_scores_gemma":[0.013082102,0.0010331544,0.00091839,0.001434749,0.00065222883,0.008354586,0.0049442356,0.002040009,0.03226869],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018008223,0.0003568222,0.0019431909,0.0017151575,0.00015130128,0.0004041764,0.0013809951,0.005546286,0.009332992,0.010940043,0.64404505,0.3223832],"study_design_scores_gemma":[0.0010392378,0.0004629476,0.0031895349,0.00048384714,0.00017181694,0.00047790605,0.0014238897,0.1430951,0.032683432,0.021570243,0.7951001,0.00030188606],"about_ca_topic_score_codex":0.02944536,"about_ca_topic_score_gemma":0.048113253,"teacher_disagreement_score":0.057215102,"about_ca_system_score_codex":0.0018191749,"about_ca_system_score_gemma":0.0037688497,"threshold_uncertainty_score":0.19140357},"labels":[],"label_agreement":null},{"id":"W2806901754","doi":"10.18653/v1/s18-2016","title":"Coarse Lexical Frame Acquisition at the Syntax–Semantics Interface Using a Latent-Variable PCFG Model","year":2018,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Deutsche Forschungsgemeinschaft","keywords":"Computer science; Natural language processing; Artificial intelligence; Rule-based machine translation; Bayesian network; Frame (networking); Merge (version control); Latent variable; Syntax; Semantics (computer science); Programming language; Information retrieval","score_opus":0.027271879303329298,"score_gpt":0.30509646108477245,"score_spread":0.27782458178144315,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2806901754","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015267628,0.000057731137,0.980335,0.00012218546,0.000013588481,0.000072113085,0.00025820918,0.002706613,0.0011669392],"genre_scores_gemma":[0.30216843,0.00009189918,0.6921954,0.00015382147,0.00003243487,0.00025621965,0.001372558,0.0010618911,0.0026673072],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987669,0.00040428786,0.000049043738,0.00046928207,0.00021370454,0.00009681699],"domain_scores_gemma":[0.9977816,0.0012878319,0.00011699695,0.00040119694,0.0003367591,0.00007563904],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014670464,0.0009583667,0.0009265375,0.0016951244,0.00076275965,0.0017925578,0.002071215,0.0013517207,0.00606952],"category_scores_gemma":[0.005399259,0.00080314284,0.0011944666,0.0012362124,0.001236375,0.0037642245,0.0017566645,0.0024159304,0.0017943665],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003967799,0.00032025552,0.004924243,0.00034030084,0.00015483738,0.00063230493,0.0019616513,0.20318757,0.048082232,0.12367085,0.013149589,0.6031794],"study_design_scores_gemma":[0.000023553011,0.0000310519,0.00077992474,0.000018992183,0.000025778845,0.00007521683,0.000083753184,0.93424714,0.008202226,0.05294152,0.0035406624,0.000030226201],"about_ca_topic_score_codex":0.010968843,"about_ca_topic_score_gemma":0.018414047,"teacher_disagreement_score":0.010968843,"about_ca_system_score_codex":0.0011963955,"about_ca_system_score_gemma":0.0021051844,"threshold_uncertainty_score":0.021809995},"labels":[],"label_agreement":null},{"id":"W2806927221","doi":"","title":"The NYU Cold Start System for TAC 2015.","year":2015,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Operating system","score_opus":0.012179140874915995,"score_gpt":0.2743210494270873,"score_spread":0.2621419085521713,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2806927221","genre_codex":"software","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009407215,0.00096802856,0.08343887,0.0010406577,0.001970939,0.0007092387,0.26394367,0.5871377,0.051383637],"genre_scores_gemma":[0.0380439,0.00034831214,0.0816812,0.000491743,0.00035929575,0.0010450305,0.7937201,0.041072484,0.043237887],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9979913,0.0005128429,0.00016677953,0.00032908356,0.0007002382,0.00029973887],"domain_scores_gemma":[0.995631,0.0006719297,0.00022848364,0.0015767642,0.0013961818,0.0004957417],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033490562,0.0016596344,0.0012809699,0.00223887,0.0013581197,0.002877211,0.0026161128,0.0012545175,0.07760987],"category_scores_gemma":[0.011758678,0.0008632532,0.00093150855,0.0022643663,0.00041603835,0.0037730683,0.002686235,0.0019834908,0.09910681],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006306013,0.000059956135,0.0007413419,0.00014110649,0.000036609996,0.000047658636,0.000076888864,0.00084052037,0.0008233931,0.0019172803,0.9746927,0.01999186],"study_design_scores_gemma":[0.0004478526,0.00015055473,0.00415415,0.0001642672,0.000058492402,0.00011584358,0.00014358608,0.04511645,0.008218508,0.014283043,0.92699265,0.00015452564],"about_ca_topic_score_codex":0.019852882,"about_ca_topic_score_gemma":0.031186987,"teacher_disagreement_score":0.07760987,"about_ca_system_score_codex":0.0010849404,"about_ca_system_score_gemma":0.0029871827,"threshold_uncertainty_score":0.25963086},"labels":[],"label_agreement":null},{"id":"W2807086563","doi":"","title":"ZJU Participation in TAC 2016 EDL task.","year":2016,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Task (project management); Computer science; Systems engineering; Engineering","score_opus":0.007137538236583994,"score_gpt":0.27888593631638076,"score_spread":0.2717483980797968,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2807086563","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20912685,0.0029781268,0.038368892,0.014941809,0.010231489,0.002187063,0.3513396,0.06720788,0.30361825],"genre_scores_gemma":[0.25160086,0.0004420957,0.039807666,0.0025810504,0.00077776756,0.0015068699,0.5552397,0.005505258,0.14253867],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9961249,0.001613502,0.00025295393,0.0005748246,0.000907119,0.0005267163],"domain_scores_gemma":[0.9926475,0.002020521,0.00018290778,0.0017172267,0.0022184006,0.0012134578],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0048472746,0.0013798224,0.0011826488,0.0012072473,0.0024122775,0.0022467128,0.0018336908,0.0021002195,0.057062548],"category_scores_gemma":[0.01492634,0.00028915785,0.00056820887,0.0010800837,0.0004974524,0.0036638253,0.004606241,0.001717993,0.042363565],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013115893,0.00041622718,0.0022091821,0.0006151923,0.00003944343,0.00029542285,0.000889627,0.0009876618,0.003739331,0.0026609644,0.905983,0.080852374],"study_design_scores_gemma":[0.0005618,0.0003830566,0.007926535,0.00025613353,0.00006112876,0.00033171137,0.0021070456,0.013563627,0.011531358,0.00863845,0.9545307,0.000108410946],"about_ca_topic_score_codex":0.012933633,"about_ca_topic_score_gemma":0.029339384,"teacher_disagreement_score":0.057062548,"about_ca_system_score_codex":0.0012157324,"about_ca_system_score_gemma":0.0031069545,"threshold_uncertainty_score":0.19089323},"labels":[],"label_agreement":null},{"id":"W2807096580","doi":"10.63317/3x9fjezosbv9","title":"Building a Constraint Grammar Parser for Plains Cree Verbs and Arguments","year":2018,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Parsing; Grammar; Constraint (computer-aided design); Natural language processing; Programming language; Linguistics; Artificial intelligence; Mathematics; Philosophy","score_opus":0.017928815541839272,"score_gpt":0.29584388899816466,"score_spread":0.2779150734563254,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2807096580","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012383081,0.00016707572,0.9143845,0.001223862,0.00031848229,0.0005540066,0.008073358,0.049791817,0.013103786],"genre_scores_gemma":[0.10288605,0.00025909086,0.8564412,0.00062492664,0.00012132878,0.0003486156,0.015519447,0.012251636,0.0115476735],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99842095,0.00021911581,0.00013524343,0.00056820764,0.0005157983,0.00014070902],"domain_scores_gemma":[0.99525636,0.0026607995,0.00016266126,0.00051807106,0.0012645747,0.0001374111],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017555426,0.0014224733,0.0015568628,0.0021026726,0.0015789948,0.004043076,0.0026252212,0.0025421423,0.029380254],"category_scores_gemma":[0.0079484815,0.0025835312,0.0022094408,0.0021973618,0.0014440425,0.005535294,0.0032692947,0.004391086,0.012322066],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036610788,0.0004457115,0.005225315,0.001466725,0.0002114811,0.003079912,0.002421494,0.03030451,0.054147035,0.2742327,0.19723956,0.43085942],"study_design_scores_gemma":[0.0002703848,0.00010549827,0.0020930073,0.00027333523,0.00024084213,0.0013902312,0.00098557,0.49841088,0.07569895,0.17388344,0.24638373,0.00026409992],"about_ca_topic_score_codex":0.015329132,"about_ca_topic_score_gemma":0.021933023,"teacher_disagreement_score":0.029380254,"about_ca_system_score_codex":0.0019488612,"about_ca_system_score_gemma":0.005716668,"threshold_uncertainty_score":0.09828675},"labels":[],"label_agreement":null},{"id":"W2807137429","doi":"","title":"SoochowNLP Team System Description for 2016 KBP Slot Filling and Nugget Detection Tasks.","year":2016,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Real-time computing; Artificial intelligence","score_opus":0.007929024833840781,"score_gpt":0.23664891234295943,"score_spread":0.22871988750911865,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2807137429","genre_codex":"software","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013312887,0.00060770963,0.13521348,0.00092714734,0.00051363994,0.0014165271,0.23232757,0.57559526,0.04008582],"genre_scores_gemma":[0.067387424,0.000378722,0.14042382,0.0010721992,0.00014045494,0.0027979608,0.71306986,0.047934093,0.026795544],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99831533,0.00030391727,0.00023773755,0.00043751716,0.00053115387,0.00017433289],"domain_scores_gemma":[0.9968437,0.00078719837,0.00011962167,0.0006832254,0.0013034432,0.0002626974],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018270427,0.0015965786,0.0008847003,0.0021677446,0.00088036415,0.002192538,0.0030310047,0.001426114,0.05318293],"category_scores_gemma":[0.00637184,0.0009758994,0.0009048782,0.0015211754,0.00030880436,0.0026909234,0.0015439388,0.0016375342,0.054189734],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009333281,0.00015242428,0.0017126951,0.0013075044,0.00010017051,0.00031674997,0.00027649078,0.0033918417,0.008945782,0.0022253725,0.9212402,0.05939742],"study_design_scores_gemma":[0.0008139822,0.00029498327,0.0055173393,0.00033131827,0.0001369463,0.00090725604,0.00032698942,0.0788926,0.043534983,0.0057020793,0.8633227,0.00021882706],"about_ca_topic_score_codex":0.013641069,"about_ca_topic_score_gemma":0.013765638,"teacher_disagreement_score":0.05318293,"about_ca_system_score_codex":0.0011622189,"about_ca_system_score_gemma":0.0028350104,"threshold_uncertainty_score":0.17791462},"labels":[],"label_agreement":null},{"id":"W2807182733","doi":"10.18653/v1/w18-1605","title":"Cross-corpus Native Language Identification via Statistical Embedding","year":2018,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Thomson Reuters (Canada)","funders":"Qatar National Research Fund; Fonds National de la Recherche Luxembourg; Qatar Foundation","keywords":"Computer science; Natural language processing; Artificial intelligence; Identification (biology); Task (project management); Embedding; Language identification; Indonesian; Arabic; Representation (politics); Language model; Linguistics; Natural language; Engineering","score_opus":0.011191389769946252,"score_gpt":0.3515243750749443,"score_spread":0.340332985304998,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2807182733","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16752408,0.0010582992,0.81926376,0.00032433143,0.00023944913,0.0001041327,0.00065796386,0.006371481,0.0044565024],"genre_scores_gemma":[0.784609,0.0004550607,0.20184204,0.00023011924,0.00014888082,0.00014390636,0.0035554485,0.00072380924,0.008291667],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985,0.0005201254,0.000087568755,0.0005303073,0.00024294722,0.00011906385],"domain_scores_gemma":[0.9966185,0.0015267992,0.00022108539,0.00080427196,0.0007256886,0.00010361125],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019960657,0.0013070606,0.00081696926,0.0017341189,0.0004660677,0.0014748598,0.0010022388,0.0009225726,0.0023635274],"category_scores_gemma":[0.0065738093,0.00040867436,0.00065231,0.0011192003,0.00053020974,0.0035005184,0.0025086233,0.0014271084,0.0024877256],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005945941,0.0004998306,0.009243605,0.00024678817,0.0003439485,0.00065399363,0.00060339516,0.083715424,0.06974175,0.0058839032,0.005748516,0.8227242],"study_design_scores_gemma":[0.000017561913,0.0002154105,0.004357282,0.000031318574,0.00007371213,0.0005721138,0.0003738117,0.9441694,0.03439653,0.009664328,0.0060587386,0.000069723515],"about_ca_topic_score_codex":0.0016695696,"about_ca_topic_score_gemma":0.0025496378,"teacher_disagreement_score":0.0023635274,"about_ca_system_score_codex":0.00028524885,"about_ca_system_score_gemma":0.0007486602,"threshold_uncertainty_score":0.01055634},"labels":[],"label_agreement":null},{"id":"W2807293832","doi":"10.18653/v1/s18-1118","title":"SUNNYNLP at SemEval-2018 Task 10: A Support-Vector-Machine-Based Method for Detecting Semantic Difference using Taxonomy and Word Embedding Features","year":2018,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Chinese University of Hong Kong; University of Toronto; Universiteit Utrecht","keywords":"SemEval; Computer science; Support vector machine; Discriminative model; Taxonomy (biology); Word embedding; Artificial intelligence; Natural language processing; Task (project management); Embedding; Word (group theory); Mathematics","score_opus":0.035713310776311694,"score_gpt":0.33118054505907013,"score_spread":0.29546723428275845,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2807293832","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1630262,0.0037126383,0.47067517,0.002109875,0.0026345712,0.002260414,0.09996145,0.2311225,0.024497334],"genre_scores_gemma":[0.2432757,0.00058826664,0.5775156,0.00067673274,0.00021105116,0.0015903796,0.16269505,0.0043327007,0.009114507],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9946813,0.0011319035,0.0005484197,0.001883097,0.0014400156,0.0003152516],"domain_scores_gemma":[0.99448895,0.002065634,0.00040586147,0.0014852141,0.0012198259,0.0003345268],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004337295,0.002979521,0.0019888189,0.00440951,0.0012671708,0.0025727323,0.00436822,0.0031073387,0.014242288],"category_scores_gemma":[0.013895223,0.0008219202,0.0016657551,0.0026963777,0.0007833656,0.00928857,0.005569259,0.003333759,0.009518417],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012225459,0.0009526937,0.0085005555,0.0020647992,0.00032568825,0.00058844296,0.0008611041,0.00457361,0.02219459,0.007131713,0.25988492,0.6916993],"study_design_scores_gemma":[0.00072167546,0.0014063249,0.018562311,0.00046947936,0.00024741027,0.0026508593,0.0018033733,0.54152286,0.096251965,0.04552129,0.2905098,0.0003326771],"about_ca_topic_score_codex":0.005041794,"about_ca_topic_score_gemma":0.007970613,"teacher_disagreement_score":0.014242288,"about_ca_system_score_codex":0.0014554898,"about_ca_system_score_gemma":0.0019822074,"threshold_uncertainty_score":0.04764521},"labels":[],"label_agreement":null},{"id":"W2807312676","doi":"10.63317/2vxw58tvfbm4","title":"Integrating Generative Lexicon Event Structures into VerbNet","year":2018,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Guelph","funders":"","keywords":"Computer science; Generative grammar; Lexicon; Event (particle physics); Artificial intelligence; Natural language processing","score_opus":0.011342971524478716,"score_gpt":0.30450558369575426,"score_spread":0.29316261217127554,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2807312676","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012472204,0.00020755886,0.9618267,0.00042537696,0.00021878202,0.00014639049,0.0017957027,0.012628613,0.010278678],"genre_scores_gemma":[0.43257588,0.0005026305,0.54425275,0.00030837877,0.00019681959,0.0001854296,0.009223984,0.0035534825,0.009200645],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99897003,0.00032585155,0.00008766898,0.00031082152,0.00022762103,0.00007814152],"domain_scores_gemma":[0.99799865,0.0010638491,0.00010100515,0.00038100424,0.00038186111,0.00007355735],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014533369,0.000902083,0.000709182,0.0025429607,0.00081128126,0.0037558617,0.0015105818,0.0009179324,0.015553054],"category_scores_gemma":[0.005857869,0.00092360243,0.0010598802,0.0021988594,0.00076724595,0.006408817,0.0022638203,0.001681201,0.005248803],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005150776,0.000338462,0.0034403787,0.00063586066,0.00019047306,0.0006340071,0.0011827226,0.04682715,0.016725393,0.34939247,0.02585275,0.55426526],"study_design_scores_gemma":[0.000059088143,0.000055467375,0.0008440541,0.00011720157,0.00012643587,0.00019255208,0.00031319645,0.5041172,0.013620203,0.4384143,0.042079046,0.00006136359],"about_ca_topic_score_codex":0.0054774513,"about_ca_topic_score_gemma":0.013567082,"teacher_disagreement_score":0.015553054,"about_ca_system_score_codex":0.0013353436,"about_ca_system_score_gemma":0.0015448227,"threshold_uncertainty_score":0.052030146},"labels":[],"label_agreement":null},{"id":"W2807399664","doi":"10.63317/5894atgjvr8a","title":"SPADE: Evaluation Dataset for Monolingual Phrase Alignment","year":2018,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Open Text (Canada)","funders":"","keywords":"Computer science; Phrase; Natural language processing; Artificial intelligence; Speech recognition; Information retrieval","score_opus":0.042956909522173264,"score_gpt":0.3728724622383085,"score_spread":0.32991555271613526,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2807399664","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.049525425,0.0027801937,0.02014691,0.00093851244,0.0014025543,0.0014927209,0.8553248,0.050076373,0.01831256],"genre_scores_gemma":[0.0071614687,0.00018199554,0.014768236,0.00017501923,0.0000553273,0.00044326237,0.9735411,0.0007403039,0.0029332396],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99461323,0.0012840703,0.0007525787,0.0014660599,0.0013588209,0.0005253265],"domain_scores_gemma":[0.99231064,0.0017625747,0.00040901493,0.0021784515,0.002485574,0.0008537743],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0038497932,0.005257766,0.0026085258,0.007280873,0.0031074716,0.002331433,0.0059504174,0.0039706137,0.025188802],"category_scores_gemma":[0.011108936,0.0012255727,0.0024219966,0.005943472,0.0011394791,0.004811015,0.0059572756,0.0032475158,0.04060262],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010985449,0.0009853512,0.0032106806,0.0021424643,0.00043496117,0.00043153652,0.00022569275,0.002196588,0.009361541,0.0018764186,0.8992683,0.07876789],"study_design_scores_gemma":[0.00433072,0.001540844,0.029211083,0.00083263445,0.000898922,0.0030378352,0.0017946081,0.05380692,0.04126592,0.008600393,0.85416,0.0005200555],"about_ca_topic_score_codex":0.02105367,"about_ca_topic_score_gemma":0.05229866,"teacher_disagreement_score":0.025188802,"about_ca_system_score_codex":0.0016066739,"about_ca_system_score_gemma":0.00546275,"threshold_uncertainty_score":0.084264934},"labels":[],"label_agreement":null},{"id":"W2807400292","doi":"","title":"TinkerBell: Cross-lingual Cold-Start Knowledge Base Construction.","year":2017,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Base (topology); Computer science; Knowledge base; Artificial intelligence; Mathematics","score_opus":0.012080456819392963,"score_gpt":0.30629773494844537,"score_spread":0.2942172781290524,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2807400292","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004013006,0.0005443302,0.84584206,0.00035875075,0.00030444982,0.00035733607,0.00578288,0.12752385,0.015273379],"genre_scores_gemma":[0.07686552,0.0006093377,0.8366255,0.0006193869,0.00010182385,0.0007046836,0.04024623,0.02148584,0.022741716],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99727637,0.00068374554,0.00025176466,0.0006456762,0.00085931434,0.00028317567],"domain_scores_gemma":[0.9941332,0.0026819764,0.00015952163,0.0016733141,0.0011165846,0.00023542585],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028807798,0.001207494,0.001194965,0.0031192983,0.0015195885,0.0041288184,0.004811147,0.0019257299,0.040981803],"category_scores_gemma":[0.015389326,0.0016154659,0.002130044,0.0029790422,0.0013472127,0.0066417865,0.00961405,0.0033687237,0.02114539],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006581369,0.00041908099,0.002683898,0.001993935,0.00050802034,0.0013123081,0.002133041,0.0064199455,0.0128731495,0.115546525,0.2829656,0.57248634],"study_design_scores_gemma":[0.00023808304,0.00017812339,0.0017769592,0.00070552505,0.0003587949,0.0009270707,0.001736556,0.12531556,0.058207586,0.23760399,0.57270324,0.00024847855],"about_ca_topic_score_codex":0.0071797874,"about_ca_topic_score_gemma":0.016895339,"teacher_disagreement_score":0.040981803,"about_ca_system_score_codex":0.0011622327,"about_ca_system_score_gemma":0.0032592707,"threshold_uncertainty_score":0.13709778},"labels":[],"label_agreement":null},{"id":"W2807441601","doi":"","title":"Overview of Linguistic Resources for the TAC KBP 2015 Evaluations: Methodologies and Results.","year":2015,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":48,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Linguistics; Natural language processing; Artificial intelligence; Philosophy","score_opus":0.10060836722971492,"score_gpt":0.41520445656747484,"score_spread":0.31459608933775995,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2807441601","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19054674,0.035751402,0.18327418,0.0055922554,0.0015450903,0.025955236,0.36233896,0.0424852,0.1525109],"genre_scores_gemma":[0.2140833,0.0057001426,0.29879063,0.0011793529,0.00029243724,0.0196595,0.4382547,0.0038592534,0.018180663],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.97406363,0.012470925,0.003000952,0.0017925982,0.0077524143,0.00091939315],"domain_scores_gemma":[0.9419708,0.023056734,0.0019323073,0.00665133,0.023915416,0.0024733848],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.021824274,0.0021610383,0.0017144573,0.013496169,0.0028699902,0.004995826,0.004658085,0.0020395156,0.018128464],"category_scores_gemma":[0.07235658,0.0009474303,0.001252483,0.009009898,0.001159618,0.007370712,0.0058726585,0.0025181868,0.013026405],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004246365,0.0030116173,0.010336243,0.016846744,0.0010393746,0.00061253493,0.003680299,0.012527798,0.0127164805,0.0084858,0.33565724,0.5908395],"study_design_scores_gemma":[0.0026126215,0.0025841447,0.0363195,0.0082339365,0.0027089003,0.0013751822,0.007168818,0.05654673,0.05172282,0.014900573,0.8150134,0.00081332336],"about_ca_topic_score_codex":0.027085708,"about_ca_topic_score_gemma":0.032076113,"teacher_disagreement_score":0.027085708,"about_ca_system_score_codex":0.0031251055,"about_ca_system_score_gemma":0.008026792,"threshold_uncertainty_score":0.11541915},"labels":[],"label_agreement":null},{"id":"W2807448824","doi":"10.63317/27a9bt4m4kvq","title":"Constructing a Lexicon of Relational Nouns","year":2018,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Lexicon; Computer science; Noun; Natural language processing; Artificial intelligence; Linguistics; Proper noun; Philosophy","score_opus":0.018468038104294548,"score_gpt":0.278183330385637,"score_spread":0.25971529228134244,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2807448824","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06521881,0.00042768178,0.8763891,0.001001706,0.00031131605,0.00082696445,0.008429156,0.021188615,0.026206674],"genre_scores_gemma":[0.24669625,0.0005380129,0.7245852,0.00023163072,0.00008433243,0.00036964146,0.01833462,0.0026837327,0.0064765858],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9989042,0.00024206158,0.00016814336,0.00034477835,0.00023924717,0.000101598715],"domain_scores_gemma":[0.9978861,0.00092708116,0.00011206519,0.00028147106,0.0006872794,0.00010589992],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007173489,0.0009181757,0.001086748,0.0038469892,0.0015462972,0.004243289,0.0016325737,0.0010418752,0.010320732],"category_scores_gemma":[0.004922613,0.0014149527,0.0017888842,0.0031657037,0.0009722993,0.0060312147,0.0030132988,0.0019167748,0.007304161],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00041192086,0.00045044947,0.009110639,0.0019725577,0.00022062333,0.0036503235,0.0047691446,0.013955369,0.075600676,0.26833728,0.059905224,0.5616158],"study_design_scores_gemma":[0.00024647402,0.00027646197,0.005335649,0.00064325816,0.0006926287,0.0030778402,0.006276228,0.2968876,0.0699634,0.3492903,0.26697585,0.0003342014],"about_ca_topic_score_codex":0.0057630436,"about_ca_topic_score_gemma":0.010840029,"teacher_disagreement_score":0.010320732,"about_ca_system_score_codex":0.0014744968,"about_ca_system_score_gemma":0.0027934439,"threshold_uncertainty_score":0.03452623},"labels":[],"label_agreement":null},{"id":"W2807453564","doi":"10.63317/2whrzuzomqc3","title":"Modeling Northern Haida Verb Morphology","year":2018,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Morphology (biology); Verb; Computer science; Artificial intelligence; Natural language processing; Linguistics; Geology; Paleontology; Philosophy","score_opus":0.016362288145748506,"score_gpt":0.2705758485018908,"score_spread":0.2542135603561423,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2807453564","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5227101,0.00025857703,0.4172341,0.00051604706,0.00009491087,0.00008624724,0.0012524675,0.002179447,0.05566815],"genre_scores_gemma":[0.9379009,0.00009689371,0.048859775,0.000028435721,0.000015831649,0.00003470342,0.0008047466,0.0003644418,0.01189415],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9998833,0.000034305092,0.0000044188196,0.000043753993,0.000017782591,0.000016469185],"domain_scores_gemma":[0.99968934,0.00015490194,0.00002205388,0.000036953912,0.000079580415,0.00001715857],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00032575618,0.0004229308,0.00032172227,0.0004453755,0.0006416442,0.0018919782,0.00061818736,0.0004383658,0.0074959933],"category_scores_gemma":[0.0016775301,0.0004333441,0.0006222077,0.00039482623,0.00036803557,0.0018968298,0.00044760166,0.0006427097,0.0014406953],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003437933,0.00015577649,0.019540012,0.00018384922,0.00008452395,0.0006791555,0.0014655042,0.5351398,0.017210908,0.21907489,0.008156728,0.197965],"study_design_scores_gemma":[0.000007825735,0.000014444665,0.0014950921,0.000008702259,0.000014896794,0.000061305196,0.00015232133,0.9701304,0.0013138716,0.02372449,0.0030686704,0.000007928743],"about_ca_topic_score_codex":0.022913245,"about_ca_topic_score_gemma":0.047696624,"teacher_disagreement_score":0.022913245,"about_ca_system_score_codex":0.0009623491,"about_ca_system_score_gemma":0.0010066848,"threshold_uncertainty_score":0.045559764},"labels":[],"label_agreement":null},{"id":"W2807600698","doi":"","title":"Description of the BOUN System for the Trilingual Entity Detection and Linking Tasks at TAC KBP 2017.","year":2017,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence","score_opus":0.01522562215205321,"score_gpt":0.27106895817904836,"score_spread":0.25584333602699516,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2807600698","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008037678,0.0010373775,0.531051,0.0010761989,0.00066801836,0.0020661885,0.040259864,0.4002945,0.015509091],"genre_scores_gemma":[0.07072348,0.0009133504,0.64349717,0.0013791021,0.00023647277,0.0038504694,0.21714096,0.03393583,0.028323183],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99840814,0.00024278244,0.00021010288,0.00045580528,0.0005192071,0.00016411587],"domain_scores_gemma":[0.99780124,0.00041840566,0.00006701634,0.0006470331,0.0008147396,0.00025156033],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021573314,0.001959633,0.0015939395,0.0034772144,0.0012924857,0.0040732916,0.004691794,0.0017138836,0.048972752],"category_scores_gemma":[0.0055370154,0.0018854807,0.0011982146,0.0023277127,0.00073563977,0.004022528,0.0034850086,0.0032024544,0.052258328],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013531742,0.0006727482,0.004323547,0.002217903,0.00051481236,0.001351647,0.00095770386,0.01493996,0.04616791,0.014877396,0.59385145,0.31877175],"study_design_scores_gemma":[0.0005914473,0.0003017516,0.005204942,0.0003641239,0.0002836706,0.0016391475,0.0005719134,0.21424781,0.06719851,0.024721902,0.6844617,0.00041302884],"about_ca_topic_score_codex":0.02158381,"about_ca_topic_score_gemma":0.02132983,"teacher_disagreement_score":0.048972752,"about_ca_system_score_codex":0.0012032556,"about_ca_system_score_gemma":0.0028046013,"threshold_uncertainty_score":0.16383016},"labels":[],"label_agreement":null},{"id":"W2807690984","doi":"10.1515/cllt-2017-0031","title":"Dependency profiles in the large-scale analysis of discourse connectives","year":2018,"lang":"en","type":"article","venue":"Corpus Linguistics and Linguistic Theory","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University; Brock University","funders":"Turun Yliopisto","keywords":"Dependency (UML); Computer science; Syntax; Natural language processing; Scope (computer science); Artificial intelligence; Linguistics; Cluster analysis; Focus (optics); Annotation","score_opus":0.00909766402107549,"score_gpt":0.2919299757573142,"score_spread":0.28283231173623874,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2807690984","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.89311427,0.00038403185,0.10123186,0.00011618293,0.000016337975,0.00013479937,0.0019174218,0.0002895597,0.0027956609],"genre_scores_gemma":[0.96506184,0.000084273604,0.03331181,0.000011103422,0.000008862354,0.00012855805,0.001068583,0.000046615736,0.00027828212],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99712247,0.0015990529,0.0001958342,0.0005567312,0.00042180068,0.0001040025],"domain_scores_gemma":[0.98065954,0.015795518,0.0012671694,0.0009758236,0.0010086865,0.00029337956],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027526973,0.0003076451,0.00038880706,0.0054364693,0.0011186486,0.0014600549,0.0005216671,0.00045809735,0.0016902561],"category_scores_gemma":[0.02017612,0.000311382,0.0003720756,0.005369746,0.00085651217,0.0023439082,0.0015862739,0.0007139984,0.00039151873],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019589346,0.00062845566,0.34148824,0.0016046172,0.00053827325,0.0025898556,0.02976373,0.036597136,0.07716006,0.059785917,0.0053761634,0.4425086],"study_design_scores_gemma":[0.000066432825,0.00028543817,0.44575045,0.00028565087,0.000279409,0.0018609713,0.013373015,0.40163377,0.04212883,0.071230754,0.02290628,0.00019888922],"about_ca_topic_score_codex":0.0026166358,"about_ca_topic_score_gemma":0.0029773503,"teacher_disagreement_score":0.0054364693,"about_ca_system_score_codex":0.00069492875,"about_ca_system_score_gemma":0.0006118932,"threshold_uncertainty_score":0.014557779},"labels":[],"label_agreement":null},{"id":"W2807692999","doi":"","title":"Illinois CCG Entity Discovery and Linking, Event Nugget Detection and Co-reference, and Slot Filler Validation Systems for TAC 2016.","year":2016,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Event (particle physics); Filler (materials); Computer science; Materials science; Composite material; Physics","score_opus":0.009420556148873268,"score_gpt":0.2642500822046072,"score_spread":0.25482952605573395,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2807692999","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01977891,0.0013350445,0.45317274,0.003979303,0.0015402611,0.001053626,0.031698644,0.43042833,0.057013147],"genre_scores_gemma":[0.15225717,0.0006407547,0.6041706,0.0012630601,0.00045655182,0.0007713653,0.15686911,0.01987996,0.06369141],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9940468,0.0017701433,0.00042551677,0.001113418,0.0021938586,0.0004502064],"domain_scores_gemma":[0.9882824,0.0021999402,0.00037351414,0.0050797146,0.0033938526,0.00067073206],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009316894,0.0014809106,0.0008523258,0.005753593,0.0023423075,0.003992083,0.003009253,0.0019997861,0.030590136],"category_scores_gemma":[0.017743919,0.0010133093,0.0010282055,0.0035462405,0.0013035298,0.008106376,0.00382587,0.0028772452,0.022182599],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009951061,0.0003777908,0.0069620186,0.00030381628,0.00013853499,0.00042484625,0.00073346845,0.008279635,0.008104255,0.033138204,0.6678919,0.27265042],"study_design_scores_gemma":[0.00022416166,0.0002447885,0.004715226,0.00021346066,0.00011175556,0.0005088931,0.00033877726,0.2499069,0.03378443,0.06037308,0.6493777,0.00020079702],"about_ca_topic_score_codex":0.037302718,"about_ca_topic_score_gemma":0.0572155,"teacher_disagreement_score":0.037302718,"about_ca_system_score_codex":0.0023995508,"about_ca_system_score_gemma":0.0060381135,"threshold_uncertainty_score":0.10233426},"labels":[],"label_agreement":null},{"id":"W2807695162","doi":"","title":"Adept Automatic Knowledge Discovery System for Cold Start Knowledge Base Population.","year":2017,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Knowledge base; Computer science; Knowledge extraction; Population; Base (topology); Data science; Artificial intelligence; Mathematics; Medicine","score_opus":0.013232088479578484,"score_gpt":0.29039802800860864,"score_spread":0.2771659395290302,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2807695162","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.076639116,0.0029206031,0.8549996,0.00075542176,0.0006214431,0.0014552195,0.008126989,0.03932918,0.01515243],"genre_scores_gemma":[0.1573119,0.00059714215,0.8089971,0.0005923703,0.00016280213,0.00087445945,0.016406002,0.0005270661,0.014531176],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99884987,0.00019850938,0.00012230648,0.00031349747,0.0004220296,0.00009379065],"domain_scores_gemma":[0.9976528,0.0008338968,0.00011267708,0.0002996595,0.00091597653,0.00018502361],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016332779,0.00072373624,0.0012569346,0.0048299143,0.0013510333,0.0019639453,0.0031005281,0.0015016642,0.007034208],"category_scores_gemma":[0.0051688,0.0004123004,0.0010824404,0.0025439686,0.00036771188,0.0024412933,0.001760086,0.0012798797,0.004101501],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006445352,0.0007720181,0.0059495578,0.0006339169,0.0003429492,0.0006294265,0.00026454305,0.009416939,0.017409556,0.006754964,0.06084508,0.8963365],"study_design_scores_gemma":[0.0005720733,0.0008501196,0.006389999,0.00026940112,0.00091624557,0.0024850604,0.0008486332,0.7834184,0.066490926,0.032212276,0.10536653,0.00018034599],"about_ca_topic_score_codex":0.003754493,"about_ca_topic_score_gemma":0.00823234,"teacher_disagreement_score":0.007034208,"about_ca_system_score_codex":0.0008940495,"about_ca_system_score_gemma":0.0024106072,"threshold_uncertainty_score":0.023531795},"labels":[],"label_agreement":null},{"id":"W2808069598","doi":"10.1007/978-3-319-93782-3_29","title":"Cross-Linguistic Projection for French-Vietnamese Named Entity Translation","year":2018,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Computer science; Vietnamese; Translation (biology); Natural language processing; Artificial intelligence; Projection (relational algebra); Machine translation; Linguistics; Algorithm; Philosophy","score_opus":0.02275696068570322,"score_gpt":0.3085441165539686,"score_spread":0.28578715586826536,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2808069598","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09432004,0.0030214821,0.81197375,0.0012508435,0.0011727212,0.00038606903,0.009691032,0.03035448,0.047829583],"genre_scores_gemma":[0.48705876,0.0017562596,0.44556612,0.00028650006,0.00022814106,0.00030562875,0.036835417,0.0027250582,0.02523814],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9989404,0.0003471322,0.00011928381,0.00029780442,0.0001669372,0.00012833107],"domain_scores_gemma":[0.9988243,0.00031842105,0.000050932158,0.00029620863,0.000449975,0.000060110062],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013680051,0.0009834784,0.0007013678,0.0011150707,0.0011462528,0.002231251,0.0008465576,0.00082276884,0.019658279],"category_scores_gemma":[0.0023215837,0.0005503422,0.000946199,0.0020114789,0.00044053528,0.0027105836,0.0030009572,0.0011210841,0.011376893],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009786759,0.00026758687,0.0022841962,0.00089501997,0.00017476827,0.0018211463,0.0011639324,0.0083417045,0.048225496,0.04127864,0.05991805,0.8346508],"study_design_scores_gemma":[0.0002080281,0.0005991269,0.009305648,0.00033888043,0.00039798295,0.0035812608,0.002851683,0.4585558,0.21905208,0.081223726,0.22364543,0.00024029054],"about_ca_topic_score_codex":0.005382581,"about_ca_topic_score_gemma":0.008342746,"teacher_disagreement_score":0.019658279,"about_ca_system_score_codex":0.0006121076,"about_ca_system_score_gemma":0.0017141104,"threshold_uncertainty_score":0.06576347},"labels":[],"label_agreement":null},{"id":"W2809877424","doi":"10.13053/rcs-145-1-4","title":"Genex+, a Semantic-based Automatic Extractor of Examples Applied to Bilingual Terms","year":2017,"lang":"en","type":"article","venue":"Research in Computing Science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Extractor; Computer science; Natural language processing; Artificial intelligence; Information retrieval; Engineering; Process engineering","score_opus":0.1216505839072852,"score_gpt":0.45331967435493553,"score_spread":0.33166909044765036,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2809877424","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021123458,0.00091964926,0.86868656,0.00021838899,0.0001507093,0.000513538,0.012351258,0.08469515,0.011341315],"genre_scores_gemma":[0.059538744,0.0004260828,0.9186243,0.00007293645,0.00006649749,0.00035797502,0.012128719,0.0042697117,0.0045149666],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9990972,0.00020868058,0.00011686733,0.00028171545,0.00025337987,0.000042285392],"domain_scores_gemma":[0.9985306,0.0007058739,0.00013172095,0.00026750617,0.00030994645,0.000054237356],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009247252,0.0015131969,0.0010499101,0.0053969645,0.0007675443,0.0016330291,0.0010090105,0.0006976234,0.022391481],"category_scores_gemma":[0.0048589343,0.0007849879,0.0007758474,0.0026059984,0.00050248014,0.0026377146,0.0018530948,0.00071931223,0.010575397],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00072452344,0.00008976189,0.0032417083,0.0034844035,0.00017203583,0.00095582067,0.0014037006,0.0023940466,0.05580666,0.025813753,0.04578315,0.86013055],"study_design_scores_gemma":[0.00028706665,0.00042845885,0.011302995,0.0007368008,0.0002952313,0.005151389,0.0014801007,0.16715306,0.12883155,0.032832284,0.6512606,0.00024043466],"about_ca_topic_score_codex":0.0014051568,"about_ca_topic_score_gemma":0.0026730963,"teacher_disagreement_score":0.022391481,"about_ca_system_score_codex":0.0005310519,"about_ca_system_score_gemma":0.00088744005,"threshold_uncertainty_score":0.074906945},"labels":[],"label_agreement":null},{"id":"W2823189734","doi":"","title":"Indigenous language technologies in Canada: Assessment, challenges, and successes","year":2018,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":123,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":true,"ca_institutions":"Carleton University; University of Alberta; University of British Columbia; University of Ottawa","funders":"","keywords":"Computer science; Indigenous language; Indigenous; Transliteration; Natural language processing; Spell; Character (mathematics); Artificial intelligence; Machine translation; Language translation; Speech recognition; Linguistics; Sociology","score_opus":0.01281198790067665,"score_gpt":0.2685093541554958,"score_spread":0.25569736625481915,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2823189734","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.734835,0.034197684,0.010231636,0.034503944,0.00047730486,0.00066404813,0.0038375885,0.0013051067,0.17994758],"genre_scores_gemma":[0.9323528,0.025505163,0.013512286,0.0015783502,0.00006403408,0.00014400265,0.0019573444,0.00017757755,0.024708513],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9888114,0.0008846718,0.00037076266,0.0006562364,0.0071839234,0.0020930427],"domain_scores_gemma":[0.976442,0.0020578061,0.0008117375,0.00038475246,0.017942198,0.0023615498],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007010282,0.0006447558,0.00045662787,0.004170801,0.00799557,0.0074126106,0.0021103565,0.00093011535,0.0028629764],"category_scores_gemma":[0.016221998,0.00027289306,0.00038412286,0.00806174,0.0036617666,0.0030623816,0.0031899621,0.0015805681,0.0006311415],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":true,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00041493296,0.00032481298,0.1810807,0.0021711355,0.00011540252,0.0012475145,0.03340927,0.0031509516,0.0053449012,0.037249994,0.022930235,0.7125601],"study_design_scores_gemma":[0.00005314712,0.00069549837,0.40669423,0.002234574,0.00034992103,0.0014213681,0.15668039,0.00854468,0.015492778,0.005835776,0.4015778,0.00041985637],"about_ca_topic_score_codex":0.9890674,"about_ca_topic_score_gemma":0.9920141,"teacher_disagreement_score":0.92621464,"about_ca_system_score_codex":0.07378538,"about_ca_system_score_gemma":0.17857307,"threshold_uncertainty_score":0.53535295},"labels":[],"label_agreement":null},{"id":"W28660552","doi":"10.1007/s11886-017-0876-4","title":"A community reference grammar of Labrador Inuttitut","year":2009,"lang":"en","type":"article","venue":"Current Cardiology Reports","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Grammar; Computer science; Linguistics; Rule-based machine translation; Speech community; Artificial intelligence; Natural language processing","score_opus":0.03529709228215173,"score_gpt":0.31759671673345014,"score_spread":0.2822996244512984,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W28660552","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0019223312,0.005440918,0.013437562,0.017160095,0.015146293,0.00021261563,0.022106094,0.0028803828,0.9216937],"genre_scores_gemma":[0.044614617,0.0075327996,0.015464727,0.0068520526,0.006188956,0.0004138962,0.03932586,0.0064088353,0.8731983],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99809223,0.00061488117,0.00026287546,0.00034214984,0.0005210579,0.000166914],"domain_scores_gemma":[0.99404246,0.0018368689,0.00043323965,0.0009315535,0.0024516883,0.00030410237],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0015596271,0.0011253485,0.001115426,0.0043747034,0.0025420515,0.005883009,0.0016567279,0.0019785508,0.3183237],"category_scores_gemma":[0.010800394,0.00040012863,0.00036995692,0.007839275,0.0011999181,0.0043644807,0.002291676,0.0023526037,0.20955849],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000061649276,0.000022604985,0.00026815233,0.00019243688,0.000003489896,0.00013641914,0.00056360976,0.00012558702,0.00028611853,0.07228967,0.87110656,0.054943714],"study_design_scores_gemma":[0.0000021889346,0.0000023728844,0.00010217521,0.000062454405,7.899645e-7,0.000045271834,0.000069332295,0.00007576076,0.00005467798,0.0013636828,0.99821764,0.000003536872],"about_ca_topic_score_codex":0.015914738,"about_ca_topic_score_gemma":0.01487504,"teacher_disagreement_score":0.98408526,"about_ca_system_score_codex":0.0025236378,"about_ca_system_score_gemma":0.0027786316,"threshold_uncertainty_score":0.9723285},"labels":[],"label_agreement":null},{"id":"W2883173034","doi":"","title":"Comment faire pour que l'opinion forgée à la sortie des urnes soit la bonne? Application au défi DEFT'07","year":2007,"lang":"fr","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Physics","score_opus":0.020600547641031898,"score_gpt":0.2709579569487284,"score_spread":0.25035740930769645,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2883173034","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03597768,0.0022012654,0.09146534,0.42960235,0.01490965,0.00014871839,0.00082464283,0.0010808794,0.42378944],"genre_scores_gemma":[0.6889024,0.0014867671,0.037505638,0.041923657,0.0057702973,0.00022480245,0.00048402124,0.001114909,0.22258759],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9887659,0.0046323407,0.00054342096,0.0016012827,0.0035223602,0.0009347159],"domain_scores_gemma":[0.98169625,0.009368455,0.0008084238,0.0010788224,0.006381606,0.0006664795],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010985208,0.000549098,0.00097372325,0.000907736,0.0057827574,0.010498618,0.0013219852,0.007107949,0.022347199],"category_scores_gemma":[0.050301455,0.00035413657,0.00070873566,0.00069218833,0.004864358,0.009597429,0.0026656217,0.0075834473,0.0054747732],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002700456,0.00002334616,0.0018948879,0.00015487579,0.00002978136,0.00032540865,0.0072551942,0.00047999766,0.002494168,0.8153665,0.12873378,0.042972036],"study_design_scores_gemma":[0.00009648915,0.000093717725,0.0033652578,0.00039129925,0.00005943736,0.00059616385,0.013385277,0.0066131325,0.004305485,0.3138026,0.6571624,0.00012874346],"about_ca_topic_score_codex":0.027932057,"about_ca_topic_score_gemma":0.020049185,"teacher_disagreement_score":0.027932057,"about_ca_system_score_codex":0.007070471,"about_ca_system_score_gemma":0.0029711328,"threshold_uncertainty_score":0.07475889},"labels":[],"label_agreement":null},{"id":"W2883231793","doi":"10.3138/cmlr.4054","title":"Intelligent Computer Assisted Language Learning (ICALL) for <i>nêhiyawêwin</i>: An In-Depth User-Experience Evaluation","year":2018,"lang":"en","type":"article","venue":"Canadian Modern Language Review/ La Revue canadienne des langues vivantes","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Mainstream; Indigenous; Think aloud protocol; Computer science; Class (philosophy); Language acquisition; Indigenous language; Human–computer interaction; Multimedia; Mathematics education; Psychology; Artificial intelligence; Usability","score_opus":0.03128640802397649,"score_gpt":0.30707221005681135,"score_spread":0.27578580203283487,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2883231793","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9656731,0.0019716586,0.008400247,0.00054440997,0.000059753915,0.0024191893,0.00034689394,0.00048327594,0.020101445],"genre_scores_gemma":[0.8883813,0.003993581,0.076014414,0.00069403526,0.000049662864,0.0044825305,0.0013145666,0.00017338243,0.024896473],"study_design_codex":"design_other","study_design_gemma":"qualitative","domain_scores_codex":[0.99796116,0.0011016737,0.0001339736,0.00014551742,0.00048063294,0.00017702149],"domain_scores_gemma":[0.99563944,0.0018670846,0.00017937185,0.00023012789,0.0016443699,0.00043960844],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0058183298,0.00042077352,0.00039437047,0.00063819037,0.00078947435,0.0010072513,0.0007791422,0.00050804025,0.0040755514],"category_scores_gemma":[0.0064102565,0.00015407868,0.00031970156,0.0005119601,0.0004682136,0.0009681706,0.0012539648,0.0004287349,0.00083614164],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002256321,0.0056181895,0.02689955,0.0042529926,0.00007354317,0.00085508975,0.046675563,0.0008878422,0.03927479,0.0019484296,0.016284907,0.85497284],"study_design_scores_gemma":[0.0013547345,0.05588217,0.32281253,0.0027032322,0.0007092181,0.003990627,0.057509363,0.011092628,0.088766776,0.0011951763,0.45357144,0.0004121479],"about_ca_topic_score_codex":0.006265237,"about_ca_topic_score_gemma":0.015027856,"teacher_disagreement_score":0.006265237,"about_ca_system_score_codex":0.0009701808,"about_ca_system_score_gemma":0.0018024774,"threshold_uncertainty_score":0.0307706},"labels":[],"label_agreement":null},{"id":"W2884082915","doi":"10.1162/coli_a_00327","title":"Anaphora With Non-nominal Antecedents in Computational Linguistics: a Survey","year":2018,"lang":"en","type":"article","venue":"Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":28,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Ruhr-Universität Bochum; Universität Hamburg","keywords":"Anaphora (linguistics); Antecedent (behavioral psychology); Computer science; Automatic summarization; Natural language processing; Linguistics; Sentence; Artificial intelligence; Machine translation; Computational linguistics; Field (mathematics); Resolution (logic); Psychology; Philosophy","score_opus":0.018271616867919945,"score_gpt":0.30533072661077887,"score_spread":0.2870591097428589,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2884082915","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0039673117,0.9221705,0.042713072,0.004644895,0.0005447061,0.00008641436,0.00017908939,0.00027205973,0.025421917],"genre_scores_gemma":[0.033568718,0.92874575,0.029298946,0.0018842806,0.0022911332,0.00016386854,0.0005530498,0.00018700442,0.0033072394],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99558914,0.001525695,0.00065061555,0.0008553069,0.0011826053,0.00019663242],"domain_scores_gemma":[0.9741697,0.022571364,0.00068555016,0.0010182614,0.0012858713,0.00026931512],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0056603244,0.0010014419,0.002127039,0.017628783,0.0023288147,0.0066418084,0.0023599032,0.0033219496,0.0055925283],"category_scores_gemma":[0.016355056,0.0014677192,0.001615489,0.025834356,0.003978933,0.018890187,0.0044171517,0.003491941,0.0023610024],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000059899266,0.00016360448,0.0034173022,0.011712641,0.00017794072,0.00050772115,0.005240136,0.0012060612,0.0006736683,0.16376095,0.02174421,0.7913359],"study_design_scores_gemma":[0.0000162344,0.00005213575,0.0034612517,0.0068646716,0.000110257664,0.0017983335,0.003143752,0.0034333663,0.0006677992,0.14865437,0.831718,0.00007979961],"about_ca_topic_score_codex":0.0028147749,"about_ca_topic_score_gemma":0.0025655585,"teacher_disagreement_score":0.017628783,"about_ca_system_score_codex":0.0021487796,"about_ca_system_score_gemma":0.003422385,"threshold_uncertainty_score":0.029935002},"labels":[],"label_agreement":null},{"id":"W2885158679","doi":"10.18653/v1/w18-2412","title":"Comparison of Assorted Models for Transliteration","year":2018,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Innovates","keywords":"Transliteration; Computer science; Discriminative model; Generative grammar; Artificial intelligence; Focus (optics); Context (archaeology); Task (project management); Natural language processing; Machine learning; Engineering","score_opus":0.05180045916560814,"score_gpt":0.3680837688659927,"score_spread":0.31628330970038454,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2885158679","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5720326,0.024708128,0.35187867,0.004855897,0.0016190177,0.00067129097,0.00488722,0.019601716,0.019745588],"genre_scores_gemma":[0.9097696,0.0032454487,0.069940455,0.0007167146,0.00027248752,0.00041530665,0.006315181,0.0015054413,0.007819335],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9942346,0.0028394922,0.0006177155,0.0012001343,0.00072143576,0.0003866225],"domain_scores_gemma":[0.9739128,0.019731458,0.00062090333,0.0029515699,0.001956445,0.0008269092],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013377366,0.0031229835,0.0023373286,0.003719448,0.0013165685,0.0025546532,0.00319565,0.0046416256,0.00500349],"category_scores_gemma":[0.028252702,0.0011945906,0.0021070743,0.0024407022,0.0009898791,0.00626551,0.0023614855,0.0031732873,0.00313351],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0053020935,0.0006907951,0.008709874,0.0013915525,0.0021326721,0.00044007468,0.0006272239,0.72106934,0.0058570825,0.008548035,0.014261642,0.23096958],"study_design_scores_gemma":[0.00018559789,0.0005471578,0.001663247,0.00008658086,0.00035703287,0.00018064924,0.0001594788,0.98154664,0.0051662293,0.008034673,0.001976505,0.0000961402],"about_ca_topic_score_codex":0.011403617,"about_ca_topic_score_gemma":0.017801546,"teacher_disagreement_score":0.013377366,"about_ca_system_score_codex":0.0028569715,"about_ca_system_score_gemma":0.0025536534,"threshold_uncertainty_score":0.07074714},"labels":[],"label_agreement":null},{"id":"W2885556514","doi":"","title":"UWaterlooMDS at the TREC 2017 Common Core Track.","year":2017,"lang":"en","type":"article","venue":"Text REtrieval Conference","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Track (disk drive); Computer science; Core (optical fiber); Telecommunications; Operating system","score_opus":0.06543898827919384,"score_gpt":0.3335348181136187,"score_spread":0.2680958298344248,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2885556514","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.003226726,0.006974822,0.012671283,0.016958805,0.014799087,0.0010111484,0.34813547,0.025094725,0.5711279],"genre_scores_gemma":[0.005177018,0.0025692913,0.008296989,0.0018929766,0.0015685314,0.00027174337,0.17785403,0.004037347,0.79833204],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99826103,0.00024714498,0.00008696298,0.00032591453,0.00085013115,0.00022877651],"domain_scores_gemma":[0.9948584,0.0004360389,0.00012546932,0.0005615867,0.0024787583,0.0015397941],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.002813818,0.0016668878,0.0027072951,0.0050477246,0.0026080355,0.006534286,0.0024186037,0.0017054431,0.636452],"category_scores_gemma":[0.0058711413,0.0008303203,0.00072121905,0.007028039,0.00068377634,0.0063693896,0.0034119033,0.0020137986,0.45084074],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000025665613,0.000028292874,0.000043327433,0.00004876966,0.000003015682,0.000009653846,0.0000101977685,0.000031853244,0.00026661594,0.00042439453,0.9867306,0.012377694],"study_design_scores_gemma":[0.000043703483,0.000026089223,0.0008075878,0.00007192578,0.000007761277,0.000024632936,0.00008208418,0.0005169187,0.0006108429,0.0024516813,0.9953354,0.00002139208],"about_ca_topic_score_codex":0.07814548,"about_ca_topic_score_gemma":0.21512763,"teacher_disagreement_score":0.636452,"about_ca_system_score_codex":0.00351377,"about_ca_system_score_gemma":0.005592791,"threshold_uncertainty_score":0.5185571},"labels":[],"label_agreement":null},{"id":"W2885711770","doi":"","title":"Comparison of Two Interactive Search Refinement Techniques","year":2004,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Information retrieval; Set (abstract data type); Query expansion; Document retrieval; Web search query; Noun phrase; Natural language processing; Search engine; Artificial intelligence; Noun; Programming language","score_opus":0.0320002516061259,"score_gpt":0.40214932323633507,"score_spread":0.37014907163020916,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2885711770","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.058894984,0.0020690914,0.9172341,0.00034419456,0.00007429269,0.0006358977,0.00027568592,0.009244183,0.011227474],"genre_scores_gemma":[0.31954423,0.0012688279,0.66601205,0.00033826716,0.000092504524,0.0006105439,0.0013093858,0.0015795426,0.00924465],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9946616,0.002063318,0.00039185447,0.00040616916,0.0021263005,0.00035082622],"domain_scores_gemma":[0.98500174,0.0103849,0.00047670255,0.002244155,0.0016736174,0.00021878174],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034006485,0.0014247334,0.001312232,0.0018916775,0.00068515074,0.0013981675,0.0038397266,0.0016688299,0.009175998],"category_scores_gemma":[0.01954108,0.00060261483,0.0013661699,0.0020109909,0.00088962936,0.003608762,0.0025189177,0.0018805204,0.00248659],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0039301463,0.001323125,0.0025942442,0.0011132807,0.0003245065,0.00024493912,0.0022452446,0.052303355,0.030612571,0.022863256,0.0070297783,0.8754156],"study_design_scores_gemma":[0.0014713896,0.0028370703,0.0045819464,0.00023264672,0.00059281365,0.0011823099,0.0013716023,0.861446,0.06514808,0.017203094,0.04360303,0.00033000397],"about_ca_topic_score_codex":0.006214433,"about_ca_topic_score_gemma":0.0069798795,"teacher_disagreement_score":0.009175998,"about_ca_system_score_codex":0.0007659131,"about_ca_system_score_gemma":0.0010793493,"threshold_uncertainty_score":0.03069675},"labels":[],"label_agreement":null},{"id":"W2886751231","doi":"","title":"Simultaneous Translation using Optimized Segmentation","year":2018,"lang":"en","type":"article","venue":"Conference of the Association for Machine Translation in the Americas","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Translation (biology); Computer science; Segmentation; Artificial intelligence; Image segmentation; Computer vision; Natural language processing","score_opus":0.04451618106901561,"score_gpt":0.33230382086866517,"score_spread":0.28778763979964955,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2886751231","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020116711,0.0013140505,0.9463629,0.00044798374,0.00096743397,0.00013137002,0.0009390943,0.016584707,0.013135782],"genre_scores_gemma":[0.21642071,0.00073539966,0.7499833,0.00035891228,0.00039721146,0.00020838212,0.0051494394,0.0050345003,0.021712119],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99814415,0.0003455135,0.00012947727,0.0007066155,0.00043607655,0.000238139],"domain_scores_gemma":[0.9984107,0.00040819886,0.00009331333,0.00060269824,0.00041885112,0.00006619109],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00106074,0.0027627426,0.0023076192,0.0027791958,0.0016996824,0.0043438813,0.0015999128,0.002487424,0.02784148],"category_scores_gemma":[0.0028932379,0.0019018151,0.0026712054,0.004052777,0.0010316132,0.0031137115,0.0032566926,0.0022218884,0.016903538],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010709291,0.00015713551,0.00077271153,0.00045519165,0.00031745966,0.000688775,0.0002959346,0.036805447,0.104637384,0.0160229,0.033489935,0.80528617],"study_design_scores_gemma":[0.00015353937,0.00029516005,0.001650485,0.00009078749,0.00034605747,0.0010185395,0.00025847025,0.7290802,0.15911606,0.041601617,0.06621517,0.0001738396],"about_ca_topic_score_codex":0.004488672,"about_ca_topic_score_gemma":0.008248411,"teacher_disagreement_score":0.02784148,"about_ca_system_score_codex":0.00089974084,"about_ca_system_score_gemma":0.002122846,"threshold_uncertainty_score":0.09313905},"labels":[],"label_agreement":null},{"id":"W2886776997","doi":"","title":"Leveraging Data Resources for Cross-Linguistic Information Retrieval Using Statistical Machine Translation","year":2018,"lang":"en","type":"article","venue":"Conference of the Association for Machine Translation in the Americas","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Machine translation; Computer science; Natural language processing; Rule-based machine translation; Artificial intelligence; Cross-language information retrieval; Translation (biology); Information retrieval","score_opus":0.08431429977728597,"score_gpt":0.37126298598252266,"score_spread":0.2869486862052367,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2886776997","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.061068058,0.0084285075,0.8855183,0.004600909,0.0012245261,0.0010082859,0.007484682,0.017290985,0.013375696],"genre_scores_gemma":[0.279588,0.004222525,0.67411864,0.0012422312,0.0005765505,0.0008273194,0.034279037,0.0016798639,0.0034658646],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99011886,0.0048496416,0.0013161664,0.0012380338,0.0020311314,0.00044618166],"domain_scores_gemma":[0.9760968,0.011477759,0.00080751994,0.007006294,0.0042720125,0.00033959234],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007437597,0.001954355,0.0026826449,0.017603086,0.0022139647,0.0055407896,0.002598087,0.0021855845,0.0064566736],"category_scores_gemma":[0.03304022,0.0011059653,0.0023405398,0.015689857,0.0013378252,0.011611374,0.008392972,0.0028219405,0.008817612],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007594438,0.0013799069,0.0046870103,0.0025715784,0.00074009487,0.002103129,0.001441989,0.015333965,0.056963515,0.020290622,0.047068402,0.8466603],"study_design_scores_gemma":[0.00033814122,0.00085010316,0.0058928323,0.0010792611,0.0012858596,0.0021550532,0.0038421897,0.5500341,0.12152977,0.13428774,0.17820229,0.00050262],"about_ca_topic_score_codex":0.0033189591,"about_ca_topic_score_gemma":0.0053224466,"teacher_disagreement_score":0.017603086,"about_ca_system_score_codex":0.0010506015,"about_ca_system_score_gemma":0.0038182593,"threshold_uncertainty_score":0.039334297},"labels":[],"label_agreement":null},{"id":"W2886787252","doi":"10.1007/978-3-319-98678-4_42","title":"An Integrated AMIS Prototype for Automated Summarization and Translation of Newscasts and Reports","year":2018,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Agence Nationale de la Recherche","keywords":"Automatic summarization; Computer science; Translation (biology); Software; Machine translation; Software engineering; Programming language; Information retrieval; Artificial intelligence","score_opus":0.014538732380786934,"score_gpt":0.28103010189461447,"score_spread":0.2664913695138275,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2886787252","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02160336,0.00075847836,0.56389934,0.0004573296,0.00087147806,0.00086327654,0.016407525,0.38194644,0.013192706],"genre_scores_gemma":[0.07788687,0.0004187311,0.8285044,0.00035662166,0.00042606582,0.0010419026,0.057480123,0.013491943,0.020393373],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99902105,0.00021572738,0.000121187986,0.00031092012,0.0002745394,0.000056606885],"domain_scores_gemma":[0.99785125,0.00084804517,0.00013957228,0.00039704988,0.00063221855,0.00013194968],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014226672,0.0015378015,0.0010345806,0.0022412145,0.00077859074,0.002375861,0.001849466,0.0010806951,0.03206567],"category_scores_gemma":[0.0048405034,0.00080160337,0.0009913618,0.001675809,0.0003567643,0.0024973287,0.0017715439,0.0012183952,0.02183151],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016796937,0.0004036304,0.0012579171,0.0018704825,0.00041389815,0.0008119025,0.001785604,0.003379187,0.12457851,0.0041381367,0.19177496,0.6679061],"study_design_scores_gemma":[0.0010728327,0.0014520488,0.0065549854,0.0003582175,0.00074629637,0.001746453,0.0020919116,0.19558173,0.2308501,0.012521297,0.546678,0.0003461159],"about_ca_topic_score_codex":0.0019171385,"about_ca_topic_score_gemma":0.0027158363,"teacher_disagreement_score":0.03206567,"about_ca_system_score_codex":0.00038840587,"about_ca_system_score_gemma":0.0008040438,"threshold_uncertainty_score":0.1072703},"labels":[],"label_agreement":null},{"id":"W2887191217","doi":"10.18653/v1/w18-2414","title":"Low-Resource Machine Transliteration Using Recurrent Neural Networks of Asian Languages","year":2018,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Transliteration; Computer science; Grapheme; Natural language processing; Artificial intelligence; Machine translation; Pronunciation; Task (project management); Sequence (biology); Vietnamese; Speech recognition; Resource (disambiguation); Artificial neural network; Linguistics","score_opus":0.011383521399224949,"score_gpt":0.2861042481436922,"score_spread":0.27472072674446724,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2887191217","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.37639162,0.0008908441,0.61004937,0.00062879093,0.00020766445,0.00008602434,0.000638759,0.005344708,0.0057622767],"genre_scores_gemma":[0.8856045,0.00025061282,0.107663475,0.00012791119,0.000049428472,0.000086771186,0.0015268016,0.00028687285,0.004403624],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99961555,0.00013535892,0.00002570581,0.00012852561,0.00004292143,0.000051848183],"domain_scores_gemma":[0.9990608,0.00047188887,0.000101127785,0.00013896004,0.00018695764,0.00004020265],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006175987,0.0007537966,0.00052847865,0.00027172914,0.000319378,0.0005381859,0.00067842775,0.0003683915,0.0019906245],"category_scores_gemma":[0.0028073243,0.00028332716,0.0005122298,0.00040966956,0.00028603914,0.0016295741,0.0007275271,0.0009725264,0.000842546],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00078160234,0.0003287731,0.0032038388,0.0005149592,0.000210912,0.0012301889,0.0008147633,0.3736697,0.08474566,0.014884394,0.011470864,0.5081444],"study_design_scores_gemma":[0.000019837897,0.00008637158,0.00043541295,0.000011051071,0.000029203904,0.0000653168,0.00005856887,0.97820055,0.013963038,0.0057596103,0.0013573782,0.0000135762975],"about_ca_topic_score_codex":0.004287685,"about_ca_topic_score_gemma":0.007361326,"teacher_disagreement_score":0.004287685,"about_ca_system_score_codex":0.00041189793,"about_ca_system_score_gemma":0.00050906383,"threshold_uncertainty_score":0.008525491},"labels":[],"label_agreement":null},{"id":"W2887836700","doi":"10.7202/1048829ar","title":"Hermeneutica, une expérience numérique de l’interprétation : Hermeneutica. Computer-assisted interpretation in the humanities, de Geoffrey Rockwell et Stéfan Sinclair","year":2017,"lang":"fr","type":"article","venue":"Sens public","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Philosophy; Physics","score_opus":0.03690512863309288,"score_gpt":0.3170315240495979,"score_spread":0.280126395416505,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2887836700","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015985437,0.21087235,0.39020228,0.1155339,0.0069283526,0.0001996957,0.00042660683,0.00093006453,0.2589213],"genre_scores_gemma":[0.5931466,0.07831747,0.16804542,0.016085675,0.0048168674,0.0009678549,0.0004496691,0.0021483041,0.13602208],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.98968387,0.0072856313,0.00028911087,0.0009218011,0.0015006666,0.0003188708],"domain_scores_gemma":[0.98968494,0.007904638,0.00044612054,0.00088740455,0.0007420477,0.0003348231],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009337739,0.0012562955,0.0009806447,0.0036584348,0.005344348,0.015512456,0.0016552772,0.004645154,0.008637944],"category_scores_gemma":[0.017892994,0.0010101141,0.0010098544,0.003923608,0.039394267,0.024878064,0.0057554445,0.0073283683,0.0017678645],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00004825777,0.000017759121,0.00030468247,0.0002529607,0.000014024083,0.00015342855,0.037891038,0.0003064037,0.00060652377,0.9039226,0.025078315,0.031403974],"study_design_scores_gemma":[0.000020582158,0.000024309458,0.0005582073,0.0005421857,0.000012959012,0.00041445647,0.013594599,0.0011125924,0.0009823198,0.51806355,0.4646155,0.000058837675],"about_ca_topic_score_codex":0.008954764,"about_ca_topic_score_gemma":0.010580482,"teacher_disagreement_score":0.015512456,"about_ca_system_score_codex":0.005832795,"about_ca_system_score_gemma":0.0044526835,"threshold_uncertainty_score":0.049383283},"labels":[],"label_agreement":null},{"id":"W2888353881","doi":"10.1109/icosc.2019.8665531","title":"You Shall Know the Most Frequent Sense by the Company it Keeps","year":2019,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Innovates","keywords":"Identification (biology); Task (project management); Computer science; Sense (electronics); Word (group theory); Translation (biology); Natural language processing; Word-sense disambiguation; Artificial intelligence; Linguistics; Philosophy; Engineering","score_opus":0.017443850595418038,"score_gpt":0.27323721407322393,"score_spread":0.2557933634778059,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2888353881","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.63216996,0.0048992177,0.24375315,0.011397195,0.0030858812,0.00022492599,0.03971751,0.0056525446,0.059099626],"genre_scores_gemma":[0.8498016,0.001979028,0.11272173,0.0007513334,0.0008786214,0.00010491755,0.012558742,0.0007001441,0.02050397],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995036,0.00006280997,0.000059005073,0.00014955607,0.00018157344,0.00004344944],"domain_scores_gemma":[0.99861395,0.0005447197,0.00021531644,0.00016853995,0.0003581322,0.00009927815],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00047434238,0.00048524947,0.00035964267,0.0023315628,0.00060546515,0.0009416644,0.00026127623,0.00080679636,0.010119706],"category_scores_gemma":[0.0035731343,0.00023290544,0.00026196195,0.0023923828,0.00048747964,0.002320973,0.0005165836,0.0005828717,0.0062147947],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012069745,0.00021859455,0.0844019,0.0011914131,0.00020496338,0.003468667,0.0036517633,0.0010530396,0.16088998,0.03647878,0.11832752,0.58890635],"study_design_scores_gemma":[0.00009297031,0.0006431522,0.28963754,0.00054260483,0.00031376255,0.022939034,0.0057421685,0.03362929,0.07327493,0.10876605,0.46391964,0.00049879286],"about_ca_topic_score_codex":0.0013640047,"about_ca_topic_score_gemma":0.0023559458,"teacher_disagreement_score":0.010119706,"about_ca_system_score_codex":0.00018302213,"about_ca_system_score_gemma":0.00018315444,"threshold_uncertainty_score":0.03385377},"labels":[],"label_agreement":null},{"id":"W2888541716","doi":"10.18653/v1/d18-1398","title":"Meta-Learning for Low-Resource Neural Machine Translation","year":2018,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":319,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Tencent; Samsung Advanced Institute of Technology; Canadian Institute for Advanced Research; Samsung; Nvidia","keywords":"Computer science; Machine translation; Artificial intelligence; Romanian; Natural language processing; Transfer of learning; Representation (politics); BLEU; Translation (biology); Resource (disambiguation); Machine learning; Frame (networking); Linguistics","score_opus":0.03958221554592636,"score_gpt":0.29513782613145806,"score_spread":0.2555556105855317,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2888541716","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017478952,0.0012550165,0.97593176,0.0003304542,0.000116521005,0.000053530825,0.00012603046,0.0029348575,0.001772853],"genre_scores_gemma":[0.5940461,0.00062662945,0.39874968,0.000599175,0.0002353413,0.00037597405,0.0010278148,0.00043997576,0.0038993931],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99890983,0.0004950097,0.00008115457,0.00028197299,0.00014968748,0.00008233185],"domain_scores_gemma":[0.99811065,0.0010338692,0.00013447962,0.0004620809,0.00020363103,0.000055357425],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020902338,0.001293904,0.001368596,0.0009583265,0.00057647435,0.0013094777,0.002406485,0.0017891907,0.0034819827],"category_scores_gemma":[0.0051221494,0.000620242,0.0012159507,0.0014362029,0.0006719816,0.0028349387,0.0016022597,0.0021782303,0.0017766427],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003584467,0.00029573153,0.0012843258,0.00039465932,0.00040160533,0.00023969586,0.0001315654,0.4990323,0.011647556,0.017411225,0.00560353,0.4631994],"study_design_scores_gemma":[0.000021322221,0.000060222297,0.00010804973,0.0000147818,0.00002696073,0.00004356051,0.000011342477,0.98249346,0.0026367174,0.013500203,0.0010734798,0.000009827055],"about_ca_topic_score_codex":0.0011238572,"about_ca_topic_score_gemma":0.0020290627,"teacher_disagreement_score":0.0034819827,"about_ca_system_score_codex":0.00086525147,"about_ca_system_score_gemma":0.0009060726,"threshold_uncertainty_score":0.0116484165},"labels":[],"label_agreement":null},{"id":"W2888817241","doi":"","title":"Interactive Search Refinement Techniques for HARD Tasks.","year":2003,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Information retrieval; Query expansion; Set (abstract data type); Web search query; Search engine; Natural language processing; Programming language","score_opus":0.02458523497708714,"score_gpt":0.3286205812999517,"score_spread":0.30403534632286455,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2888817241","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008870636,0.0013702216,0.977448,0.00034439383,0.00004569599,0.00040325898,0.00022601413,0.0062792846,0.0050124065],"genre_scores_gemma":[0.12335085,0.0009229784,0.86549383,0.0003518452,0.00010593758,0.0006297339,0.0009910446,0.0012300751,0.006923656],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9939622,0.0032192855,0.0002712718,0.00050128077,0.001738367,0.0003076581],"domain_scores_gemma":[0.9849392,0.0113502955,0.00035500998,0.0024906578,0.0006423761,0.00022236028],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0050161756,0.0018498517,0.0013494098,0.0014771938,0.0008233257,0.001503963,0.0035890397,0.001363993,0.017947862],"category_scores_gemma":[0.022528462,0.00081164466,0.0013290574,0.0017228927,0.0016146224,0.005265878,0.0038204475,0.0033503408,0.0053656725],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019383839,0.00062933384,0.0014109624,0.002134152,0.00022424763,0.00031459864,0.0021471768,0.036362737,0.038826797,0.07820759,0.03811671,0.79968727],"study_design_scores_gemma":[0.00089301984,0.0010621767,0.0014158939,0.000352543,0.0002309728,0.0012044483,0.001004699,0.6959531,0.0412926,0.14916162,0.10724532,0.0001835716],"about_ca_topic_score_codex":0.0035065897,"about_ca_topic_score_gemma":0.007627704,"teacher_disagreement_score":0.017947862,"about_ca_system_score_codex":0.00074486685,"about_ca_system_score_gemma":0.0009357839,"threshold_uncertainty_score":0.060041547},"labels":[],"label_agreement":null},{"id":"W2888880755","doi":"","title":"Do Character-Level Neural Network Language Models Capture Knowledge of Multiword Expression Compositionality?","year":2018,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of New Brunswick","funders":"","keywords":"Principle of compositionality; Computer science; Artificial intelligence; Natural language processing; Character (mathematics); Treebank; Expression (computer science); Artificial neural network; Programming language; Annotation","score_opus":0.06360377069358782,"score_gpt":0.3540654245744966,"score_spread":0.2904616538809088,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2888880755","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1854207,0.00089776336,0.80238473,0.001625228,0.00018013513,0.00009253507,0.0007118744,0.0011807518,0.0075062644],"genre_scores_gemma":[0.9132646,0.00069783186,0.08027994,0.00037930222,0.00008473181,0.00016711222,0.0009781008,0.00015353429,0.0039949366],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99956506,0.00014967842,0.000022166325,0.00015975896,0.00005730129,0.00004601953],"domain_scores_gemma":[0.99767035,0.0013340241,0.00028954566,0.0002489868,0.0003803847,0.00007668759],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010711154,0.0007585135,0.0005765493,0.0005699777,0.0002523402,0.0015210385,0.0010954484,0.0008327164,0.0023233604],"category_scores_gemma":[0.007353724,0.00045337435,0.0005262528,0.00052459486,0.0005139881,0.0056121275,0.0006485967,0.0017654097,0.0012580599],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007655436,0.00041580663,0.028925953,0.0007555518,0.00069042674,0.00044156812,0.0010206278,0.34045026,0.06436678,0.0629559,0.004677464,0.4945341],"study_design_scores_gemma":[0.000009184239,0.00004854711,0.002398079,0.000029542914,0.00004591726,0.0000543551,0.00007391087,0.96428317,0.0034299449,0.028309215,0.001295674,0.0000223931],"about_ca_topic_score_codex":0.0028107543,"about_ca_topic_score_gemma":0.0039940323,"teacher_disagreement_score":0.0028107543,"about_ca_system_score_codex":0.0005696893,"about_ca_system_score_gemma":0.00058619154,"threshold_uncertainty_score":0.007772386},"labels":[],"label_agreement":null},{"id":"W2889109524","doi":"","title":"Knowledge of Language and Knowledge Science.","year":2018,"lang":"en","type":"article","venue":"New Trends in Software Methodologies, Tools and Techniques","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec","funders":"","keywords":"Computer science; Knowledge management; Natural language processing","score_opus":0.10654051370661888,"score_gpt":0.41372093511566926,"score_spread":0.3071804214090504,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2889109524","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010628825,0.0839489,0.57039076,0.042797696,0.002490788,0.00013056496,0.001009837,0.001108462,0.28749412],"genre_scores_gemma":[0.5270367,0.097024746,0.27696484,0.009861461,0.006145606,0.00047676064,0.002910974,0.00038194231,0.079196915],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9965249,0.0012568495,0.00032870815,0.0006902498,0.0010640997,0.00013527283],"domain_scores_gemma":[0.98475397,0.010253354,0.0006493446,0.0025388103,0.0012842698,0.00052034756],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036449356,0.0005179116,0.0009456898,0.0037454083,0.0012518633,0.0088200085,0.0018631608,0.0019046231,0.013572156],"category_scores_gemma":[0.016160544,0.0004081641,0.0007325962,0.00287525,0.009987732,0.018689519,0.0033598228,0.0036129376,0.002975259],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00002097205,0.000028698843,0.00037242792,0.000484879,0.000041140218,0.00013034843,0.001245058,0.00057696935,0.000411352,0.90100336,0.010710009,0.084974684],"study_design_scores_gemma":[0.000004448147,0.000007321714,0.00018958049,0.00021739218,0.0000103728025,0.00009994439,0.00039004852,0.00089157996,0.00026069582,0.9160462,0.08187214,0.0000103859475],"about_ca_topic_score_codex":0.0027630932,"about_ca_topic_score_gemma":0.0028174222,"teacher_disagreement_score":0.013572156,"about_ca_system_score_codex":0.0020930085,"about_ca_system_score_gemma":0.003374548,"threshold_uncertainty_score":0.04540336},"labels":[],"label_agreement":null},{"id":"W2890045624","doi":"10.18653/v1/w19-4002","title":"WiRe57 : A Fine-Grained Benchmark for Open Information Extraction","year":2019,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Granularity; Task (project management); Benchmark (surveying); Tuple; Inference; Information extraction; Annotation; Information retrieval; Data mining; Artificial intelligence; Programming language; Mathematics; Engineering","score_opus":0.020855440689569498,"score_gpt":0.3239822860998366,"score_spread":0.3031268454102671,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2890045624","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.23275651,0.010942599,0.3383929,0.004136631,0.0015431986,0.002161338,0.17024712,0.1224635,0.11735622],"genre_scores_gemma":[0.22629364,0.0023687237,0.31514373,0.0006516641,0.00024172539,0.0012049511,0.42159978,0.0126217175,0.019874034],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99098843,0.0022789533,0.0016951046,0.0013285211,0.0030681763,0.00064073474],"domain_scores_gemma":[0.9756631,0.010225533,0.0010374498,0.007564204,0.004931328,0.0005783432],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006962998,0.0017504126,0.001039682,0.0075641717,0.0022016284,0.0040323264,0.0031561344,0.00221413,0.011691604],"category_scores_gemma":[0.034404058,0.000784937,0.001196338,0.009032816,0.0011219939,0.0062481593,0.0037287609,0.0016175845,0.0070348145],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014445976,0.0013396955,0.013496896,0.0050850543,0.0004864424,0.00087454193,0.0016365346,0.0294151,0.021143813,0.03341568,0.40276983,0.4888918],"study_design_scores_gemma":[0.0007184108,0.0013819985,0.023153635,0.0011735588,0.00033523623,0.0020636404,0.0026110928,0.18437816,0.116958015,0.05192712,0.61492646,0.0003726487],"about_ca_topic_score_codex":0.008511967,"about_ca_topic_score_gemma":0.010293964,"teacher_disagreement_score":0.011691604,"about_ca_system_score_codex":0.0015949284,"about_ca_system_score_gemma":0.002347861,"threshold_uncertainty_score":0.03911227},"labels":[],"label_agreement":null},{"id":"W2890231653","doi":"10.4324/9781315561271-19","title":"The Root-Word Method for Building Proficient Second-Language Speakers of Polysynthetic Languages","year":2018,"lang":"en","type":"book-chapter","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Root (linguistics); Linguistics; Word (group theory); Psychology; Computer science; Philosophy","score_opus":0.01274332393520683,"score_gpt":0.31227650313511934,"score_spread":0.2995331791999125,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2890231653","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.056690693,0.01594569,0.17947988,0.00487055,0.0010825999,0.00050301076,0.00028356517,0.0030562207,0.73808783],"genre_scores_gemma":[0.12404899,0.011330517,0.23219381,0.0010185607,0.00015715152,0.0005262937,0.00053401885,0.0018264148,0.62836426],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9997497,0.00004772029,0.000010633811,0.000045914738,0.0001181334,0.000027903625],"domain_scores_gemma":[0.99975437,0.000094736184,0.000011049243,0.000029500019,0.000060261125,0.000050000555],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00053986796,0.00059314567,0.00019652338,0.000772196,0.00094196724,0.0013337274,0.0007154472,0.000578659,0.01834128],"category_scores_gemma":[0.0008004797,0.00025620373,0.00021900037,0.00036855118,0.001148865,0.0025880188,0.0018900504,0.0017795102,0.0061690854],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003991205,0.0002100905,0.0010773289,0.00057871913,0.000005515403,0.00039719685,0.022111965,0.00029368617,0.014655624,0.17013371,0.098890066,0.6916062],"study_design_scores_gemma":[0.000010966774,0.0001166652,0.0020328318,0.0002584951,0.000007675905,0.0018162936,0.003487109,0.0006719902,0.007708933,0.023952361,0.9599127,0.000023890512],"about_ca_topic_score_codex":0.0010679679,"about_ca_topic_score_gemma":0.0047435625,"teacher_disagreement_score":0.01834128,"about_ca_system_score_codex":0.0006945173,"about_ca_system_score_gemma":0.0015601069,"threshold_uncertainty_score":0.061357677},"labels":[],"label_agreement":null},{"id":"W2890328620","doi":"10.18653/v1/d18-1097","title":"Utilizing Character and Word Embeddings for Text Normalization with Sequence-to-Sequence Models","year":2018,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"York University; New York University Abu Dhabi","keywords":"Computer science; Natural language processing; Normalization (sociology); Artificial intelligence; Word (group theory); Artificial neural network; Language model; Character (mathematics); Recurrent neural network; Task (project management); Sequence (biology); Arabic; Linguistics","score_opus":0.043959955845430145,"score_gpt":0.3089185288247624,"score_spread":0.26495857297933223,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2890328620","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.033811715,0.00094433076,0.94137585,0.0007078433,0.0005468128,0.00023378598,0.0017405,0.017974647,0.002664437],"genre_scores_gemma":[0.34695405,0.0010547069,0.61518145,0.00073168386,0.00039089602,0.0006202111,0.0152652785,0.0017787473,0.01802303],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985291,0.00041543797,0.00012126874,0.00055464596,0.0002787002,0.00010091667],"domain_scores_gemma":[0.9973132,0.0009581562,0.0002999932,0.0006372685,0.0007077355,0.00008364319],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016387461,0.001734823,0.00081990357,0.0017170127,0.00054397236,0.0012162043,0.0016177432,0.0012745649,0.0033075083],"category_scores_gemma":[0.0077062873,0.0004138675,0.0010434103,0.0020718488,0.00072222215,0.0036883065,0.0013112376,0.0030575986,0.0050818427],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00038603842,0.00038477755,0.002408602,0.0003133038,0.000118596545,0.00030490474,0.00022658314,0.13781458,0.03126303,0.00894955,0.02802657,0.78980345],"study_design_scores_gemma":[0.00001936667,0.00010150923,0.0005376591,0.000023229544,0.00002768201,0.00013278979,0.000058885733,0.9529775,0.019701878,0.016499124,0.009891768,0.000028690347],"about_ca_topic_score_codex":0.0059392503,"about_ca_topic_score_gemma":0.010093793,"teacher_disagreement_score":0.0059392503,"about_ca_system_score_codex":0.0011857903,"about_ca_system_score_gemma":0.0014385429,"threshold_uncertainty_score":0.011809349},"labels":[],"label_agreement":null},{"id":"W2890460364","doi":"10.1023/a:1006500224529","title":"The Berkeley UNIX Consultant Project","year":2000,"lang":"en","type":"article","venue":"Artificial Intelligence Review","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"RTDS Technologies (Canada)","funders":"","keywords":"Computer science; Unix; Programming language; Utterance; Natural language; Component (thermodynamics); Artificial intelligence; Knowledge representation and reasoning; Natural language processing; Human–computer interaction","score_opus":0.049952052016624185,"score_gpt":0.3552141471328705,"score_spread":0.3052620951162463,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2890460364","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0019797718,0.03584377,0.00290174,0.069630556,0.020693075,0.00011479866,0.0052993367,0.0034154986,0.8601214],"genre_scores_gemma":[0.0042891456,0.007626456,0.0020745294,0.001942544,0.0011882769,0.00003862234,0.0020595577,0.00035020226,0.98043066],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9982805,0.00028486404,0.000062770574,0.00027791562,0.00080102665,0.00029293672],"domain_scores_gemma":[0.9952519,0.00056364364,0.00024371382,0.0006444346,0.0019354133,0.0013608661],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.002291495,0.00090467813,0.0007733882,0.003111538,0.0017753494,0.004036281,0.0014364218,0.002476882,0.36735696],"category_scores_gemma":[0.00524267,0.00044579638,0.00044646196,0.0044729426,0.00081039465,0.003492487,0.0023281001,0.0024306583,0.21194713],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000051106876,0.000017276896,0.000092709015,0.0000713815,0.0000020107893,0.00001705688,0.000014580758,0.000030571635,0.00008659572,0.008477912,0.9106826,0.08045621],"study_design_scores_gemma":[0.000016248447,0.000009970219,0.0003137015,0.000048837635,0.0000026220887,0.00002859229,0.000026539346,0.000041017738,0.00009940745,0.0011715556,0.99823856,0.0000029125135],"about_ca_topic_score_codex":0.010761659,"about_ca_topic_score_gemma":0.0126428595,"teacher_disagreement_score":0.36735696,"about_ca_system_score_codex":0.0026135212,"about_ca_system_score_gemma":0.006473449,"threshold_uncertainty_score":0.9023885},"labels":[],"label_agreement":null},{"id":"W2890620592","doi":"10.18653/v1/w18-5805","title":"String Transduction with Target Language Models and Insertion Handling","year":2018,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Advanced Research Projects Agency; Defense Advanced Research Projects Agency; University of Alberta","keywords":"Transduction (biophysics); Grapheme; Computer science; String (physics); Natural language processing; Sequence (biology); Word (group theory); Projection (relational algebra); Artificial intelligence; Syllabification; Inflection; Language model; Speech recognition; Linguistics; Algorithm; Biology; Mathematics; Syllable; Engineering","score_opus":0.017088273484584995,"score_gpt":0.2616792634933338,"score_spread":0.24459099000874882,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2890620592","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01436957,0.00013385311,0.96428454,0.00029258188,0.0001476017,0.00006353891,0.0007756567,0.01613206,0.0038005696],"genre_scores_gemma":[0.4702257,0.00046222948,0.49967542,0.0005806906,0.00028112275,0.00030197093,0.008084153,0.0068462966,0.013542388],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9978854,0.00062695396,0.0001822462,0.0006657391,0.0004942781,0.00014532331],"domain_scores_gemma":[0.9971973,0.0011114209,0.00014675206,0.0009875511,0.00048988493,0.00006704673],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017738189,0.001563583,0.0010295837,0.0011661574,0.00059960474,0.0026039393,0.0025127826,0.001832942,0.011491325],"category_scores_gemma":[0.006710083,0.0009912031,0.0016315449,0.0016765768,0.0013518217,0.006282211,0.0038394297,0.0039012583,0.012365188],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00072510296,0.00043932055,0.0027627442,0.0010089929,0.0002100857,0.0015335015,0.0012745436,0.15421854,0.091173664,0.15361302,0.026950365,0.56609017],"study_design_scores_gemma":[0.000027798853,0.0001251449,0.00035299125,0.000035719833,0.000051627587,0.0005464676,0.00014707836,0.7595673,0.05654448,0.16702838,0.015510879,0.000062120715],"about_ca_topic_score_codex":0.00074458856,"about_ca_topic_score_gemma":0.0010623063,"teacher_disagreement_score":0.011491325,"about_ca_system_score_codex":0.00053656194,"about_ca_system_score_gemma":0.00094496657,"threshold_uncertainty_score":0.038442314},"labels":[],"label_agreement":null},{"id":"W2890631628","doi":"10.1016/j.dib.2018.08.099","title":"MiBio: A dataset for OCR post-processing evaluation","year":2018,"lang":"en","type":"article","venue":"Data in Brief","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Computer science; Ground truth; Natural language processing; Preprocessor; Artificial intelligence; Benchmark (surveying); Sentence; Segmentation; Information retrieval; Optical character recognition; Word (group theory); Text segmentation; Linguistics; Image (mathematics); Cartography","score_opus":0.06855534351809031,"score_gpt":0.38595050366115724,"score_spread":0.3173951601430669,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2890631628","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023107827,0.0024823707,0.018486638,0.000569408,0.00064573233,0.0019946848,0.9030472,0.03174589,0.01792024],"genre_scores_gemma":[0.010458813,0.0002953499,0.022930296,0.0001705563,0.00009293534,0.0012787126,0.9592789,0.00095222803,0.004542176],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99375486,0.0008579418,0.00088382623,0.0011399708,0.0029877364,0.00037561537],"domain_scores_gemma":[0.990155,0.0018813881,0.000767187,0.002582924,0.0042140414,0.00039951692],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027166798,0.0042230207,0.0015941217,0.008879711,0.0016047645,0.0021580965,0.004030414,0.002968336,0.020838734],"category_scores_gemma":[0.010312528,0.00062026444,0.0017282121,0.0055274502,0.0007036009,0.0027269751,0.0027219597,0.0019037775,0.031317893],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005310346,0.00071403343,0.0034415491,0.0026362622,0.00021660198,0.00023418292,0.00014090099,0.0030358143,0.014033501,0.0012599394,0.8248726,0.14888361],"study_design_scores_gemma":[0.0006837379,0.0008002653,0.048163872,0.0007955172,0.00033426768,0.0017375294,0.0006400877,0.036313064,0.06902209,0.004428689,0.83666146,0.00041932784],"about_ca_topic_score_codex":0.012042031,"about_ca_topic_score_gemma":0.023357138,"teacher_disagreement_score":0.020838734,"about_ca_system_score_codex":0.0021807025,"about_ca_system_score_gemma":0.0022636263,"threshold_uncertainty_score":0.06971252},"labels":[],"label_agreement":null},{"id":"W2891670090","doi":"10.1109/icmla.2018.00109","title":"A Novel Neural Sequence Model with Multiple Attentions for Word Sense Disambiguation","year":2018,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Computer science; Bridging (networking); Word (group theory); Artificial intelligence; Natural language processing; Sentence; Sequence (biology); Encoder; Word-sense disambiguation; Linguistics","score_opus":0.052927163113188826,"score_gpt":0.3138115097703926,"score_spread":0.2608843466572038,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2891670090","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023951247,0.0010402118,0.9684987,0.0008969123,0.0003447439,0.00007158443,0.00031117877,0.001658009,0.0032274404],"genre_scores_gemma":[0.6950868,0.0012435949,0.2812599,0.0010815643,0.00031695465,0.00034227886,0.00096528075,0.00041522327,0.019288497],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995314,0.00011740607,0.000027404267,0.00018141612,0.00008122892,0.000061080944],"domain_scores_gemma":[0.9992717,0.00038441762,0.00006263917,0.00006802382,0.00015254373,0.000060664333],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009313125,0.0010451783,0.0010955532,0.0010744695,0.00058802444,0.001028107,0.0028541922,0.001766956,0.0045407545],"category_scores_gemma":[0.0027089487,0.00074837275,0.0010583351,0.001406606,0.0007824271,0.0029099318,0.0015484861,0.001983619,0.0012636472],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035871737,0.00021840923,0.0012800262,0.00020596784,0.00017944029,0.00044238588,0.00029788716,0.6314111,0.011192999,0.045949228,0.006857383,0.30160657],"study_design_scores_gemma":[0.000011216943,0.00002204611,0.00007331602,0.0000067896335,0.00001648891,0.000038359693,0.000006976785,0.98521835,0.0007443046,0.013149861,0.0007043528,0.000008026639],"about_ca_topic_score_codex":0.013077621,"about_ca_topic_score_gemma":0.018971521,"teacher_disagreement_score":0.013077621,"about_ca_system_score_codex":0.001297303,"about_ca_system_score_gemma":0.0019498653,"threshold_uncertainty_score":0.026003003},"labels":[],"label_agreement":null},{"id":"W2891958973","doi":"10.18653/v1/d18-1181","title":"Auto-Encoding Dictionary Definitions into Consistent Word Embeddings","year":2018,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":43,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"Natural Sciences and Engineering Research Council of Canada; Compute Canada","keywords":"Computer science; Word (group theory); Natural language processing; Artificial intelligence; Consistency (knowledge bases); Exploit; Encoding (memory); Simple (philosophy); Similarity (geometry); Semantics (computer science); Linguistics; Programming language","score_opus":0.029005931297088793,"score_gpt":0.2843189780748317,"score_spread":0.2553130467777429,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2891958973","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08180005,0.002190896,0.8801024,0.0009134843,0.0008132461,0.00024620118,0.012706396,0.010213358,0.011013939],"genre_scores_gemma":[0.35645562,0.0021197777,0.60251504,0.00043949822,0.00021262294,0.000249954,0.02853862,0.0029037022,0.006565204],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99815863,0.0005621931,0.00026963517,0.0006322197,0.00029165385,0.0000856824],"domain_scores_gemma":[0.994978,0.0016354058,0.00038437903,0.0017868239,0.0010892452,0.00012618354],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011334249,0.00085300184,0.0006693224,0.0029469808,0.00046871393,0.0020261623,0.0010720534,0.00058558263,0.007119608],"category_scores_gemma":[0.009132064,0.0006086561,0.00058068166,0.0036019408,0.00068832625,0.007436851,0.003823737,0.0015475847,0.0062268227],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00045370805,0.00037330293,0.008362723,0.0017852071,0.00030665303,0.00037535993,0.001697186,0.008534846,0.038379394,0.073239625,0.04971177,0.81678015],"study_design_scores_gemma":[0.0002117542,0.00064154447,0.00919461,0.0010734455,0.00047530714,0.002231875,0.004247467,0.21723206,0.10006191,0.3018422,0.36245972,0.00032817246],"about_ca_topic_score_codex":0.00090251974,"about_ca_topic_score_gemma":0.0034878748,"teacher_disagreement_score":0.007119608,"about_ca_system_score_codex":0.00041568375,"about_ca_system_score_gemma":0.0012006847,"threshold_uncertainty_score":0.02381742},"labels":[],"label_agreement":null},{"id":"W2892239351","doi":"10.18653/v1/d18-1276","title":"Free as in Free Word Order: An Energy Based Model for Word Segmentation and Morphological Tagging in Sanskrit","year":2018,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Sanskrit; Word order; Computer science; Natural language processing; Word (group theory); Artificial intelligence; Text segmentation; Linguistics; Segmentation; Philosophy","score_opus":0.021127114783148156,"score_gpt":0.2926867301244836,"score_spread":0.27155961534133544,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2892239351","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.45570245,0.0032488785,0.5206236,0.0029372424,0.00050022727,0.00013770116,0.00297317,0.0037605006,0.010116146],"genre_scores_gemma":[0.88592315,0.0006785068,0.09437165,0.00029498478,0.00012206828,0.00015878475,0.0032520434,0.0010005061,0.014198219],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996574,0.00013519029,0.000022724204,0.00010711151,0.000027270866,0.00005028922],"domain_scores_gemma":[0.99798226,0.0013191687,0.00011249253,0.00019341038,0.00028660824,0.00010603645],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013913282,0.00086164044,0.0009985747,0.0010967003,0.00094337115,0.0018725457,0.002115028,0.0016031311,0.0044150758],"category_scores_gemma":[0.0029158706,0.00074936857,0.0011675507,0.0010315217,0.0006702809,0.003737939,0.0014104478,0.0022253576,0.002296029],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015827277,0.00052095077,0.013695187,0.00026602362,0.00026945682,0.00072466146,0.001769621,0.7011802,0.0076986104,0.059350032,0.015351004,0.19759153],"study_design_scores_gemma":[0.000011257476,0.000019850486,0.00078872475,0.0000103171105,0.000019820862,0.000038718565,0.00006197562,0.97743875,0.00036029253,0.020627784,0.00060770725,0.000014897358],"about_ca_topic_score_codex":0.030262038,"about_ca_topic_score_gemma":0.06959287,"teacher_disagreement_score":0.030262038,"about_ca_system_score_codex":0.0012721095,"about_ca_system_score_gemma":0.0012727524,"threshold_uncertainty_score":0.060171843},"labels":[],"label_agreement":null},{"id":"W2893999801","doi":"","title":"The Canadian Writing Research Collaboratory: Infrastructure Development through Partnership.","year":2011,"lang":"en","type":"article","venue":"DH","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Collaboratory; General partnership; Political science; Computer science; World Wide Web","score_opus":0.09178839047304764,"score_gpt":0.3469579641236295,"score_spread":0.25516957365058185,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2893999801","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008748882,0.0052548037,0.033663835,0.1527484,0.01205048,0.0026428774,0.1613608,0.0146910455,0.6088389],"genre_scores_gemma":[0.04124943,0.002209968,0.035847694,0.00417353,0.0004210102,0.00063185353,0.030446375,0.0030833138,0.8819369],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.98965377,0.0010271909,0.0003872253,0.0009626513,0.006302628,0.0016665403],"domain_scores_gemma":[0.9021358,0.0041409424,0.0010885948,0.0042735967,0.071885474,0.016475609],"candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.012037919,0.00094564195,0.0008971419,0.0043175365,0.008059833,0.00756801,0.0025423744,0.0016810001,0.17263168],"category_scores_gemma":[0.037727427,0.0007250448,0.0006161436,0.006757905,0.0017217548,0.0045805275,0.0042735455,0.0026445552,0.056462917],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":true,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010609877,0.000024448582,0.0011226315,0.00019262457,0.00000971311,0.000107572014,0.0011240011,0.000111164474,0.000756583,0.01124756,0.9330881,0.052109398],"study_design_scores_gemma":[0.000029354711,0.000013960817,0.0024372516,0.000104050254,0.000011807535,0.000061970524,0.00145095,0.00027623013,0.0006112285,0.0011550576,0.9938285,0.000019673718],"about_ca_topic_score_codex":0.9021063,"about_ca_topic_score_gemma":0.91449046,"teacher_disagreement_score":0.992432,"about_ca_system_score_codex":0.031729463,"about_ca_system_score_gemma":0.22855668,"threshold_uncertainty_score":0.5775105},"labels":[],"label_agreement":null},{"id":"W2896672690","doi":"","title":"Identifying Infrequent Translations by Aligning Non Parallel Sentences.","year":2012,"lang":"en","type":"article","venue":"Conference of the Association for Machine Translation in the Americas","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Task (project management); Natural language processing; Simple (philosophy); Sequence (biology); Artificial intelligence; Machine translation","score_opus":0.045029642877369824,"score_gpt":0.31914862471164296,"score_spread":0.27411898183427313,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2896672690","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15351152,0.0011768875,0.8355861,0.000581983,0.00045127686,0.0003279002,0.0012395483,0.004065737,0.0030591218],"genre_scores_gemma":[0.37492865,0.00070841104,0.6101091,0.0005134111,0.00042284903,0.00037265345,0.0065622553,0.0011651632,0.005217524],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99705887,0.0011427894,0.0002777254,0.0008860521,0.00050986663,0.00012469773],"domain_scores_gemma":[0.9903616,0.0047185584,0.0012664903,0.0020217877,0.0014178407,0.00021386442],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002482415,0.0017447851,0.0012214978,0.0015842324,0.0012098385,0.0012144378,0.0014770285,0.001639977,0.0035784557],"category_scores_gemma":[0.013329964,0.00078972714,0.0008601209,0.0020641417,0.000876109,0.0030859269,0.0017342123,0.0022015152,0.0052249962],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001056261,0.0005293131,0.009472358,0.0013547088,0.0003658239,0.0017827515,0.0014029731,0.014513963,0.31141365,0.008528554,0.013655859,0.6359238],"study_design_scores_gemma":[0.00032601808,0.0016562983,0.018377144,0.00024721786,0.00066562,0.0053870506,0.002378393,0.4995561,0.33024135,0.07839905,0.062502764,0.00026294254],"about_ca_topic_score_codex":0.00073204073,"about_ca_topic_score_gemma":0.0021941033,"teacher_disagreement_score":0.0035784557,"about_ca_system_score_codex":0.00027758654,"about_ca_system_score_gemma":0.0012130245,"threshold_uncertainty_score":0.0131284},"labels":[],"label_agreement":null},{"id":"W2896787225","doi":"10.1007/978-3-030-01081-2_40","title":"Experiments in Learning to Solve Formal Analogical Equations","year":2018,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Artificial intelligence; Analogy; Translation (biology); Analogical reasoning; Natural language processing; Machine translation; Machine learning; Theoretical computer science; Linguistics","score_opus":0.027987214451111925,"score_gpt":0.30492314535687853,"score_spread":0.2769359309057666,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2896787225","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8851703,0.0008653091,0.07288857,0.0016713693,0.00059271435,0.0006115974,0.00090142206,0.0015652159,0.035733588],"genre_scores_gemma":[0.9452859,0.00049516064,0.039503496,0.00043188184,0.00010865106,0.0004349494,0.0014394772,0.00022126103,0.0120792845],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9972294,0.0011126928,0.0003330637,0.0005682675,0.00058296067,0.00017352472],"domain_scores_gemma":[0.9335599,0.059680894,0.0013701594,0.0036871524,0.0009811536,0.000720722],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028909044,0.0014219262,0.0006862211,0.00041615573,0.0004895418,0.0014818036,0.0026005793,0.0024517179,0.017990688],"category_scores_gemma":[0.05184861,0.0006563159,0.0006387484,0.00051238877,0.002079277,0.0052973754,0.0016683683,0.003160219,0.0016377385],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.018601237,0.036954574,0.0118881995,0.0048197447,0.00076342054,0.000906932,0.00648957,0.12702502,0.06599871,0.17294885,0.026433455,0.52717036],"study_design_scores_gemma":[0.009202161,0.015568296,0.0066186544,0.00043819094,0.00029004988,0.0006974673,0.001642754,0.4595186,0.08563501,0.39173397,0.028434088,0.00022074186],"about_ca_topic_score_codex":0.0010618035,"about_ca_topic_score_gemma":0.00069479796,"teacher_disagreement_score":0.017990688,"about_ca_system_score_codex":0.00081090716,"about_ca_system_score_gemma":0.00070709124,"threshold_uncertainty_score":0.060184836},"labels":[],"label_agreement":null},{"id":"W2896826757","doi":"10.1007/s10579-018-9430-2","title":"VERTa: a linguistic approach to automatic machine translation evaluation","year":2018,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Ministerio de Asuntos Económicos y Transformación Digital, Gobierno de España; Alberta-Pacific Forest Industries","keywords":"Metric (unit); Computer science; Machine translation; Natural language processing; Variety (cybernetics); Artificial intelligence; Linguistics; Engineering","score_opus":0.030387430137635726,"score_gpt":0.32967932798055893,"score_spread":0.2992918978429232,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2896826757","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022251097,0.0014121534,0.8927895,0.0006326675,0.0006652174,0.0015060542,0.00542806,0.058305547,0.0170097],"genre_scores_gemma":[0.13944376,0.0006178573,0.8248838,0.0004997357,0.00031803295,0.0016725303,0.014484224,0.0074338075,0.010646231],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.968594,0.020351551,0.002028133,0.0019133639,0.006417021,0.0006959033],"domain_scores_gemma":[0.9791027,0.009897619,0.00076051516,0.003166916,0.00660485,0.00046746773],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013276025,0.0022322086,0.0021224096,0.008677003,0.0023646296,0.0067567336,0.0036784755,0.0023178966,0.010808187],"category_scores_gemma":[0.030831141,0.0012158672,0.0015460697,0.0046509625,0.0013937664,0.006629603,0.005087656,0.0029534765,0.0055472804],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013630812,0.0010324456,0.0029721237,0.0019975563,0.0009867551,0.00048035118,0.0011558372,0.017858395,0.044398576,0.04210081,0.10496585,0.7806882],"study_design_scores_gemma":[0.0006484146,0.0011830562,0.0039980654,0.00047647796,0.00072166027,0.0010569688,0.0011184325,0.7118655,0.08275644,0.07799259,0.117762595,0.00041984738],"about_ca_topic_score_codex":0.004772209,"about_ca_topic_score_gemma":0.008031053,"teacher_disagreement_score":0.013276025,"about_ca_system_score_codex":0.0015547674,"about_ca_system_score_gemma":0.0035587277,"threshold_uncertainty_score":0.07021117},"labels":[],"label_agreement":null},{"id":"W28970257","doi":"10.1096/fj.201700599rr","title":"Structure morphosyntaxique et modélisation informatique de la langue malgache","year":2000,"lang":"en","type":"dissertation","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; National Natural Science Foundation of China","keywords":"Humanities; Philosophy","score_opus":0.004623504935908257,"score_gpt":0.2759809216660904,"score_spread":0.27135741673018215,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W28970257","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09015418,0.0030841802,0.8608774,0.0027925787,0.00080777623,0.00012664072,0.0030549415,0.0055650943,0.03353733],"genre_scores_gemma":[0.60841435,0.0038970998,0.34363005,0.0005548946,0.00026256964,0.0006260122,0.0044445987,0.0014526439,0.03671777],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99976844,0.00006570285,0.000008555953,0.00005810102,0.00007478241,0.00002448225],"domain_scores_gemma":[0.9997781,0.00010804096,0.000018143399,0.000035895784,0.000046520156,0.00001338677],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00039063737,0.00073243,0.00068402593,0.00079545396,0.000611437,0.0019194996,0.00088106445,0.001346393,0.0094594145],"category_scores_gemma":[0.0012524733,0.00038415013,0.001111385,0.0007542353,0.0009902074,0.0008640869,0.00074333197,0.0011600489,0.0023663817],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020773779,0.000050510043,0.0023212119,0.00038893227,0.00010965621,0.00065100245,0.0002998818,0.69225246,0.026192082,0.19131637,0.0084589785,0.07775122],"study_design_scores_gemma":[0.00004513397,0.000029399918,0.0011844169,0.000037499256,0.000017151177,0.00014097449,0.000059851267,0.92399746,0.0038865418,0.030476479,0.04009374,0.0000313148],"about_ca_topic_score_codex":0.024117613,"about_ca_topic_score_gemma":0.013983349,"teacher_disagreement_score":0.024117613,"about_ca_system_score_codex":0.0015947828,"about_ca_system_score_gemma":0.001177676,"threshold_uncertainty_score":0.0479545},"labels":[],"label_agreement":null},{"id":"W2897075944","doi":"10.3968/10526","title":"The Preliminary Study on the Construction of the Energy Power Corpus","year":2018,"lang":"en","type":"article","venue":"Studies in literature and language","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Selection (genetic algorithm); Energy (signal processing); Corpus linguistics; Power (physics); Computer science; Psychology; Natural language processing; Linguistics; Artificial intelligence; Statistics","score_opus":0.009398006402956523,"score_gpt":0.28287762761024066,"score_spread":0.27347962120728414,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2897075944","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.28615668,0.0026299506,0.3463759,0.014104529,0.0023533697,0.015284689,0.058346935,0.0029718499,0.2717761],"genre_scores_gemma":[0.3188232,0.0017709426,0.51723576,0.0015394281,0.00057581224,0.015848758,0.08179955,0.0028921152,0.05951446],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9955058,0.0028522993,0.00041290626,0.000504162,0.00057518744,0.00014962364],"domain_scores_gemma":[0.9780961,0.012412158,0.00044618486,0.0021371192,0.0062600677,0.00064838986],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0055113253,0.0003945021,0.0004902278,0.004203695,0.0034808232,0.0033333662,0.0008979124,0.0008168786,0.02468791],"category_scores_gemma":[0.025070604,0.00065483485,0.00029012488,0.0058076703,0.0019017463,0.0036249459,0.0027445913,0.0028114228,0.008519919],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007081391,0.0009140896,0.012752091,0.0032656763,0.000022915605,0.0037478816,0.08491981,0.0028909536,0.05847429,0.13804384,0.20413262,0.49012768],"study_design_scores_gemma":[0.00007275078,0.00026600092,0.0141020715,0.0006938774,0.000027861068,0.0011462842,0.022341287,0.0040206388,0.019099412,0.009752093,0.92838013,0.00009751633],"about_ca_topic_score_codex":0.0058177034,"about_ca_topic_score_gemma":0.00801219,"teacher_disagreement_score":0.02468791,"about_ca_system_score_codex":0.0015020117,"about_ca_system_score_gemma":0.003599592,"threshold_uncertainty_score":0.08258933},"labels":[],"label_agreement":null},{"id":"W2897455045","doi":"10.1515/cjal-2018-0013","title":"Stress as a Suprasegmental Phonological Shift in Translation: A New Category of Linguistic Shifts","year":2018,"lang":"en","type":"article","venue":"Chinese Journal of Applied Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Linguistics; Stress (linguistics); Phonology; Translation (biology); Categorization; Computer science; Grammar; Dynamic and formal equivalence; Natural language processing; Psychology; Machine translation; Philosophy","score_opus":0.014800754529321268,"score_gpt":0.2920307838065182,"score_spread":0.2772300292771969,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2897455045","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.69314295,0.0015214715,0.11091271,0.004802886,0.00071420224,0.00008290613,0.00012386621,0.00020948594,0.18848959],"genre_scores_gemma":[0.99416417,0.00014637434,0.0032588665,0.00013477656,0.00012229345,0.000020950796,0.000020915384,0.000026002943,0.0021057543],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"qualitative","domain_scores_codex":[0.99929035,0.00021640318,0.00004883816,0.0001485531,0.00021173319,0.000084095074],"domain_scores_gemma":[0.99864405,0.00051934883,0.00017173492,0.0003170051,0.00025919854,0.00008859704],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00064960937,0.00041523227,0.00027948793,0.0015223549,0.0013935746,0.0030277933,0.00054797524,0.0011154759,0.0028038505],"category_scores_gemma":[0.0017655998,0.00016825313,0.00043793576,0.0011266378,0.006821849,0.0049883984,0.0022202418,0.0016113089,0.00032176494],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018867152,0.00006616958,0.007289901,0.0001479065,0.000016658867,0.00079102186,0.02440041,0.0006919843,0.015156758,0.88945186,0.0011690145,0.060629707],"study_design_scores_gemma":[0.000030738967,0.00037814537,0.02613886,0.00015916802,0.00006189582,0.002060466,0.020973437,0.018139886,0.0108607365,0.8789148,0.042168338,0.00011358456],"about_ca_topic_score_codex":0.0006632522,"about_ca_topic_score_gemma":0.00059136923,"teacher_disagreement_score":0.0030277933,"about_ca_system_score_codex":0.0012210105,"about_ca_system_score_gemma":0.0006680042,"threshold_uncertainty_score":0.009379864},"labels":[],"label_agreement":null},{"id":"W2897977929","doi":"10.25205/1818-7900-2018-16-3-74-86","title":"Developing the System for Automatic Summarization of Scientific Texts.","year":2018,"lang":"en","type":"article","venue":"Vestnik NSU Series Information Technologies","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Automatic summarization; Computer science; Rhetorical question; Natural language processing; Computational linguistics; Linguistics; Artificial intelligence; Information retrieval; Library science; World Wide Web; Philosophy","score_opus":0.01290734020264808,"score_gpt":0.25144144180618017,"score_spread":0.23853410160353208,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2897977929","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0062001385,0.0017342463,0.69303644,0.0014747266,0.0008437795,0.0022729496,0.017088829,0.26895595,0.008392861],"genre_scores_gemma":[0.018079732,0.00075460365,0.9221721,0.00041105493,0.0003307111,0.0016688498,0.038443644,0.005490699,0.012648548],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9978015,0.0006007695,0.0003676271,0.00058784045,0.0005214362,0.00012081574],"domain_scores_gemma":[0.9963086,0.0012817198,0.00029858155,0.00042614347,0.0014770637,0.00020788924],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037636254,0.0020100437,0.0016587852,0.0041056927,0.001368181,0.0039524646,0.0018793036,0.0017769574,0.026379531],"category_scores_gemma":[0.009710948,0.0014591056,0.0015119093,0.0025928617,0.0005415703,0.0048167533,0.0028434761,0.002033307,0.040887024],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005436277,0.00016352064,0.0011564567,0.0027175492,0.00023684467,0.00077479763,0.0013702023,0.0017625502,0.05820713,0.0052779475,0.22200893,0.7057804],"study_design_scores_gemma":[0.00081882137,0.0010386931,0.0051728725,0.0007133756,0.00050035294,0.0026941916,0.0011203253,0.090988986,0.143866,0.016142039,0.7366221,0.0003223223],"about_ca_topic_score_codex":0.0017334998,"about_ca_topic_score_gemma":0.0019871811,"teacher_disagreement_score":0.026379531,"about_ca_system_score_codex":0.0008924188,"about_ca_system_score_gemma":0.0022610312,"threshold_uncertainty_score":0.08824831},"labels":[],"label_agreement":null},{"id":"W2898671076","doi":"10.25159/1013-8471/3010","title":"The Accordance Hebrew Syntactic Database Project","year":2018,"lang":"en","type":"article","venue":"Journal for Semitics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Hebrew; Syntax; Database; Computer science; Linguistics; Hebrew Bible; History; Natural language processing; Philosophy; Biblical studies; Archaeology","score_opus":0.028687413463313518,"score_gpt":0.3544127414779752,"score_spread":0.3257253280146617,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2898671076","genre_codex":"methods","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007323444,0.0035466019,0.38613683,0.0090395,0.003700841,0.0011840595,0.27156678,0.039135724,0.2783663],"genre_scores_gemma":[0.044561055,0.0046977326,0.24427257,0.002574588,0.0010980814,0.0019225996,0.5424135,0.016492803,0.14196718],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9896653,0.003585829,0.001728241,0.0015079752,0.0030084683,0.0005042183],"domain_scores_gemma":[0.9769125,0.004169539,0.0011899952,0.009821997,0.0063320673,0.0015739087],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013553566,0.0012202878,0.0013916568,0.0073959576,0.0028584744,0.011431146,0.0030597781,0.002251992,0.123797856],"category_scores_gemma":[0.031095749,0.0014399724,0.0010460254,0.011860451,0.0017921993,0.015997184,0.00559382,0.0028750761,0.11592201],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00034103543,0.00008329512,0.000810955,0.0004894793,0.000049416423,0.0001433884,0.00083974237,0.000440954,0.002309328,0.21233867,0.6143591,0.16779467],"study_design_scores_gemma":[0.000038308623,0.000020054977,0.000509593,0.00015752633,0.000017692211,0.000096352145,0.00022951815,0.00073141226,0.0013121625,0.028134646,0.96872,0.000032525004],"about_ca_topic_score_codex":0.012754233,"about_ca_topic_score_gemma":0.0053892536,"teacher_disagreement_score":0.123797856,"about_ca_system_score_codex":0.0032862676,"about_ca_system_score_gemma":0.010272073,"threshold_uncertainty_score":0.4141451},"labels":[],"label_agreement":null},{"id":"W2898685865","doi":"10.18653/v1/k18-3015","title":"Combining Neural and Non-Neural Methods for Low-Resource Morphological Reinflection","year":2018,"lang":"en","type":"article","venue":"Proceedings of the","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Innovates","keywords":"Computer science; Task (project management); Resource (disambiguation); Artificial intelligence; Artificial neural network; Transduction (biophysics); Machine learning; Natural language processing; Biology","score_opus":0.02031761143215918,"score_gpt":0.3287717414730686,"score_spread":0.30845413004090944,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2898685865","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07136609,0.0027353393,0.7816242,0.0012583772,0.0008651158,0.00050894334,0.0015914041,0.1167343,0.023316287],"genre_scores_gemma":[0.25404283,0.0009114625,0.70688933,0.0010639566,0.00028457496,0.00032543333,0.0078647025,0.005699332,0.022918386],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99657875,0.00062961376,0.0002786247,0.0012150495,0.0009673298,0.00033075703],"domain_scores_gemma":[0.9938114,0.0019041965,0.00028171425,0.002547881,0.0012345267,0.00022034373],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004677345,0.0032596332,0.0019436901,0.00287168,0.0014045266,0.0036951047,0.005257884,0.002530496,0.01286686],"category_scores_gemma":[0.00828168,0.001058985,0.001595376,0.0020952795,0.0013690067,0.0073224073,0.0053880136,0.003303003,0.014673848],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00045425745,0.00043969945,0.002018333,0.00070736086,0.00030647425,0.00036819198,0.00031123863,0.015741866,0.070949,0.0044664135,0.024399424,0.87983775],"study_design_scores_gemma":[0.00017254418,0.0005039889,0.0036027979,0.00024455838,0.0003561722,0.0013730078,0.00054604764,0.73998684,0.17479059,0.028258068,0.049928308,0.00023701217],"about_ca_topic_score_codex":0.0038342609,"about_ca_topic_score_gemma":0.011054052,"teacher_disagreement_score":0.01286686,"about_ca_system_score_codex":0.0012000633,"about_ca_system_score_gemma":0.0014146166,"threshold_uncertainty_score":0.04304391},"labels":[],"label_agreement":null},{"id":"W2898954967","doi":"10.1515/cllt-2018-0033","title":"An information-theoretic view on language complexity and register variation: Compressing naturalistic corpus data","year":2018,"lang":"en","type":"article","venue":"Corpus Linguistics and Linguistic Theory","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Variation (astronomy); Register (sociolinguistics); Linguistics; Formality; Context (archaeology); Conversation; Sentence; Natural language processing; Artificial intelligence","score_opus":0.030986459640575396,"score_gpt":0.31101361548343714,"score_spread":0.28002715584286175,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2898954967","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.52276623,0.00083631976,0.46586278,0.0017027998,0.00006656212,0.00013543585,0.0023377207,0.00031443837,0.00597774],"genre_scores_gemma":[0.9300447,0.00028098043,0.06617871,0.00011028059,0.000102883336,0.00020949445,0.0022455975,0.000091502836,0.00073588546],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99485743,0.0025011997,0.0004999235,0.00070893375,0.0011968627,0.00023569129],"domain_scores_gemma":[0.93919164,0.046296556,0.0040483326,0.0069313087,0.002918628,0.0006135644],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0062937005,0.00044196035,0.0008239108,0.0073994272,0.0010224422,0.004275642,0.001121405,0.00076094904,0.0020327652],"category_scores_gemma":[0.06627839,0.00039685465,0.0006943898,0.0066708843,0.0046994584,0.0063383807,0.0029210101,0.0015518802,0.00025482487],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014370024,0.0002432365,0.08093922,0.00083637715,0.00044628055,0.0007308841,0.008534563,0.14817846,0.0155187445,0.38819987,0.0041202763,0.35081506],"study_design_scores_gemma":[0.000055237644,0.0003093527,0.07125997,0.00021410424,0.000113471375,0.0007067573,0.0033803245,0.43906128,0.0067387405,0.4703301,0.0076021664,0.00022860568],"about_ca_topic_score_codex":0.004119723,"about_ca_topic_score_gemma":0.0032378032,"teacher_disagreement_score":0.0073994272,"about_ca_system_score_codex":0.0020709608,"about_ca_system_score_gemma":0.0009964107,"threshold_uncertainty_score":0.033284664},"labels":[],"label_agreement":null},{"id":"W2900029582","doi":"10.22215/etd/2018-13317","title":"Disregarding Linguistics: A Critical Study of Google Translate's Syntactic/Semantic Errors in Rendering Multiword Units in English to Persian Translations","year":2018,"lang":"en","type":"dissertation","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Natural language processing; Machine translation; Persian; Artificial intelligence; Rendering (computer graphics); Linguistics; Source text; Corpus linguistics; Rule-based machine translation","score_opus":0.025541176067927128,"score_gpt":0.342479134050637,"score_spread":0.3169379579827099,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2900029582","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9575394,0.003546364,0.008089044,0.008054685,0.00034263454,0.00020062375,0.0003323795,0.00019885815,0.021695929],"genre_scores_gemma":[0.98367983,0.0019793953,0.009367545,0.0012833008,0.00014480487,0.00006864205,0.00031790836,0.00040068862,0.0027579179],"study_design_codex":"qualitative","study_design_gemma":"observational","domain_scores_codex":[0.9888206,0.0059374054,0.00092229666,0.0008098983,0.003168897,0.00034092495],"domain_scores_gemma":[0.8861027,0.0865295,0.0048593883,0.003329962,0.018807674,0.00037081298],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0130656995,0.00038074795,0.0003522149,0.004118833,0.0037363805,0.0038502084,0.00069966516,0.0008073774,0.0011187983],"category_scores_gemma":[0.08244186,0.00038603344,0.00020440455,0.005091077,0.0056139403,0.0037780886,0.0017068994,0.0015443704,0.00046655067],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00063940673,0.00019103954,0.050032463,0.0023976713,0.000054740693,0.0052488353,0.5687976,0.0013344204,0.016075255,0.022961847,0.016039215,0.31622753],"study_design_scores_gemma":[0.0000716557,0.00064566376,0.16192368,0.0020032178,0.0002569226,0.009670094,0.435163,0.007561785,0.0745772,0.017592302,0.29029772,0.00023680332],"about_ca_topic_score_codex":0.0148086455,"about_ca_topic_score_gemma":0.025891768,"teacher_disagreement_score":0.0148086455,"about_ca_system_score_codex":0.0026737729,"about_ca_system_score_gemma":0.0033206518,"threshold_uncertainty_score":0.06909883},"labels":[],"label_agreement":null},{"id":"W2900379213","doi":"10.1007/978-3-030-03840-3_9","title":"CH1: A Conversational System to Calculate Carbohydrates in a Meal","year":2018,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Meal; Computer science; Food science; Chemistry","score_opus":0.011343527461812492,"score_gpt":0.2508860938175988,"score_spread":0.2395425663557863,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2900379213","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.100891575,0.00091549085,0.6647603,0.0010365322,0.00042261035,0.0010647549,0.005767765,0.19025458,0.034886416],"genre_scores_gemma":[0.51933074,0.00035886804,0.433214,0.0007339338,0.00017140845,0.00086223544,0.0068687964,0.0029015318,0.035558365],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995034,0.00016338372,0.00003077722,0.00015626772,0.00009400341,0.00005221291],"domain_scores_gemma":[0.99897975,0.0006414683,0.000039778595,0.00009638808,0.0001329032,0.00010967731],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009649083,0.0011935419,0.00045954372,0.0004680632,0.0006170554,0.0009687043,0.0016546596,0.0012902361,0.020110095],"category_scores_gemma":[0.0027748432,0.00043778803,0.00040623156,0.00031479128,0.00032008308,0.0015668516,0.0014727442,0.00074005063,0.006614882],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0048615793,0.0010366295,0.0076487577,0.0024519581,0.00025585174,0.0018308361,0.009042968,0.010868385,0.13677405,0.013351592,0.1408926,0.67098475],"study_design_scores_gemma":[0.0006847835,0.0013970007,0.013774452,0.00033942828,0.00039269187,0.0031949957,0.002670479,0.5344003,0.14378685,0.022557404,0.27621526,0.00058637693],"about_ca_topic_score_codex":0.003693376,"about_ca_topic_score_gemma":0.0023927893,"teacher_disagreement_score":0.020110095,"about_ca_system_score_codex":0.00048157555,"about_ca_system_score_gemma":0.000811893,"threshold_uncertainty_score":0.06727499},"labels":[],"label_agreement":null},{"id":"W2901196901","doi":"10.3390/info9110290","title":"Annotating a Low-Resource Language with LLOD Technology: Sumerian Morphology and Syntax","year":2018,"lang":"en","type":"article","venue":"Information","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"National Endowment for the Humanities; International Institute of Information Technology, Hyderabad; Bundesministerium für Bildung und Forschung; University of California, Los Angeles; Social Sciences and Humanities Research Council of Canada; Deutsche Forschungsgemeinschaft","keywords":"Sumerian; Syntax; Computer science; Resource (disambiguation); Annotation; Languages of Asia; Linguistics; Natural language processing; Artificial intelligence","score_opus":0.0029036296242360123,"score_gpt":0.22280456179319832,"score_spread":0.2199009321689623,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2901196901","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3565407,0.0026169375,0.44521323,0.0043783677,0.0005653943,0.0004887828,0.031293575,0.0055030617,0.15339996],"genre_scores_gemma":[0.6451112,0.0016074871,0.30191353,0.0006606661,0.00011771455,0.0006954488,0.02897847,0.0016005277,0.019314926],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.998998,0.0003322668,0.00016625156,0.00026840175,0.00018537135,0.000049677372],"domain_scores_gemma":[0.9976622,0.00080478954,0.00027443518,0.0007329478,0.0004654239,0.00006026368],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012896445,0.00044754657,0.0003839356,0.005719357,0.0018245911,0.0024430335,0.0006504457,0.00051586545,0.0071877483],"category_scores_gemma":[0.005074523,0.00032923368,0.0003330631,0.0050473367,0.0017431222,0.004414556,0.0036879492,0.0007902277,0.0018235762],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00027137098,0.00012634121,0.035647437,0.003290666,0.00010988004,0.0028373932,0.04663411,0.00351502,0.05619928,0.23916782,0.03377233,0.57842845],"study_design_scores_gemma":[0.000011043371,0.00003306949,0.04154538,0.0009192476,0.00007453647,0.002097194,0.009791815,0.009911637,0.026723273,0.052144717,0.856631,0.000117053605],"about_ca_topic_score_codex":0.0046672863,"about_ca_topic_score_gemma":0.0067483964,"teacher_disagreement_score":0.0071877483,"about_ca_system_score_codex":0.0017522065,"about_ca_system_score_gemma":0.0016136043,"threshold_uncertainty_score":0.024045408},"labels":[],"label_agreement":null},{"id":"W2901256112","doi":"","title":"Proceedings of the EACL 2012 Joint Workshop of LINGVIS & UNCLH","year":2012,"lang":"en","type":"article","venue":"Conference of the European Chapter of the Association for Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":35,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Joint (building); Engineering; Architectural engineering","score_opus":0.039615400340741805,"score_gpt":0.262296469403868,"score_spread":0.22268106906312624,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2901256112","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.073083155,0.026583051,0.3451776,0.15931077,0.11372226,0.0012765597,0.01784534,0.024360131,0.23864117],"genre_scores_gemma":[0.120989785,0.005623701,0.11167577,0.010568208,0.00788062,0.00065603567,0.028688,0.008760893,0.705157],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9974216,0.0011552777,0.00010411092,0.00038408686,0.0005794085,0.00035549322],"domain_scores_gemma":[0.99274373,0.0016110007,0.00012816463,0.0009845545,0.0029968154,0.001535744],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006566669,0.001408004,0.001497508,0.0013659851,0.003047231,0.007731616,0.0021925587,0.0027640422,0.07442675],"category_scores_gemma":[0.008555559,0.00061946514,0.00088555645,0.0012629769,0.0010327637,0.0048972866,0.004806362,0.0030559218,0.028885089],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009567216,0.00034411575,0.00075226545,0.00029218514,0.000031959586,0.000322719,0.000698695,0.0007707875,0.004118605,0.005317551,0.835488,0.15090643],"study_design_scores_gemma":[0.00013534972,0.00012790042,0.0014194795,0.00017751439,0.00006268343,0.0002537977,0.0011031732,0.005530657,0.0066431905,0.00737358,0.9771114,0.000061242936],"about_ca_topic_score_codex":0.0118119735,"about_ca_topic_score_gemma":0.039951816,"teacher_disagreement_score":0.07442675,"about_ca_system_score_codex":0.002458793,"about_ca_system_score_gemma":0.0046981894,"threshold_uncertainty_score":0.24898225},"labels":[],"label_agreement":null},{"id":"W2901297050","doi":"10.1556/084.2018.19.2.5","title":"How to approach translation in a financial news corpus?","year":2018,"lang":"en","type":"article","venue":"Across Languages and Cultures","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Concordia University; Université de Montréal","funders":"","keywords":"Newspaper; Corpus linguistics; Translation (biology); Relation (database); Linguistics; Translation studies; Computer science; Advertising; Artificial intelligence; Business","score_opus":0.013505810497312554,"score_gpt":0.3138604127573024,"score_spread":0.3003546022599899,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2901297050","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01524612,0.006162219,0.9324817,0.016819129,0.0025234455,0.0015108723,0.0057805856,0.0034666713,0.01600934],"genre_scores_gemma":[0.057647146,0.002402069,0.9197248,0.0022122737,0.00087276025,0.0037259525,0.008323645,0.0014647547,0.0036264819],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9512018,0.035076033,0.0041528456,0.0047853673,0.0041322513,0.0006516853],"domain_scores_gemma":[0.8935286,0.06743171,0.0040387483,0.016261559,0.01748477,0.0012547055],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04115063,0.0018494794,0.002470668,0.015415574,0.004692765,0.015281307,0.0037595727,0.0041446257,0.009595068],"category_scores_gemma":[0.15057297,0.002060548,0.0018838823,0.0179552,0.0053005996,0.024267377,0.0107022915,0.0060564512,0.006615537],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002898341,0.00032677766,0.007037487,0.0057805344,0.00057534646,0.0013209773,0.022100622,0.005852066,0.009862665,0.20007128,0.048537225,0.6982452],"study_design_scores_gemma":[0.00026881424,0.00021052458,0.0074211336,0.0028646295,0.000332693,0.0019861753,0.030462133,0.048532855,0.0112786,0.47487742,0.42136294,0.00040214375],"about_ca_topic_score_codex":0.007268463,"about_ca_topic_score_gemma":0.0073425695,"teacher_disagreement_score":0.04115063,"about_ca_system_score_codex":0.0036259657,"about_ca_system_score_gemma":0.007032013,"threshold_uncertainty_score":0.21762794},"labels":[],"label_agreement":null},{"id":"W2902103202","doi":"10.1075/ts.18002.can","title":"Of ostriches, pyramids, and Swiss cheese","year":2018,"lang":"en","type":"article","venue":"Translation Spaces","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canfor (Canada)","funders":"","keywords":"Damages; Translation (biology); Risk analysis (engineering); Computer science; Business; Forensic engineering; Engineering; Political science; Law","score_opus":0.020890222124875076,"score_gpt":0.28423578720950166,"score_spread":0.2633455650846266,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2902103202","genre_codex":"empirical","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5853155,0.0028520257,0.0832105,0.00917892,0.0004063735,0.00015087359,0.0013177883,0.0007756041,0.31679243],"genre_scores_gemma":[0.9709425,0.0004843123,0.006641435,0.0002537507,0.00004491705,0.000028864184,0.0004025776,0.00010063518,0.021101063],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9989133,0.0003757857,0.00006564753,0.00021667314,0.00031399704,0.000114604074],"domain_scores_gemma":[0.99726933,0.0011075401,0.00061619503,0.00053813134,0.000376462,0.00009224445],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013253727,0.0004801139,0.00040747446,0.0014506729,0.0015530969,0.0026834987,0.00057184935,0.0010280116,0.03292725],"category_scores_gemma":[0.007835994,0.00023008822,0.000696232,0.0016957838,0.0037447196,0.003964036,0.0018853731,0.0011743014,0.0022544854],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006331081,0.00009107939,0.013616007,0.00033389998,0.00010447215,0.0025845543,0.0063858773,0.016163476,0.0027266121,0.85418105,0.012394525,0.0907854],"study_design_scores_gemma":[0.00006360008,0.00020170047,0.025317928,0.00026832282,0.00007069382,0.0012982256,0.0048893075,0.040296555,0.0023251222,0.84801376,0.077183,0.00007179305],"about_ca_topic_score_codex":0.0103641655,"about_ca_topic_score_gemma":0.0071733594,"teacher_disagreement_score":0.03292725,"about_ca_system_score_codex":0.0016803147,"about_ca_system_score_gemma":0.0005777023,"threshold_uncertainty_score":0.1101526},"labels":[],"label_agreement":null},{"id":"W2903119096","doi":"10.3138/cmlr.2017-0093","title":"Productive Collocation Knowledge at Advanced CEFR Levels: Evidence from the Development of a Test for Advanced L2 French","year":2018,"lang":"fr","type":"article","venue":"Canadian Modern Language Review/ La Revue canadienne des langues vivantes","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Political science; Philosophy","score_opus":0.04152639122049247,"score_gpt":0.2967829733305588,"score_spread":0.25525658211006635,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2903119096","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9914892,0.00019866113,0.0006808863,0.00020988639,0.000019926803,0.000104781844,0.00036135362,0.000029736953,0.006905522],"genre_scores_gemma":[0.99475855,0.000092715665,0.0010617811,0.00017807052,0.000013501265,0.0003766655,0.00064456527,0.000028024537,0.0028462035],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9892218,0.0054079248,0.0005522721,0.0018262377,0.0022343697,0.0007574531],"domain_scores_gemma":[0.917508,0.050892904,0.008157589,0.0052981446,0.014743423,0.0034000298],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.015339623,0.0010394502,0.00076039555,0.0022078224,0.001647071,0.003832529,0.0014903062,0.0017071652,0.007704238],"category_scores_gemma":[0.05552871,0.0005056725,0.0009873888,0.0017952492,0.0021997257,0.003925619,0.002907028,0.0015610467,0.0023453003],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020467665,0.0016101367,0.8143234,0.00069076096,0.0006011261,0.0011236375,0.08597057,0.0009906482,0.005387558,0.002331278,0.0033441724,0.081579864],"study_design_scores_gemma":[0.00012554579,0.0016781006,0.9685177,0.00016720603,0.00012575154,0.0002631007,0.016560655,0.0013448824,0.0015828885,0.0007830809,0.008770098,0.00008102058],"about_ca_topic_score_codex":0.06143039,"about_ca_topic_score_gemma":0.062212262,"teacher_disagreement_score":0.06143039,"about_ca_system_score_codex":0.0029680424,"about_ca_system_score_gemma":0.0024212776,"threshold_uncertainty_score":0.12214565},"labels":[],"label_agreement":null},{"id":"W2903182367","doi":"10.18653/v1/w18-6480","title":"Measuring sentence parallelism using Mahalanobis distances: The NRC unsupervised submissions to the WMT18 Parallel Corpus Filtering shared task","year":2018,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Computer science; Natural language processing; Task (project management); Artificial intelligence; Sample (material); Word (group theory); Sentence; Parallelism (grammar); Speech recognition; Linguistics; Parallel computing","score_opus":0.056489036960198226,"score_gpt":0.2793418193547273,"score_spread":0.22285278239452908,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2903182367","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4684553,0.0032351506,0.34130645,0.003369701,0.004417259,0.0033284682,0.083684966,0.06294405,0.029258642],"genre_scores_gemma":[0.4752846,0.00037329248,0.3301123,0.0011208578,0.00063864735,0.004535564,0.16395752,0.010577745,0.013399402],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9779375,0.009440991,0.0020179146,0.00499182,0.004666046,0.00094569457],"domain_scores_gemma":[0.9380894,0.026931826,0.0017065132,0.014177699,0.016394295,0.0027002634],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.016204178,0.0036062477,0.0022346315,0.0039298725,0.0031552177,0.0039621214,0.0032709872,0.0034441787,0.008341022],"category_scores_gemma":[0.08477108,0.0010474944,0.0020140547,0.003550501,0.0014256883,0.0042474456,0.00795058,0.00515909,0.009981167],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0034787671,0.001520134,0.017171035,0.0029356664,0.0012284438,0.0009253456,0.004328132,0.018652387,0.058509555,0.0036286958,0.38588092,0.501741],"study_design_scores_gemma":[0.0025425523,0.0038765797,0.08684748,0.0006305628,0.00090349483,0.0022858835,0.0054401113,0.41264507,0.15041101,0.037801635,0.29536733,0.0012483827],"about_ca_topic_score_codex":0.0068715084,"about_ca_topic_score_gemma":0.014884226,"teacher_disagreement_score":0.016204178,"about_ca_system_score_codex":0.0017511502,"about_ca_system_score_gemma":0.003859224,"threshold_uncertainty_score":0.085696936},"labels":[],"label_agreement":null},{"id":"W2903213390","doi":"10.1007/s11050-018-9146-2","title":"Factive islands and meaning-driven unacceptability","year":2018,"lang":"en","type":"article","venue":"Natural Language Semantics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":34,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Social Sciences and Humanities Research Council of Canada; Agence Nationale de la Recherche","keywords":"Meaning (existential); Triviality; Linguistics; Epistemology; Sentence; Psychology; Denotation (semiotics); Philosophy of language; Logical consequence; Philosophy; Metaphysics; Mathematics; Semiotics","score_opus":0.0069199576637714155,"score_gpt":0.27415016142679355,"score_spread":0.26723020376302214,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2903213390","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16320303,0.00061358686,0.77327985,0.0062439293,0.00039879826,0.00010956737,0.00047977432,0.0026733659,0.052998032],"genre_scores_gemma":[0.9501458,0.00020234754,0.042686693,0.00050274585,0.0002216858,0.00009970268,0.00028029535,0.0008360886,0.005024708],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99383646,0.0021382284,0.000457326,0.0009659661,0.0018583196,0.0007436966],"domain_scores_gemma":[0.96188635,0.027388915,0.0011220055,0.005732762,0.0032757786,0.0005942443],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0043784515,0.0006319366,0.0012338138,0.0020491711,0.0026762432,0.005114884,0.002529684,0.002556041,0.009089757],"category_scores_gemma":[0.036212165,0.0013096749,0.0021848143,0.0014662064,0.008612994,0.02053835,0.0066810125,0.0073420587,0.00082701974],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000041454685,0.00003004172,0.0006619164,0.00008167939,0.000020562933,0.00024185829,0.0011353922,0.0027678304,0.00088757224,0.9872479,0.0013162084,0.0055676284],"study_design_scores_gemma":[0.0000061273554,0.0000042715624,0.00009206135,0.0000072741946,0.000012717055,0.00006023757,0.00009937133,0.009162115,0.00074100436,0.9887931,0.0010084254,0.000013313995],"about_ca_topic_score_codex":0.0021377697,"about_ca_topic_score_gemma":0.0018664908,"teacher_disagreement_score":0.009089757,"about_ca_system_score_codex":0.0017053531,"about_ca_system_score_gemma":0.0014264154,"threshold_uncertainty_score":0.030408323},"labels":[],"label_agreement":null},{"id":"W2903297715","doi":"10.18653/v1/w18-6481","title":"Accurate semantic textual similarity for cleaning noisy parallel corpora using semantic machine translation evaluation metric: The NRC supervised submissions to the Parallel Corpus Filtering task","year":2018,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":35,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Machine translation; Semantic similarity; Redundancy (engineering); Similarity (geometry); Metric (unit); Task (project management); Word (group theory); Fluency; Information retrieval","score_opus":0.08794199865200938,"score_gpt":0.3452897288338443,"score_spread":0.2573477301818349,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2903297715","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.49210894,0.0025518392,0.3946506,0.0025947182,0.0023110243,0.0023773103,0.015996438,0.062294412,0.025114745],"genre_scores_gemma":[0.44384375,0.00038828937,0.48336515,0.000526301,0.00041121052,0.0017427681,0.049709864,0.007612716,0.012399868],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9824157,0.00684691,0.0016590884,0.0024912185,0.0058953897,0.00069161947],"domain_scores_gemma":[0.9556187,0.009622455,0.0016593179,0.00950869,0.02211878,0.0014720889],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011535069,0.0022591215,0.0026774146,0.0042607444,0.0028955964,0.002637095,0.0021461893,0.002075648,0.004677496],"category_scores_gemma":[0.043567687,0.0008788451,0.0012373816,0.0039296392,0.0014479783,0.0026269525,0.0035777227,0.0024286266,0.0045692185],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0022824032,0.0014359083,0.011426358,0.0027931684,0.000561348,0.0016303731,0.0028444272,0.03295928,0.08865234,0.004919149,0.16886601,0.68162924],"study_design_scores_gemma":[0.0010062216,0.002403883,0.026312796,0.0003291142,0.00042402477,0.0027693217,0.0020933782,0.5517488,0.264708,0.010720575,0.13692077,0.0005630883],"about_ca_topic_score_codex":0.00863625,"about_ca_topic_score_gemma":0.013748791,"teacher_disagreement_score":0.011535069,"about_ca_system_score_codex":0.0018843443,"about_ca_system_score_gemma":0.005185815,"threshold_uncertainty_score":0.061003983},"labels":[],"label_agreement":null},{"id":"W2903299149","doi":"10.5539/ijel.v8n7p59","title":"Modal Verbs Hedging: The Uses and Functions of “Will” and “Shall” in Nigerian Legal Discourse","year":2018,"lang":"en","type":"article","venue":"International Journal of English Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Natural language processing; Modal; Computer science; Linguistics; Concordance; Modal verb; Artificial intelligence; Verb; Philosophy; Medicine; Chemistry","score_opus":0.01043005624016267,"score_gpt":0.29505902337570705,"score_spread":0.28462896713554436,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2903299149","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.95234984,0.00294357,0.0036389423,0.0005325008,0.00006300135,0.00016229347,0.0010646253,0.000027127937,0.039217964],"genre_scores_gemma":[0.99073714,0.001426081,0.0041333893,0.000071112685,0.00001442085,0.00012686447,0.00067266746,0.000029335772,0.00278902],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.99898285,0.00049719797,0.000118563126,0.00014736758,0.00019118018,0.000062760664],"domain_scores_gemma":[0.9952791,0.003811403,0.0004145718,0.00015451109,0.00028275684,0.000057788242],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018030643,0.00023836643,0.00032598764,0.0030179792,0.0018064256,0.0022458846,0.0002623496,0.0004539272,0.0029287154],"category_scores_gemma":[0.0053545046,0.0002407528,0.000090458256,0.0038651915,0.0021594628,0.002504785,0.0012545532,0.00060241297,0.00023625133],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00044318466,0.00012698791,0.059926156,0.0035114842,0.00002258912,0.0034673917,0.5480885,0.00062269473,0.017983716,0.12020053,0.00719238,0.23841451],"study_design_scores_gemma":[0.00003101104,0.000118880176,0.19265336,0.0033823384,0.00007141554,0.004019029,0.45523202,0.0027516766,0.012506851,0.014389492,0.31475192,0.0000920408],"about_ca_topic_score_codex":0.004932246,"about_ca_topic_score_gemma":0.010984322,"teacher_disagreement_score":0.004932246,"about_ca_system_score_codex":0.0014769258,"about_ca_system_score_gemma":0.0012548568,"threshold_uncertainty_score":0.010715902},"labels":[],"label_agreement":null},{"id":"W2904840750","doi":"10.48550/arxiv.1812.05272","title":"Towards a General-Purpose Linguistic Annotation Backend","year":2018,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"National Science Foundation","keywords":"Computer science; Documentation; Annotation; Natural language processing; Upload; Transcription (linguistics); Process (computing); Artificial intelligence; Natural language; Linguistics; World Wide Web; Programming language","score_opus":0.05514361144347381,"score_gpt":0.22260096463400986,"score_spread":0.16745735319053606,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2904840750","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0046424735,0.00023387809,0.8155714,0.0007513434,0.0006356296,0.00031230532,0.004083077,0.1659922,0.0077776574],"genre_scores_gemma":[0.07463098,0.00047162018,0.8246367,0.0024046013,0.0005780191,0.0009393781,0.039338253,0.024468765,0.032531686],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9970999,0.00042146884,0.00025841495,0.00085025566,0.0010631805,0.0003067551],"domain_scores_gemma":[0.98998404,0.0020828792,0.00026110475,0.0042250305,0.003014191,0.000432712],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0043414272,0.002491117,0.0019993398,0.0023860468,0.0017568977,0.009130679,0.005834224,0.0026219757,0.028259065],"category_scores_gemma":[0.013138288,0.0017159913,0.0019221735,0.002246746,0.0012829852,0.00850424,0.00712388,0.0064484687,0.039060947],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021215575,0.0006695766,0.002670332,0.001065664,0.0002311734,0.0013417176,0.0018292073,0.009345827,0.073665746,0.04270172,0.291232,0.57312554],"study_design_scores_gemma":[0.00024371374,0.0003186258,0.001921434,0.0004714331,0.0001913775,0.00083173614,0.0009049677,0.29062605,0.18179134,0.09833812,0.42407975,0.0002814828],"about_ca_topic_score_codex":0.0046095327,"about_ca_topic_score_gemma":0.0055461866,"teacher_disagreement_score":0.028259065,"about_ca_system_score_codex":0.0015879301,"about_ca_system_score_gemma":0.0020899572,"threshold_uncertainty_score":0.09453601},"labels":[],"label_agreement":null},{"id":"W2905137389","doi":"10.1609/aaai.v33i01.33017476","title":"Generating Character Descriptions for Automatic Summarization of Fiction","year":2019,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Automatic summarization; Character (mathematics); Computer science; Ranking (information retrieval); Information retrieval; Sample (material); Quality (philosophy); Natural language processing; Artificial intelligence; Mathematics","score_opus":0.0536322909089638,"score_gpt":0.29526777414468663,"score_spread":0.24163548323572284,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2905137389","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.36517203,0.0060622473,0.43075353,0.0020882823,0.0005520252,0.0029169235,0.13501027,0.042948812,0.014495828],"genre_scores_gemma":[0.32332453,0.0010544917,0.45040014,0.00015691253,0.00023705505,0.0009786256,0.21902388,0.0005636977,0.0042606685],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99869126,0.0004252737,0.00015653555,0.00032695517,0.00030243341,0.000097532844],"domain_scores_gemma":[0.9917105,0.0040690512,0.0011689644,0.0007932631,0.0019352153,0.00032307723],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011545983,0.0013335472,0.00062492065,0.007427599,0.0006229111,0.0013706974,0.0010585205,0.001006718,0.0043504094],"category_scores_gemma":[0.012551337,0.0003660568,0.0006518243,0.0033646468,0.00031160947,0.002730663,0.0009739005,0.0009220598,0.003609906],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00089065754,0.0004863247,0.02638099,0.0038094823,0.0002159846,0.0009159098,0.0026567625,0.013509382,0.036813263,0.0058909617,0.10351371,0.80491656],"study_design_scores_gemma":[0.0003118874,0.0012058945,0.057219956,0.000771889,0.000447416,0.0022842842,0.0057057566,0.56327313,0.09819743,0.017119767,0.25320104,0.0002616588],"about_ca_topic_score_codex":0.002397204,"about_ca_topic_score_gemma":0.0043581137,"teacher_disagreement_score":0.007427599,"about_ca_system_score_codex":0.00091258285,"about_ca_system_score_gemma":0.00082714896,"threshold_uncertainty_score":0.0145536065},"labels":[],"label_agreement":null},{"id":"W2905726793","doi":"10.22148/16.030","title":"Critical Search: A Procedure for Guided Reading in Large-Scale Textual Corpora","year":2018,"lang":"en","type":"article","venue":"Journal of Cultural Analytics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Reading (process); Computer science; Scale (ratio); Information retrieval; Natural language processing; Artificial intelligence; Linguistics; Geography; Philosophy; Cartography","score_opus":0.052274167746249496,"score_gpt":0.3675912135698836,"score_spread":0.3153170458236341,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2905726793","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0045067226,0.0003631263,0.91762483,0.0012764712,0.0004470203,0.005114101,0.012553626,0.043338496,0.014775572],"genre_scores_gemma":[0.015139824,0.00018500912,0.9639943,0.00027359062,0.00012937612,0.0045722276,0.006637223,0.0050001605,0.0040682023],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.980667,0.009337437,0.0034993608,0.003136174,0.002910244,0.00044986297],"domain_scores_gemma":[0.9090465,0.06456757,0.003455837,0.010299528,0.01144255,0.0011880518],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.026808972,0.003155011,0.0020895754,0.017969718,0.0053208596,0.0067435107,0.0035415206,0.002251031,0.068493485],"category_scores_gemma":[0.12207549,0.0023824195,0.0024213365,0.013433795,0.004461799,0.00823401,0.010115671,0.00359151,0.02723692],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007503851,0.0002468888,0.0019079207,0.005095237,0.00030746663,0.0012122493,0.02700516,0.0017679769,0.01591217,0.08144737,0.28513357,0.5792136],"study_design_scores_gemma":[0.00051890657,0.00020408827,0.0034410367,0.0018023038,0.00024220438,0.0012760669,0.016967969,0.03772385,0.023071729,0.2710851,0.6430191,0.0006477128],"about_ca_topic_score_codex":0.0048285485,"about_ca_topic_score_gemma":0.012357637,"teacher_disagreement_score":0.068493485,"about_ca_system_score_codex":0.002258291,"about_ca_system_score_gemma":0.0096213445,"threshold_uncertainty_score":0.22913355},"labels":[],"label_agreement":null},{"id":"W2907526436","doi":"10.4000/books.aaccademia.4512","title":"Overview of the EVALITA 2018 Solving language games (NLP4FUN) Task","year":2018,"lang":"en","type":"book-chapter","venue":"Accademia University Press eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"National Research Council Canada; Università degli Studi di Napoli Federico II","keywords":"Task (project management); Computer science; Set (abstract data type); Artificial intelligence; Range (aeronautics); Word (group theory); Natural language processing; World Wide Web; Multimedia; Human–computer interaction; Linguistics; Programming language; Engineering","score_opus":0.03451380042961945,"score_gpt":0.26568931946801544,"score_spread":0.231175519038396,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2907526436","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009166528,0.043570433,0.19603775,0.004427037,0.0031328434,0.0018868122,0.022104463,0.050760835,0.6689133],"genre_scores_gemma":[0.04232929,0.022455359,0.31198207,0.002491576,0.0017696865,0.0031673724,0.14006563,0.016928133,0.45881087],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99891114,0.0001859496,0.000078161356,0.0002578945,0.00044353108,0.00012333985],"domain_scores_gemma":[0.99930334,0.00020074745,0.00002359119,0.00011818995,0.00021085396,0.00014329393],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00095484365,0.0018926447,0.0010443989,0.0024989638,0.001194089,0.004229412,0.0024987736,0.0017537769,0.07725825],"category_scores_gemma":[0.0022349004,0.000721373,0.00080975244,0.0025859997,0.0005941842,0.004039604,0.0029861447,0.0022658836,0.072465196],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001536427,0.0003288789,0.00028001706,0.0011294675,0.000025146102,0.00011019983,0.00035943693,0.001535079,0.004165563,0.013185358,0.50285524,0.47587198],"study_design_scores_gemma":[0.000025921554,0.000063095766,0.0005359203,0.00015250774,0.000007087316,0.00021462045,0.000061013012,0.0023572322,0.0015026216,0.006586403,0.98846906,0.000024553694],"about_ca_topic_score_codex":0.004221042,"about_ca_topic_score_gemma":0.006768084,"teacher_disagreement_score":0.07725825,"about_ca_system_score_codex":0.0014637671,"about_ca_system_score_gemma":0.0017356375,"threshold_uncertainty_score":0.25845462},"labels":[],"label_agreement":null},{"id":"W2907532612","doi":"","title":"PSOA Prova: PSOA Translation of Pure Production Rules to the Prova Engine.","year":2018,"lang":"en","type":"article","venue":"Publikationsdatenbank der Fraunhofer-Gesellschaft (Fraunhofer-Gesellschaft)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of New Brunswick","funders":"","keywords":"Production (economics); Computer science","score_opus":0.01961148003526085,"score_gpt":0.2708945695527625,"score_spread":0.2512830895175016,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2907532612","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0020643545,0.00021767011,0.88019437,0.000710754,0.000434583,0.00048628368,0.007543689,0.08216939,0.026178937],"genre_scores_gemma":[0.0837725,0.00056587224,0.8334948,0.0017595267,0.00028663076,0.001122261,0.019835394,0.02796892,0.031194186],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9974523,0.0006464683,0.0003144796,0.0006091063,0.0008141017,0.00016346104],"domain_scores_gemma":[0.9944825,0.00274649,0.0003243581,0.0011530769,0.0011463218,0.00014728295],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026994492,0.0011297873,0.00057614205,0.001494172,0.00067448936,0.0041722567,0.0023703012,0.0012453265,0.032416973],"category_scores_gemma":[0.010428002,0.0016001619,0.0018978952,0.00076974684,0.0016390996,0.0037601548,0.0031100502,0.0033602368,0.019911194],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006337218,0.0003981272,0.0020705292,0.003527284,0.00019930286,0.0017266962,0.0025825906,0.010709139,0.027486907,0.37566337,0.22009893,0.35490346],"study_design_scores_gemma":[0.000114880124,0.00009066056,0.00057102577,0.00038657806,0.00007015753,0.0011431603,0.0002792221,0.046865672,0.032200117,0.09865691,0.8195024,0.00011917787],"about_ca_topic_score_codex":0.0028900565,"about_ca_topic_score_gemma":0.0035167183,"teacher_disagreement_score":0.032416973,"about_ca_system_score_codex":0.0011811486,"about_ca_system_score_gemma":0.0029536267,"threshold_uncertainty_score":0.108445585},"labels":[],"label_agreement":null},{"id":"W2907550307","doi":"","title":"Compact rule extraction for hierarchical phrase-based translation","year":2012,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia; Simon Fraser University","funders":"","keywords":"Computer science; Rule-based machine translation; Machine translation; Artificial intelligence; Natural language processing; Translation (biology); Phrase; Synchronous context-free grammar; Bayesian inference; Bayesian probability; Example-based machine translation","score_opus":0.037023097454397164,"score_gpt":0.33478243805467595,"score_spread":0.29775934060027875,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2907550307","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0039032556,0.00012245573,0.99246216,0.00007407784,0.000024191642,0.000070024085,0.0003649784,0.0020357727,0.0009430317],"genre_scores_gemma":[0.08668807,0.0002533065,0.90827394,0.00012271784,0.000041170417,0.00022515864,0.0025079257,0.00063816423,0.0012495628],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9980724,0.00056358497,0.00021971697,0.00041433636,0.00065953756,0.00007041373],"domain_scores_gemma":[0.9963617,0.0019474287,0.00025125706,0.00082747015,0.00054480927,0.00006744806],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015671493,0.0010160867,0.0011963727,0.0021239463,0.00074711395,0.0015876702,0.001284406,0.000863298,0.0048013534],"category_scores_gemma":[0.007594623,0.00084482227,0.0015308742,0.002141287,0.00094144203,0.0023455748,0.0022299923,0.0014607246,0.003263332],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015762121,0.00014780308,0.0018379495,0.00077975827,0.00019326029,0.00072717515,0.000616636,0.104771405,0.038522467,0.060820952,0.011436406,0.7799886],"study_design_scores_gemma":[0.000051280582,0.00006933559,0.0006418763,0.000091236885,0.00008884842,0.0004131684,0.00015127614,0.84709305,0.022803938,0.112140864,0.016395215,0.00005988644],"about_ca_topic_score_codex":0.0016373314,"about_ca_topic_score_gemma":0.004151506,"teacher_disagreement_score":0.0048013534,"about_ca_system_score_codex":0.0006449088,"about_ca_system_score_gemma":0.0017388496,"threshold_uncertainty_score":0.01606214},"labels":[],"label_agreement":null},{"id":"W2907719305","doi":"10.4000/books.aaccademia.4854","title":"Computer challenges guillotine: how an artificial player can solve a complex language TV game with web data analysis","year":2018,"lang":"en","type":"book-chapter","venue":"Accademia University Press eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"National Research Council Canada; Università degli Studi di Napoli Federico II","keywords":"Computer science; Simple (philosophy); Intersection (aeronautics); Matching (statistics); Artificial intelligence; World Wide Web; Engineering; Mathematics","score_opus":0.07876672343533694,"score_gpt":0.26868174237557385,"score_spread":0.1899150189402369,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2907719305","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.031455405,0.0052936305,0.7780737,0.012748602,0.0010388913,0.00039512062,0.0006249442,0.0051531605,0.16521658],"genre_scores_gemma":[0.15251397,0.0025869424,0.7029612,0.0030963563,0.00033600503,0.0004892909,0.0014851983,0.0018787822,0.13465233],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9984812,0.00065707404,0.000065174405,0.00033554953,0.00037668698,0.000084265645],"domain_scores_gemma":[0.99772984,0.0017093822,0.000057042904,0.00018915822,0.00018099729,0.00013349699],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019412874,0.0011529916,0.00040847727,0.0012126502,0.0010409802,0.0051638666,0.0021732654,0.0018488792,0.015236611],"category_scores_gemma":[0.0067282007,0.00037129322,0.0006501467,0.0006481663,0.0025114492,0.005571581,0.0031023296,0.0023898077,0.005764877],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004142705,0.00025295076,0.0030495543,0.00099609,0.00014141269,0.0011086591,0.0070851943,0.009498754,0.009219188,0.3327414,0.15509777,0.4803948],"study_design_scores_gemma":[0.000039851726,0.00012931971,0.0007416458,0.00027792124,0.000027198503,0.0018066918,0.0014867962,0.046925634,0.0048159882,0.10704017,0.8366425,0.000066309454],"about_ca_topic_score_codex":0.0011085116,"about_ca_topic_score_gemma":0.0025166126,"teacher_disagreement_score":0.015236611,"about_ca_system_score_codex":0.0012295593,"about_ca_system_score_gemma":0.0008045162,"threshold_uncertainty_score":0.050971568},"labels":[],"label_agreement":null},{"id":"W2907972816","doi":"10.4000/books.aaccademia.4482","title":"Overview of the Evalita 2018 – itaLIan Speech acT labEliNg (iLISTEN) Task","year":2018,"lang":"en","type":"book-chapter","venue":"Accademia University Press eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"National Research Council Canada; Università degli Studi di Napoli Federico II","keywords":"Task (project management); Speech act; Computer science; Statement (logic); Natural (archaeology); Annotation; Natural language processing; Natural language; Baseline (sea); Artificial intelligence; Speech recognition; Linguistics; Political science; Engineering","score_opus":0.04912885133115008,"score_gpt":0.27099033264748307,"score_spread":0.221861481316333,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2907972816","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01895329,0.024307262,0.33027917,0.0059261126,0.004475048,0.004144402,0.12068288,0.326698,0.16453384],"genre_scores_gemma":[0.04644526,0.0041138674,0.29515052,0.0022909408,0.0016018614,0.0048502544,0.5796869,0.015238846,0.050621636],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99563,0.0013574278,0.00033097455,0.0013155901,0.0010554745,0.0003104498],"domain_scores_gemma":[0.9974602,0.00064603204,0.0001344298,0.0006769022,0.0007612503,0.00032122922],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004065086,0.004531072,0.0020433324,0.004575801,0.0018338183,0.0043360307,0.00339582,0.0025564085,0.03932056],"category_scores_gemma":[0.0049679885,0.0010935346,0.0015771429,0.0027531348,0.0009038756,0.0043347296,0.005717961,0.0036036836,0.08369165],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000254808,0.00027165693,0.00074357266,0.0013093328,0.0000787375,0.0002060183,0.0004520878,0.002160123,0.009512821,0.0034383147,0.68491375,0.29665878],"study_design_scores_gemma":[0.00015404627,0.00023531898,0.0040519712,0.0004623711,0.00008331296,0.0010395779,0.00039387902,0.041724667,0.010900005,0.011669955,0.929108,0.00017678968],"about_ca_topic_score_codex":0.010257544,"about_ca_topic_score_gemma":0.013063689,"teacher_disagreement_score":0.03932056,"about_ca_system_score_codex":0.0017669634,"about_ca_system_score_gemma":0.0030346666,"threshold_uncertainty_score":0.13154036},"labels":[],"label_agreement":null},{"id":"W2908288256","doi":"","title":"Apprentissage de fonctions d'ordonnancement avec un flux de données non-étiquetées","year":2009,"lang":"fr","type":"preprint","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Political science; Physics; Philosophy","score_opus":0.01926213404433508,"score_gpt":0.27050537324356566,"score_spread":0.2512432391992306,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2908288256","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2470886,0.0011085105,0.7345086,0.0008726959,0.0003611442,0.00016901533,0.000740206,0.0077785808,0.0073726685],"genre_scores_gemma":[0.61909944,0.00064715254,0.36596602,0.00020562777,0.00017360869,0.00018717247,0.0010495315,0.0012745789,0.011396956],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9921532,0.0021133688,0.00051305053,0.0018090917,0.002838898,0.0005723397],"domain_scores_gemma":[0.97906166,0.014484282,0.000712562,0.00211193,0.0030577471,0.0005718941],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006019783,0.0027583635,0.0018385261,0.0037831133,0.0016958734,0.0068735853,0.001457855,0.0028256248,0.008990215],"category_scores_gemma":[0.02918814,0.00091195747,0.0034038709,0.0022012733,0.0019050365,0.0038570322,0.002016408,0.0026704269,0.0017523334],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0070925057,0.0008150748,0.02865124,0.001974411,0.00096383266,0.004220666,0.004070317,0.15045232,0.14753094,0.0964675,0.008247664,0.5495136],"study_design_scores_gemma":[0.00022903278,0.00067464553,0.019102104,0.0002612291,0.00029502806,0.00230028,0.0012024323,0.7910333,0.13387752,0.027805772,0.02301964,0.00019891693],"about_ca_topic_score_codex":0.014291348,"about_ca_topic_score_gemma":0.008370469,"teacher_disagreement_score":0.014291348,"about_ca_system_score_codex":0.0024101657,"about_ca_system_score_gemma":0.0015395969,"threshold_uncertainty_score":0.031836033},"labels":[],"label_agreement":null},{"id":"W2910606374","doi":"10.1145/3265752","title":"Low-Resource Machine Transliteration Using Recurrent Neural Networks","year":2019,"lang":"en","type":"article","venue":"ACM Transactions on Asian and Low-Resource Language Information Processing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Transliteration; Computer science; Grapheme; Pronunciation; Artificial intelligence; Machine translation; Word error rate; Natural language processing; Speech recognition; Artificial neural network; Vietnamese; Translation (biology); Recurrent neural network; Sequence (biology); Linguistics","score_opus":0.006320644909480506,"score_gpt":0.24000113453451388,"score_spread":0.23368048962503338,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2910606374","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03970216,0.0005108735,0.9510797,0.00013823312,0.000084577696,0.00005974155,0.00017182257,0.006079283,0.002173571],"genre_scores_gemma":[0.6053345,0.0004832689,0.38591716,0.00016397596,0.00007108618,0.00021057198,0.0013126218,0.0004806114,0.0060262172],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99947757,0.0001644809,0.000044009834,0.00015709987,0.000104791885,0.000052032614],"domain_scores_gemma":[0.99897254,0.00051313126,0.000107950815,0.00016130324,0.00022341557,0.000021739106],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006718987,0.00085177744,0.00066554593,0.00041986076,0.0002670654,0.0005302611,0.0009454533,0.000536689,0.0020943775],"category_scores_gemma":[0.0023413852,0.00028572517,0.00056329457,0.0005404521,0.00029509325,0.0011215423,0.0006336065,0.0009080397,0.0013668832],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00039714455,0.00017804216,0.0007125206,0.0003101395,0.00016247296,0.00040183272,0.00018009706,0.4202937,0.056901526,0.0056110076,0.004195207,0.5106563],"study_design_scores_gemma":[0.000011763191,0.000060729926,0.00013373133,0.0000074849586,0.000025021991,0.00005504137,0.000012285479,0.98150796,0.015253255,0.0018713013,0.0010520881,0.000009321801],"about_ca_topic_score_codex":0.004412966,"about_ca_topic_score_gemma":0.007189263,"teacher_disagreement_score":0.004412966,"about_ca_system_score_codex":0.00053215167,"about_ca_system_score_gemma":0.00068516383,"threshold_uncertainty_score":0.008774579},"labels":[],"label_agreement":null},{"id":"W2911149996","doi":"10.1109/wi.2018.00-54","title":"Hybrid Question Answering Using Heuristic Methods and Linked Data Schema","year":2018,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Question answering; SPARQL; RDF; Computer science; Schema (genetic algorithms); RDF Schema; Information retrieval; Linked data; Natural language; Heuristic; Semantic Web; Artificial intelligence; Natural language processing","score_opus":0.06727751609059875,"score_gpt":0.41864574676636657,"score_spread":0.35136823067576783,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2911149996","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.037045795,0.0010737525,0.9529479,0.000843434,0.000060558483,0.00036811602,0.0006880442,0.004112265,0.0028601564],"genre_scores_gemma":[0.23055956,0.00037272612,0.76339597,0.00046629042,0.000087927925,0.00048141374,0.0028633242,0.00032036868,0.0014523634],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98858637,0.00687702,0.00064535387,0.0015423703,0.002084123,0.00026473126],"domain_scores_gemma":[0.97944367,0.015621302,0.0006488291,0.0024016676,0.0016137316,0.000270769],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009174302,0.0010338929,0.0013258606,0.0053532124,0.00092577515,0.0034562112,0.0024988092,0.0018841102,0.0030842433],"category_scores_gemma":[0.021224225,0.0005569503,0.0017732903,0.0037363928,0.00149721,0.0049386644,0.0031420174,0.0014139102,0.00088507467],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006093895,0.0010870834,0.007786644,0.0013122674,0.000635557,0.00057992956,0.0017710896,0.15994418,0.014353191,0.087432265,0.012893569,0.71159476],"study_design_scores_gemma":[0.00012548355,0.00011763459,0.0011509148,0.00009606322,0.000097720345,0.00028733932,0.0004844534,0.8771902,0.009869556,0.09831783,0.012204133,0.000058687056],"about_ca_topic_score_codex":0.0044214814,"about_ca_topic_score_gemma":0.0042714537,"teacher_disagreement_score":0.009174302,"about_ca_system_score_codex":0.0015121736,"about_ca_system_score_gemma":0.0016332881,"threshold_uncertainty_score":0.048518956},"labels":[],"label_agreement":null},{"id":"W2911307508","doi":"","title":"Using Neural Transfer Learning for Morpho-syntactic Tagging of South-Slavic Languages Tweets","year":2018,"lang":"en","type":"article","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Computer science; Morpho; Slavic languages; Natural language processing; Artificial intelligence; Serbian; Domain (mathematical analysis); Task (project management); Artificial neural network; Annotation; Transfer of learning; Word (group theory); Character (mathematics); Linguistics","score_opus":0.020753866818919157,"score_gpt":0.27521177863937435,"score_spread":0.2544579118204552,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2911307508","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7092123,0.0010324212,0.25069094,0.0014243887,0.0011416213,0.00033236382,0.004626378,0.010613285,0.02092622],"genre_scores_gemma":[0.9392439,0.00022121468,0.04315981,0.00014805872,0.00020259223,0.00012140375,0.0073795384,0.00028781782,0.009235655],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994469,0.0001722192,0.000032393025,0.00016564448,0.00006238698,0.0001203097],"domain_scores_gemma":[0.9980221,0.0010123719,0.00010551712,0.00026473874,0.0005101431,0.00008518008],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010441139,0.0007905309,0.00047031126,0.0019114728,0.0009792136,0.0012192864,0.00073000236,0.0009579403,0.0038840484],"category_scores_gemma":[0.0034116388,0.00026390978,0.00058782863,0.0017191464,0.0003905752,0.0021003424,0.001478497,0.0014041001,0.004294055],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013639631,0.000851122,0.017413614,0.00041572054,0.00021687317,0.00071311026,0.0010673467,0.040956296,0.056084193,0.004721325,0.024694027,0.85150254],"study_design_scores_gemma":[0.000044719767,0.00016000749,0.008001256,0.00004458945,0.00009034141,0.00010110543,0.0009195418,0.9503614,0.022830855,0.011245975,0.0061542694,0.000045962028],"about_ca_topic_score_codex":0.008410321,"about_ca_topic_score_gemma":0.010770892,"teacher_disagreement_score":0.008410321,"about_ca_system_score_codex":0.0007260584,"about_ca_system_score_gemma":0.0009876188,"threshold_uncertainty_score":0.016722739},"labels":[],"label_agreement":null},{"id":"W2911380458","doi":"","title":"Proceedings of the 12th international conference on String Processing and Information Retrieval","year":2005,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Information retrieval; Political science","score_opus":0.013615486048400013,"score_gpt":0.26125655570359646,"score_spread":0.24764106965519644,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2911380458","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.031000068,0.088311635,0.6501102,0.017668605,0.0659283,0.0008284051,0.0042574657,0.011273011,0.13062243],"genre_scores_gemma":[0.08474308,0.05173619,0.34305212,0.004868216,0.015188339,0.0005249689,0.02275597,0.002836521,0.47429463],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9982357,0.00053155544,0.00018576255,0.00027683924,0.0005844991,0.00018560707],"domain_scores_gemma":[0.9966323,0.00090220646,0.00010551126,0.00062692136,0.0013563175,0.0003767635],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028331545,0.0010870689,0.0029085572,0.0023385186,0.0011759022,0.0061625466,0.0017442602,0.0015605816,0.057116836],"category_scores_gemma":[0.004918432,0.0004404307,0.0012273788,0.002624546,0.0010929707,0.0051024347,0.0017032071,0.002312732,0.033785023],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00065108994,0.00045356192,0.0010534895,0.00053712487,0.00017975685,0.00023440075,0.00026892062,0.0006244338,0.011597424,0.011845568,0.40302905,0.56952524],"study_design_scores_gemma":[0.00010750814,0.00027841513,0.002911135,0.00026705558,0.0003138394,0.0009633205,0.00033526233,0.019445667,0.013656649,0.02633563,0.93529683,0.00008859451],"about_ca_topic_score_codex":0.0032851228,"about_ca_topic_score_gemma":0.0045864712,"teacher_disagreement_score":0.057116836,"about_ca_system_score_codex":0.00087061134,"about_ca_system_score_gemma":0.0019637723,"threshold_uncertainty_score":0.19107485},"labels":[],"label_agreement":null},{"id":"W2911438565","doi":"","title":"Difficulté des listes thématiques d'un ouvrage bilingue selon la fréquence d’usage des mots","year":2018,"lang":"fr","type":"article","venue":"DOAJ (DOAJ: Directory of Open Access Journals)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Humanities; Philosophy","score_opus":0.1740448245446496,"score_gpt":0.5121763661077307,"score_spread":0.33813154156308106,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2911438565","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.97986263,0.00039661617,0.009206589,0.00016838977,0.000030094438,0.000114933195,0.0003936588,0.00021534547,0.0096115535],"genre_scores_gemma":[0.98382723,0.00024456446,0.008926234,0.00005374685,0.000014724334,0.00013856679,0.0009767645,0.00008316508,0.0057349526],"study_design_codex":"observational","study_design_gemma":"not_applicable","domain_scores_codex":[0.99444795,0.0017237557,0.000837367,0.000884053,0.0018023599,0.0003044159],"domain_scores_gemma":[0.9587852,0.030774089,0.0030381898,0.0012100723,0.0054414244,0.0007510712],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0040423083,0.00066409376,0.0005034128,0.0030936159,0.0011628524,0.0041500884,0.00073402986,0.0012703714,0.006915054],"category_scores_gemma":[0.038206264,0.000318264,0.0005400685,0.0019899823,0.0008754285,0.004141883,0.001962107,0.00081386126,0.0017083797],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012191614,0.00036708364,0.50029045,0.0021519826,0.00031722104,0.0020526492,0.10227405,0.0030743629,0.06680581,0.00907348,0.0053392686,0.30703443],"study_design_scores_gemma":[0.00008142657,0.0010322022,0.694598,0.0008760922,0.00045288543,0.005138295,0.13476302,0.031080494,0.044992596,0.0070416634,0.07964189,0.0003014469],"about_ca_topic_score_codex":0.0120551875,"about_ca_topic_score_gemma":0.007938328,"teacher_disagreement_score":0.0120551875,"about_ca_system_score_codex":0.0011815036,"about_ca_system_score_gemma":0.0011218976,"threshold_uncertainty_score":0.023970068},"labels":[],"label_agreement":null},{"id":"W2911903568","doi":"","title":"The subject of ROAR in the mind and in the corpus: What divergent results can teach us","year":2019,"lang":"en","type":"article","venue":"Monash University Research Portal (Monash University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Izaak Walton Killam Health Centre; Dalhousie University; University of Alberta","funders":"","keywords":"Convergence (economics); Divergence (linguistics); Linguistics; Subject (documents); Point (geometry); Reflection (computer programming); Computer science; Verb; Artificial intelligence; Epistemology; Natural language processing; Cognitive science; Psychology; Mathematics; Philosophy","score_opus":0.023343477177280955,"score_gpt":0.2624011187660794,"score_spread":0.23905764158879844,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2911903568","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.27507156,0.08924913,0.10447065,0.29826322,0.008961978,0.001136914,0.002668303,0.0008090861,0.2193692],"genre_scores_gemma":[0.8843482,0.016729984,0.055630244,0.026907777,0.0017814516,0.0014521797,0.0018523071,0.0016942004,0.009603647],"study_design_codex":"qualitative","study_design_gemma":"observational","domain_scores_codex":[0.8079468,0.14241345,0.008321123,0.012684708,0.025454981,0.0031788356],"domain_scores_gemma":[0.4582212,0.4136702,0.014644266,0.06266164,0.04650258,0.0043001864],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.19808769,0.0009530853,0.0025629655,0.013062533,0.0071663707,0.0238198,0.003993355,0.0043203146,0.007933561],"category_scores_gemma":[0.45513463,0.001237417,0.0014765582,0.010232116,0.03011032,0.04503784,0.013234493,0.0077011543,0.0037654194],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009815816,0.00035586927,0.05132897,0.0071762423,0.0012595268,0.0009443861,0.40797645,0.00043707178,0.0038074139,0.23039158,0.028757704,0.26658323],"study_design_scores_gemma":[0.00017389108,0.0003778408,0.074025325,0.017230392,0.0007837972,0.0012852317,0.38130456,0.0015228466,0.007083947,0.33180556,0.1839793,0.0004272897],"about_ca_topic_score_codex":0.006761103,"about_ca_topic_score_gemma":0.009904426,"teacher_disagreement_score":0.19808769,"about_ca_system_score_codex":0.006429625,"about_ca_system_score_gemma":0.0058798254,"threshold_uncertainty_score":0.98890066},"labels":[],"label_agreement":null},{"id":"W2912030431","doi":"","title":"Proceedings of the Thirteenth Conference on Computational Natural Language Learning","year":2009,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":44,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Session (web analytics); Computer science; Artificial intelligence; Perspective (graphical); Natural language; Natural language processing; Language acquisition; Natural (archaeology); Conjunction (astronomy); Focus (optics); Mathematics education; World Wide Web; Psychology; History","score_opus":0.00882063526391917,"score_gpt":0.2603617773115712,"score_spread":0.251541142047652,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2912030431","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010771428,0.10083696,0.5479127,0.06317411,0.09046663,0.0015196895,0.0149000855,0.0102904495,0.16012791],"genre_scores_gemma":[0.07340579,0.06532245,0.36834234,0.013911927,0.02208446,0.0021189381,0.073593065,0.0046871854,0.37653387],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9948473,0.0019860647,0.00048239285,0.0009255885,0.001485249,0.00027334344],"domain_scores_gemma":[0.98935425,0.004571188,0.00028847402,0.0020093354,0.0025728215,0.0012038624],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007232822,0.0013435455,0.0021000917,0.0020406255,0.0011940742,0.0069984854,0.0033442734,0.0021021732,0.08722753],"category_scores_gemma":[0.01668234,0.0005683857,0.001280859,0.00191155,0.0022910926,0.007622231,0.0041999537,0.00629164,0.031186866],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002250448,0.00016815693,0.00072717544,0.0006833234,0.0001823816,0.00015384397,0.0002337659,0.001429002,0.0017726925,0.021740904,0.6842051,0.2884786],"study_design_scores_gemma":[0.000038004175,0.00006665299,0.000764314,0.00027471202,0.000052397514,0.00020243287,0.00015377767,0.007837159,0.0011922315,0.038729444,0.9506443,0.000044618897],"about_ca_topic_score_codex":0.0022814027,"about_ca_topic_score_gemma":0.0042054118,"teacher_disagreement_score":0.08722753,"about_ca_system_score_codex":0.0016215923,"about_ca_system_score_gemma":0.0035911223,"threshold_uncertainty_score":0.29180515},"labels":[],"label_agreement":null},{"id":"W2912505270","doi":"","title":"Proceedings of the NAACL HLT 2010 Workshop on Computational Linguistics in a World of Social Media","year":2010,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Marketing buzz; Social media; World Wide Web; Social media optimization; Resource (disambiguation); Computer science; Quarter (Canadian coin); Multitude; Media studies; Microblogging; Sociology; Political science; History","score_opus":0.014323600361222393,"score_gpt":0.2772745184614393,"score_spread":0.2629509181002169,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2912505270","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016521383,0.06546683,0.4191093,0.25565046,0.08848562,0.0018303555,0.01368147,0.013461381,0.12579314],"genre_scores_gemma":[0.09085444,0.049487077,0.39317593,0.043548808,0.04815964,0.0037349516,0.06945619,0.009903561,0.2916794],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.98804283,0.007594998,0.00083499576,0.0014994977,0.0015630691,0.00046463832],"domain_scores_gemma":[0.9773358,0.01414828,0.00037437567,0.00246111,0.0033237587,0.002356736],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.019211823,0.0019855108,0.0031266243,0.0027447504,0.0030701538,0.013856049,0.0045391642,0.004027205,0.05358961],"category_scores_gemma":[0.026984414,0.0015828226,0.0021106668,0.0022947842,0.004244387,0.016815893,0.009201476,0.009213949,0.022774957],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020698084,0.0003695291,0.00063150673,0.0005794609,0.00013376134,0.00029230822,0.0016938319,0.0006157326,0.0014692511,0.014456635,0.8680061,0.11154493],"study_design_scores_gemma":[0.00016429154,0.00006509601,0.0013507729,0.00070895953,0.00011924627,0.0005542001,0.0015406146,0.006036241,0.0012178215,0.058556937,0.92957896,0.00010686211],"about_ca_topic_score_codex":0.008790427,"about_ca_topic_score_gemma":0.023050372,"teacher_disagreement_score":0.05358961,"about_ca_system_score_codex":0.004912207,"about_ca_system_score_gemma":0.007687324,"threshold_uncertainty_score":0.1792751},"labels":[],"label_agreement":null},{"id":"W2912628563","doi":"10.3115/1067807.1067845","title":"Topological parsing","year":2003,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Phrase structure grammar; L-attributed grammar; Parsing; S-attributed grammar; Natural language processing; Rule-based machine translation; Artificial intelligence; Tree-adjoining grammar; Slavic languages; Parsing expression grammar; Top-down parsing; Word order; Programming language; Linguistics; Context-free grammar","score_opus":0.01876191324332163,"score_gpt":0.2819508270667275,"score_spread":0.2631889138234059,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2912628563","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0016194014,0.00018932666,0.98038816,0.00034048167,0.00016239776,0.00007436097,0.00085922796,0.0039685136,0.01239813],"genre_scores_gemma":[0.08065066,0.0006871528,0.8944143,0.0004500544,0.00027098416,0.00028109908,0.005154381,0.0028600583,0.015231323],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.998432,0.0004121889,0.00016161388,0.00044150706,0.00039382197,0.00015885964],"domain_scores_gemma":[0.9976755,0.0007607584,0.00012557393,0.00091726054,0.00042014153,0.000100725934],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016877708,0.001117122,0.0008912444,0.002720028,0.0016426999,0.0039075767,0.0022763805,0.0015104732,0.027793918],"category_scores_gemma":[0.004503439,0.0007465293,0.0023151846,0.00254811,0.0022007255,0.007443383,0.0035042376,0.0018603376,0.008099283],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000030294075,0.00003073835,0.00022565814,0.00024295901,0.000031947766,0.00019918839,0.000309879,0.007397624,0.0023808046,0.9003626,0.013483566,0.075304724],"study_design_scores_gemma":[0.00002018754,0.000028786562,0.00017110973,0.00007788303,0.000053609037,0.00029864386,0.000099771256,0.043425675,0.0038591868,0.8020977,0.14981395,0.000053514745],"about_ca_topic_score_codex":0.0019645006,"about_ca_topic_score_gemma":0.0032158683,"teacher_disagreement_score":0.027793918,"about_ca_system_score_codex":0.0013292808,"about_ca_system_score_gemma":0.0017444554,"threshold_uncertainty_score":0.09297991},"labels":[],"label_agreement":null},{"id":"W2912796987","doi":"","title":"Proceedings of the NAACL-HLT 2012 Workshop on Computational Linguistics for Literature","year":2012,"lang":"en","type":"article","venue":"North American Chapter of the Association for Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Computational linguistics; Linguistics; Natural language processing; Philosophy","score_opus":0.013664324693595478,"score_gpt":0.27642005078663484,"score_spread":0.2627557260930394,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2912796987","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.028322754,0.05597597,0.4548256,0.17703946,0.06921454,0.0020111161,0.029367816,0.01572698,0.1675158],"genre_scores_gemma":[0.12378785,0.03218349,0.35604775,0.018458208,0.01970271,0.002128405,0.1560612,0.010332932,0.28129748],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9917237,0.00498149,0.0007596897,0.0010937778,0.0011147393,0.00032658083],"domain_scores_gemma":[0.97931033,0.009439401,0.0004965748,0.0034487452,0.0048262416,0.0024786438],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013547912,0.0014548351,0.002865105,0.005133531,0.004271756,0.017126212,0.0030383782,0.0026361598,0.043578554],"category_scores_gemma":[0.02383636,0.0014027153,0.0017472593,0.0036664596,0.0027719594,0.018560365,0.01110659,0.007866414,0.020914562],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028342305,0.00037564093,0.00085798785,0.00080442545,0.000091377246,0.0002551888,0.0027970304,0.00045963295,0.0028612255,0.03666759,0.8084598,0.14608663],"study_design_scores_gemma":[0.000077881625,0.000028841745,0.0008468859,0.0005638821,0.0000752777,0.0002777825,0.0015952263,0.003305534,0.0018745528,0.044339344,0.9469571,0.000057826834],"about_ca_topic_score_codex":0.006851045,"about_ca_topic_score_gemma":0.019111255,"teacher_disagreement_score":0.043578554,"about_ca_system_score_codex":0.003540784,"about_ca_system_score_gemma":0.007946497,"threshold_uncertainty_score":0.1457848},"labels":[],"label_agreement":null},{"id":"W2912854696","doi":"","title":"Proceedings of the 2007 international conference on Advances in rule interchange and applications","year":2007,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of New Brunswick","funders":"","keywords":"Computer science; Data science; Engineering ethics; Engineering","score_opus":0.01657633828653678,"score_gpt":0.30551294473709756,"score_spread":0.2889366064505608,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2912854696","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01570854,0.024106685,0.86706233,0.008760914,0.01802611,0.00068090996,0.0028349706,0.0073510683,0.055468418],"genre_scores_gemma":[0.05692228,0.021280555,0.7775529,0.004051118,0.0035821414,0.0005559938,0.022358844,0.0028987108,0.1107974],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99549663,0.0016481781,0.0006138799,0.000697622,0.0012772284,0.00026640866],"domain_scores_gemma":[0.98800045,0.004227872,0.0002973949,0.003299809,0.0035094258,0.00066500757],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010366649,0.0012058337,0.003272326,0.0022614421,0.0014895877,0.009936334,0.0040822364,0.002050362,0.042044215],"category_scores_gemma":[0.015745288,0.0008618359,0.0023242226,0.0030641959,0.0016949638,0.00873537,0.003527764,0.00416719,0.021113167],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004754237,0.0007897122,0.0011589527,0.0005768433,0.0002928471,0.00034659033,0.0005,0.0022412485,0.0056782486,0.051923223,0.1858534,0.7501636],"study_design_scores_gemma":[0.000114184375,0.00019612718,0.0015069334,0.00045781984,0.0003649223,0.0011781588,0.00047060667,0.051071577,0.011875397,0.10422962,0.82842356,0.00011109851],"about_ca_topic_score_codex":0.0049534333,"about_ca_topic_score_gemma":0.0055994377,"teacher_disagreement_score":0.042044215,"about_ca_system_score_codex":0.0012239738,"about_ca_system_score_gemma":0.0037454348,"threshold_uncertainty_score":0.14065194},"labels":[],"label_agreement":null},{"id":"W2912867848","doi":"10.7202/1055437ar","title":"Documenting Linguistic Knowledge in an Inuit Language Atlas","year":2019,"lang":"en","type":"article","venue":"Études/Inuit/Studies","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":true,"ca_institutions":"Inuit Tapiriit Kanatami; Carleton University","funders":"","keywords":"Documentation; Atlas (anatomy); Computer science; Linguistics; Vitality; Natural language processing; World Wide Web; Medicine; Programming language","score_opus":0.025840432921686637,"score_gpt":0.3535766539383909,"score_spread":0.32773622101670424,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2912867848","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.76830745,0.0006924543,0.12984821,0.0015364828,0.00015294923,0.00047332415,0.0078701,0.00531954,0.08579953],"genre_scores_gemma":[0.8406088,0.000587779,0.13648209,0.00011056948,0.000028449367,0.00015306118,0.0037687102,0.00032026126,0.01794043],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9996742,0.000094919036,0.000029882445,0.00009236114,0.000077770834,0.000030790176],"domain_scores_gemma":[0.9978346,0.0008253851,0.00016869292,0.0003980595,0.00062868086,0.0001446012],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000655862,0.00021221917,0.00018513534,0.0025512315,0.0015683935,0.0028586988,0.00042990496,0.00026162073,0.0042908555],"category_scores_gemma":[0.0027770188,0.00020958394,0.00011246433,0.0025619576,0.0009262206,0.0029358952,0.0017931055,0.0003817813,0.0008105148],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003117139,0.00014482156,0.03156451,0.0006134386,0.000040186165,0.0024445865,0.16838726,0.002862261,0.06799899,0.04622758,0.0235056,0.655899],"study_design_scores_gemma":[0.000033233933,0.00022010168,0.0664171,0.00030735688,0.000106590036,0.003281115,0.09497201,0.03184015,0.063778006,0.014680962,0.7241991,0.0001642295],"about_ca_topic_score_codex":0.0859774,"about_ca_topic_score_gemma":0.21377148,"teacher_disagreement_score":0.9140226,"about_ca_system_score_codex":0.0018354197,"about_ca_system_score_gemma":0.0027752907,"threshold_uncertainty_score":0.17095393},"labels":[],"label_agreement":null},{"id":"W2913240921","doi":"","title":"Proceedings of the 1st North American chapter of the Association for Computational Linguistics conference","year":2000,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":29,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Presentation (obstetrics); Library science; Operations research; Media studies; Engineering; Sociology; Computer science; Medicine","score_opus":0.010956207484767513,"score_gpt":0.2466134796207427,"score_spread":0.2356572721359752,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2913240921","genre_codex":"editorial","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009805082,0.10890374,0.053292744,0.21319212,0.30243477,0.0017147383,0.013412207,0.0073418645,0.28990275],"genre_scores_gemma":[0.026127448,0.053346697,0.04304193,0.04027029,0.03972957,0.0025039874,0.0351677,0.0048803645,0.754932],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9936173,0.0023825397,0.000499177,0.0013092614,0.0017908721,0.00040089962],"domain_scores_gemma":[0.9857368,0.0042954376,0.00048075293,0.0013679237,0.006112502,0.0020066928],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011109025,0.0011868841,0.0021389287,0.002473077,0.002899454,0.010360137,0.0027436414,0.0037256766,0.12227741],"category_scores_gemma":[0.018081568,0.0008064501,0.0010134648,0.0019390159,0.0020714132,0.008096701,0.004369919,0.0060861404,0.0750887],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008331988,0.0000525741,0.00033194423,0.00024482628,0.000027166183,0.00007173189,0.00026008787,0.00004319838,0.00042449232,0.0018432863,0.9466256,0.049991634],"study_design_scores_gemma":[0.000020211484,0.000019424851,0.0007801204,0.00029481147,0.000021770742,0.00008394069,0.0002756799,0.00025730152,0.00018662278,0.0021883952,0.9958467,0.000025149233],"about_ca_topic_score_codex":0.0069745043,"about_ca_topic_score_gemma":0.010780575,"teacher_disagreement_score":0.12227741,"about_ca_system_score_codex":0.0032808774,"about_ca_system_score_gemma":0.0075614443,"threshold_uncertainty_score":0.4090587},"labels":[],"label_agreement":null},{"id":"W2913626674","doi":"10.18653/v1/w18-6002","title":"Using Universal Dependencies in cross-linguistic complexity research","year":2018,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Linguistics; Natural language processing; Artificial intelligence; Programming language; Philosophy","score_opus":0.22929479630989005,"score_gpt":0.4611939874606157,"score_spread":0.23189919115072563,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2913626674","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6194616,0.0018495971,0.36409122,0.00045305258,0.00014356093,0.00023210695,0.0016643802,0.0010061614,0.011098381],"genre_scores_gemma":[0.9040333,0.00042521703,0.09170195,0.00010142647,0.000082770974,0.00030405272,0.002406034,0.00039634324,0.0005489177],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.98246163,0.008299313,0.0017478736,0.0028823102,0.004023832,0.0005850552],"domain_scores_gemma":[0.8425798,0.10961308,0.010981547,0.022817018,0.012389975,0.0016186016],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.017098103,0.0013317905,0.0010260048,0.013646524,0.0019195664,0.0045980476,0.0017117718,0.0013302389,0.0024807055],"category_scores_gemma":[0.12583123,0.000555678,0.0011251211,0.010133248,0.0026218088,0.009257852,0.0067784935,0.0016425428,0.00048577893],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013157442,0.0009272809,0.24315941,0.002299673,0.0026134476,0.0010218375,0.01044691,0.14061227,0.040771928,0.11722288,0.0051716515,0.43443695],"study_design_scores_gemma":[0.00012253689,0.00096705393,0.22745101,0.0006571559,0.0011541764,0.0012909481,0.0038995456,0.44668627,0.05199636,0.24305502,0.02214805,0.0005719579],"about_ca_topic_score_codex":0.00362445,"about_ca_topic_score_gemma":0.0045024706,"teacher_disagreement_score":0.017098103,"about_ca_system_score_codex":0.0020295358,"about_ca_system_score_gemma":0.0012252097,"threshold_uncertainty_score":0.09042448},"labels":[],"label_agreement":null},{"id":"W2914180170","doi":"10.1093/llc/fqy074","title":"Toward Kurdish language processing: Experiments in collecting and processing the AsoSoft text corpus","year":2018,"lang":"en","type":"article","venue":"Digital Scholarship in the Humanities","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Perplexity; Computer science; Natural language processing; Text corpus; Artificial intelligence; n-gram; Language model; Text processing; Annotation; Zipf's law; Linguistics","score_opus":0.06108343550245978,"score_gpt":0.3139606671856046,"score_spread":0.25287723168314485,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2914180170","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.91502154,0.0015702465,0.04081346,0.0015706008,0.0006203067,0.001617123,0.010955425,0.011877567,0.015953714],"genre_scores_gemma":[0.7203261,0.00084881904,0.19887444,0.0011063688,0.0002202468,0.0020258967,0.066419,0.0014123387,0.008766886],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9952123,0.00221894,0.00044548747,0.0011527714,0.0006797513,0.00029071476],"domain_scores_gemma":[0.987661,0.008897391,0.00021378057,0.0016100594,0.0011972137,0.00042058466],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004656058,0.0020167122,0.0013471729,0.0017088734,0.0021156832,0.0017100851,0.0019513161,0.0017780098,0.0049083955],"category_scores_gemma":[0.016688967,0.0004666741,0.0009493587,0.003256578,0.0014862325,0.003629856,0.0023815618,0.0020977287,0.0040684277],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.005892112,0.0072858143,0.012006756,0.0047748825,0.0005566829,0.004308332,0.0090795895,0.04896782,0.058016725,0.006137922,0.110642634,0.7323307],"study_design_scores_gemma":[0.0027330774,0.0038402246,0.0444665,0.0005304298,0.00062369864,0.003649428,0.021062499,0.60067105,0.16460513,0.01526966,0.14191167,0.0006366257],"about_ca_topic_score_codex":0.012760317,"about_ca_topic_score_gemma":0.010165108,"teacher_disagreement_score":0.012760317,"about_ca_system_score_codex":0.0010282476,"about_ca_system_score_gemma":0.0015650645,"threshold_uncertainty_score":0.025372088},"labels":[],"label_agreement":null},{"id":"W2914220664","doi":"","title":"Deep Models for Arabic Dialect Identification on Benchmarked Data","year":2018,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":61,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Benchmark (surveying); Computer science; Artificial intelligence; Task (project management); Deep learning; Arabic; Natural language processing; Binary classification; Deep neural networks; Identification (biology); Machine learning; Artificial neural network; Recurrent neural network; Test data; Binary number; Speech recognition; Support vector machine; Linguistics; Mathematics; Geography","score_opus":0.10115388688345246,"score_gpt":0.3765717506027052,"score_spread":0.27541786371925275,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2914220664","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7793805,0.011807667,0.0618514,0.0033831275,0.0021883887,0.0006476797,0.09620999,0.024425967,0.020105287],"genre_scores_gemma":[0.67011416,0.0013390309,0.08052433,0.0009237159,0.00031667895,0.00049372576,0.23421347,0.0006441923,0.011430647],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9975515,0.000904968,0.00020474827,0.0006413599,0.00044139335,0.00025599077],"domain_scores_gemma":[0.99372697,0.0025973618,0.00028610558,0.001557822,0.0015385837,0.00029313684],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0038226657,0.0031538743,0.0011546368,0.0028566427,0.0011630057,0.0017738052,0.0028360605,0.0022790865,0.006617148],"category_scores_gemma":[0.013735471,0.00051750697,0.0013434809,0.0026688003,0.0010683541,0.0030461634,0.00257073,0.003556804,0.005859227],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0027233937,0.0025742683,0.019387111,0.0017075799,0.00081503636,0.0008286241,0.0006264319,0.23742059,0.011102963,0.0036541803,0.23026352,0.48889625],"study_design_scores_gemma":[0.00040961913,0.00054991373,0.011789776,0.00023360949,0.00016039368,0.00032465387,0.0008291264,0.93164015,0.016729506,0.008715757,0.028488608,0.00012889814],"about_ca_topic_score_codex":0.031149788,"about_ca_topic_score_gemma":0.040514827,"teacher_disagreement_score":0.031149788,"about_ca_system_score_codex":0.0022522502,"about_ca_system_score_gemma":0.0014017673,"threshold_uncertainty_score":0.061936915},"labels":[],"label_agreement":null},{"id":"W2914344421","doi":"10.1002/pra2.2018.14505501004","title":"A word‐level language identification strategy for resource‐scarce languages","year":2018,"lang":"en","type":"article","venue":"Proceedings of the Association for Information Science and Technology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Computer science; Yoruba; Trigram; Natural language processing; Artificial intelligence; Igbo; Language identification; Hausa; Word (group theory); Identification (biology); Character (mathematics); Natural language; Linguistics; Mathematics","score_opus":0.01446866564535378,"score_gpt":0.29309677827317837,"score_spread":0.2786281126278246,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2914344421","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.22657874,0.00025471186,0.76150596,0.0004863007,0.00004374553,0.00038033412,0.00048434638,0.0052432953,0.0050225942],"genre_scores_gemma":[0.6094644,0.00012192128,0.38605747,0.00016569137,0.000027434473,0.00025207206,0.00075413764,0.00032750718,0.002829365],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99895537,0.0003702045,0.00009295848,0.0003401882,0.00014322124,0.00009791778],"domain_scores_gemma":[0.9968484,0.0014595212,0.00033584356,0.0003617895,0.00088327273,0.00011112895],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012630407,0.001076832,0.00088763115,0.0039896755,0.0011705344,0.0016916519,0.0009992531,0.0007184942,0.0040150178],"category_scores_gemma":[0.0051232255,0.00049379206,0.0005601563,0.0015413117,0.0006701764,0.0026992976,0.001867509,0.00087834324,0.0032782697],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005619392,0.00046233583,0.016912118,0.0005161537,0.000103395774,0.0010271925,0.0033043274,0.009484167,0.18850514,0.008589737,0.003705776,0.7668277],"study_design_scores_gemma":[0.00013018468,0.0007366123,0.022233969,0.00016396478,0.00037479587,0.0024965715,0.0087622665,0.6373588,0.27708507,0.026043521,0.024393413,0.00022080756],"about_ca_topic_score_codex":0.0025184362,"about_ca_topic_score_gemma":0.0036160734,"teacher_disagreement_score":0.0040150178,"about_ca_system_score_codex":0.0005200365,"about_ca_system_score_gemma":0.0014613763,"threshold_uncertainty_score":0.013431609},"labels":[],"label_agreement":null},{"id":"W2914442349","doi":"","title":"Multiple-Attribute Text Rewriting.","year":2018,"lang":"en","type":"article","venue":"International Conference on Learning Representations","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":192,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Rewriting; Computer science; Programming language; Natural language processing; Information retrieval","score_opus":0.046905022059950514,"score_gpt":0.360655210535885,"score_spread":0.3137501884759345,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2914442349","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01230159,0.0014550546,0.94049037,0.0018626414,0.0010479128,0.00031747358,0.0036345369,0.014416498,0.024473904],"genre_scores_gemma":[0.27073085,0.0009528874,0.6695504,0.001020769,0.00048271063,0.0002162384,0.011110617,0.0036845994,0.042250857],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99661714,0.0011086835,0.00039061913,0.000799555,0.0008811556,0.00020287908],"domain_scores_gemma":[0.9933149,0.0024662728,0.00026177798,0.002632218,0.0011418724,0.00018289326],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001967766,0.0007864202,0.0009265008,0.0012169651,0.0008655207,0.0021218765,0.0018434674,0.0009564161,0.013955842],"category_scores_gemma":[0.010768398,0.00041975814,0.0013493006,0.0014039215,0.0008746064,0.0037467303,0.0028459572,0.0023750125,0.0076757655],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024638543,0.00039257618,0.0012685962,0.0011795305,0.00027068198,0.00067144993,0.00092506374,0.012806173,0.027623769,0.24766038,0.1201694,0.58678603],"study_design_scores_gemma":[0.00008240078,0.0001402808,0.00078140706,0.00018205159,0.00030166935,0.0010796712,0.00046399204,0.15612927,0.070242554,0.4956906,0.27481192,0.000094109855],"about_ca_topic_score_codex":0.0013343355,"about_ca_topic_score_gemma":0.0028274823,"teacher_disagreement_score":0.013955842,"about_ca_system_score_codex":0.00082212745,"about_ca_system_score_gemma":0.00116045,"threshold_uncertainty_score":0.046686888},"labels":[],"label_agreement":null},{"id":"W2914704304","doi":"","title":"Proceedings of the International Workshop on Cross-Language Knowledge Induction","year":2006,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Natural language processing; Linguistics; Artificial intelligence; Task (project management); Verb; Word (group theory)","score_opus":0.011917923526544562,"score_gpt":0.29584394512396794,"score_spread":0.28392602159742336,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2914704304","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022580918,0.02729407,0.8409997,0.01552567,0.012935851,0.0011159475,0.0036803426,0.0108899465,0.0649775],"genre_scores_gemma":[0.13783373,0.014131627,0.6604108,0.005978996,0.0042327396,0.0015570323,0.03595318,0.0044247946,0.13547711],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99527127,0.0020126218,0.0004582545,0.0011091187,0.0007742576,0.00037455792],"domain_scores_gemma":[0.9898928,0.0041931663,0.00020224953,0.0027488614,0.00233619,0.00062677974],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01028006,0.001963595,0.0028135933,0.0026677346,0.0019241084,0.0066023604,0.0049370737,0.002557065,0.051176555],"category_scores_gemma":[0.0146142,0.0011928066,0.0024789716,0.002552086,0.0020834752,0.010935909,0.0063855033,0.003962624,0.01627614],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00083082606,0.00086162676,0.0014552763,0.00058500585,0.00037296664,0.0005877314,0.0009816738,0.003345326,0.0055254293,0.019223439,0.25768894,0.70854175],"study_design_scores_gemma":[0.0002791638,0.00030971688,0.0054641957,0.00085793796,0.0005367871,0.0012689348,0.0011833697,0.07012126,0.012717085,0.13163288,0.77546585,0.00016280297],"about_ca_topic_score_codex":0.006187349,"about_ca_topic_score_gemma":0.0065917606,"teacher_disagreement_score":0.051176555,"about_ca_system_score_codex":0.0017638828,"about_ca_system_score_gemma":0.0026298012,"threshold_uncertainty_score":0.1712026},"labels":[],"label_agreement":null},{"id":"W2914910200","doi":"","title":"Fusion probabiliste appliquée à la détection et classification d'opinions","year":2009,"lang":"fr","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science","score_opus":0.02458517896365925,"score_gpt":0.27485084289948053,"score_spread":0.2502656639358213,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2914910200","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.046890203,0.00086029054,0.94678426,0.0010449521,0.00018011907,0.00009617175,0.00041391648,0.0018456326,0.0018845631],"genre_scores_gemma":[0.6534082,0.0009234021,0.33302838,0.00043900285,0.00066070905,0.00033422204,0.0014818056,0.00031843732,0.009405717],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99458367,0.0016747231,0.000397814,0.0011786312,0.0017610485,0.000404214],"domain_scores_gemma":[0.9877438,0.009140404,0.00039226303,0.0006722987,0.0018586272,0.00019265337],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0065960083,0.0011522636,0.0020728882,0.0020475744,0.001020515,0.0043455763,0.0013914522,0.0031269332,0.0046385117],"category_scores_gemma":[0.015412181,0.0008405633,0.002326059,0.0015990728,0.0010555736,0.0029474108,0.0022304293,0.0033989423,0.0020342767],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018684838,0.00047651626,0.005888508,0.00030597267,0.00054026395,0.00032478283,0.00040741803,0.19698687,0.033470474,0.016559746,0.00839619,0.7347748],"study_design_scores_gemma":[0.000024258557,0.00005549472,0.001298475,0.000011020019,0.000042471398,0.00009167263,0.000026814847,0.98637635,0.0061142896,0.0049958993,0.0009470299,0.00001628172],"about_ca_topic_score_codex":0.0100214295,"about_ca_topic_score_gemma":0.008979838,"teacher_disagreement_score":0.0100214295,"about_ca_system_score_codex":0.0019353393,"about_ca_system_score_gemma":0.0018294974,"threshold_uncertainty_score":0.0348835},"labels":[],"label_agreement":null},{"id":"W2915030876","doi":"10.18653/v1/2021.eacl-srw","title":"Proceedings of the 16th Conference of the European Chapter of the Association for Computational Linguistics: Student Research Workshop","year":2021,"lang":"en","type":"paratext","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Atomic Energy of Canada Limited","keywords":"Computational linguistics; Association (psychology); Computer science; Library science; Applied linguistics; Linguistics; Mathematics education; Natural language processing; Philosophy; Epistemology; Psychology","score_opus":0.09510324792952576,"score_gpt":0.3824149229462771,"score_spread":0.28731167501675137,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2915030876","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025996063,0.035831712,0.12598623,0.23930241,0.27096725,0.0023061044,0.017429931,0.009840544,0.27233973],"genre_scores_gemma":[0.027889123,0.007936443,0.026700461,0.008334186,0.02015278,0.0011748509,0.018260209,0.004572536,0.8849793],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9932661,0.0031839313,0.00044527825,0.00095988414,0.0015694108,0.0005753395],"domain_scores_gemma":[0.98276824,0.003270239,0.00038777612,0.0015444981,0.005795181,0.006234064],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012895502,0.0010769151,0.0016284984,0.0020172563,0.0022389095,0.009158867,0.0026396662,0.0033922666,0.15702078],"category_scores_gemma":[0.012866339,0.0005682427,0.0011110869,0.0021544378,0.0015567836,0.007343323,0.0060950257,0.0038548508,0.11621549],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013914988,0.00010993352,0.00025470933,0.00023408353,0.000016732281,0.000112194924,0.0004008061,0.00015663599,0.0008843805,0.0025740706,0.9435048,0.051612414],"study_design_scores_gemma":[0.000025619885,0.000048924187,0.0006008213,0.00011176558,0.0000103162565,0.000100536134,0.00033454184,0.00037569256,0.00038819786,0.0021488543,0.9958364,0.000018372464],"about_ca_topic_score_codex":0.0020455897,"about_ca_topic_score_gemma":0.004296266,"teacher_disagreement_score":0.15702078,"about_ca_system_score_codex":0.0023681058,"about_ca_system_score_gemma":0.0063621867,"threshold_uncertainty_score":0.5252868},"labels":[],"label_agreement":null},{"id":"W2915114476","doi":"","title":"RPI BLENDER TAC-KBP2016 System Description.","year":2016,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science","score_opus":0.0084808463073163,"score_gpt":0.2374701232353894,"score_spread":0.2289892769280731,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2915114476","genre_codex":"software","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006349008,0.0011197131,0.20922491,0.00095110404,0.00043403282,0.0013319029,0.2431684,0.4583951,0.07902582],"genre_scores_gemma":[0.04659793,0.00093222345,0.13860278,0.001194876,0.00014310378,0.002675366,0.71296304,0.046924558,0.04996604],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99845266,0.00022729691,0.0002139019,0.0002793737,0.0006791607,0.00014760843],"domain_scores_gemma":[0.9982157,0.00031354235,0.00009648416,0.0005365484,0.0007300604,0.00010770168],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016297167,0.0018416484,0.0009639169,0.002204984,0.00055144355,0.0026301562,0.003578886,0.0014931457,0.07127291],"category_scores_gemma":[0.005426274,0.0009347621,0.00088499545,0.0015374742,0.00035238385,0.0035468016,0.0016902761,0.002068986,0.07092577],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010134785,0.00019248111,0.0012287481,0.002520946,0.00011920417,0.00045725817,0.00028255387,0.008375384,0.012762619,0.011642694,0.8410868,0.12031789],"study_design_scores_gemma":[0.00026671405,0.0001430523,0.0015753345,0.00028680105,0.00009656123,0.00075572834,0.000079837904,0.039586723,0.021753391,0.008429886,0.9269075,0.00011851549],"about_ca_topic_score_codex":0.008332448,"about_ca_topic_score_gemma":0.006624571,"teacher_disagreement_score":0.07127291,"about_ca_system_score_codex":0.001159899,"about_ca_system_score_gemma":0.0018309349,"threshold_uncertainty_score":0.23843163},"labels":[],"label_agreement":null},{"id":"W2915258134","doi":"","title":"Semeval-2012 Task 8: Cross-lingual Textual Entailment for Content Synchronization","year":2012,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":28,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Textual entailment; Task (project management); Natural language processing; SemEval; Logical consequence; Inference; Synchronization (alternating current); Artificial intelligence; Process (computing); Programming language","score_opus":0.03341947877365044,"score_gpt":0.3240185823605777,"score_spread":0.29059910358692725,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2915258134","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.32065526,0.0038848093,0.49739438,0.0035797346,0.0017062282,0.0049162805,0.059371937,0.07417946,0.034311958],"genre_scores_gemma":[0.36180338,0.00046225762,0.47529787,0.00084684254,0.0003820899,0.001973269,0.14168508,0.0051513584,0.0123978015],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98973876,0.004404526,0.0010811177,0.0030420125,0.0011867981,0.00054682774],"domain_scores_gemma":[0.9824157,0.00869923,0.0007366086,0.004342258,0.003101473,0.0007047149],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009855504,0.0024009724,0.0023619595,0.0031107068,0.0027768696,0.0027584184,0.0033380254,0.0036777381,0.019067785],"category_scores_gemma":[0.027197093,0.00093279197,0.0021628004,0.0022192767,0.0015247762,0.0077033415,0.00882631,0.0035582527,0.009387188],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004778834,0.0020262299,0.0075089545,0.006375344,0.0008744977,0.0018628252,0.005210535,0.012368248,0.07164001,0.018643558,0.24139501,0.62731594],"study_design_scores_gemma":[0.0030350764,0.0024011345,0.025200069,0.000692374,0.0011421611,0.005662485,0.007480496,0.19560811,0.30360678,0.04377081,0.41078672,0.0006138453],"about_ca_topic_score_codex":0.0046802717,"about_ca_topic_score_gemma":0.005689038,"teacher_disagreement_score":0.019067785,"about_ca_system_score_codex":0.0018910274,"about_ca_system_score_gemma":0.004103371,"threshold_uncertainty_score":0.063788116},"labels":[],"label_agreement":null},{"id":"W2915427573","doi":"","title":"Proceedings of the 10th International Conference on Parsing Technologies","year":2007,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Parsing; Czech; Beijing; China; Library science; Event (particle physics); Computer science; History; Artificial intelligence; Linguistics; Archaeology","score_opus":0.022128069754142058,"score_gpt":0.28885467525338177,"score_spread":0.2667266054992397,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2915427573","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0051121013,0.061639976,0.5279755,0.02477131,0.03945894,0.00088936766,0.009983912,0.037528396,0.29264054],"genre_scores_gemma":[0.027843231,0.04483442,0.36998463,0.010229361,0.00916505,0.00095019344,0.054198768,0.010654081,0.47214034],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99616987,0.00096438377,0.00047930374,0.0008838347,0.0012102434,0.00029242274],"domain_scores_gemma":[0.99566615,0.0011848664,0.00011962721,0.0012303648,0.0014219605,0.0003770552],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003789905,0.0019857716,0.0016490679,0.002624693,0.0015188319,0.008164766,0.0029339332,0.0029431009,0.123557076],"category_scores_gemma":[0.008714556,0.00107436,0.0016609408,0.0034713305,0.0013937956,0.012109315,0.0036247442,0.0048040883,0.083460175],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018055248,0.00012647302,0.00039431412,0.00039610988,0.00007262375,0.0002321905,0.00027706515,0.00042759452,0.0037049716,0.019066805,0.65658677,0.31853437],"study_design_scores_gemma":[0.000015614683,0.000023807868,0.00030991776,0.0001742021,0.000031717787,0.0002485602,0.00008760351,0.0015754984,0.0012504059,0.0112851225,0.9849755,0.000022098387],"about_ca_topic_score_codex":0.0034579276,"about_ca_topic_score_gemma":0.003907569,"teacher_disagreement_score":0.123557076,"about_ca_system_score_codex":0.0013765574,"about_ca_system_score_gemma":0.0023242335,"threshold_uncertainty_score":0.4133396},"labels":[],"label_agreement":null},{"id":"W2916153972","doi":"","title":"PolyU at TAC 2009.","year":2009,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Automatic summarization; Computer science; Task (project management); Word (group theory); Multi-document summarization; Natural language processing; Information retrieval; Set (abstract data type); Statement (logic); Artificial intelligence; Linguistics; Programming language","score_opus":0.00538132808384473,"score_gpt":0.2542616198187059,"score_spread":0.24888029173486115,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2916153972","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03390638,0.008453703,0.163144,0.017696923,0.017343068,0.0021205433,0.07909436,0.064766616,0.6134744],"genre_scores_gemma":[0.109524876,0.0018628302,0.12868847,0.0019074151,0.0022954647,0.0012336409,0.094319426,0.0069489055,0.6532191],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99763215,0.0006307035,0.000075353055,0.0006996521,0.00066021155,0.00030190442],"domain_scores_gemma":[0.99644995,0.0004766025,0.00011728356,0.0005179123,0.0015875436,0.00085067766],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035225076,0.0014763261,0.001563269,0.001708675,0.002389334,0.0048065125,0.0017921337,0.001530318,0.17922613],"category_scores_gemma":[0.004483108,0.00047469497,0.0005535642,0.0020076805,0.0004564812,0.003323325,0.0019866498,0.0016094622,0.099672385],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008801942,0.00027428725,0.000879366,0.00024949465,0.000033898257,0.00025443782,0.00020393635,0.00082873605,0.0053413054,0.007657056,0.8014019,0.18199532],"study_design_scores_gemma":[0.00016285738,0.00024648255,0.0017779886,0.00004483394,0.000029457096,0.00015851748,0.00021656649,0.010073199,0.005843791,0.004770674,0.9766335,0.00004214917],"about_ca_topic_score_codex":0.0069210036,"about_ca_topic_score_gemma":0.009622217,"teacher_disagreement_score":0.17922613,"about_ca_system_score_codex":0.0024188005,"about_ca_system_score_gemma":0.0021054416,"threshold_uncertainty_score":0.5995711},"labels":[],"label_agreement":null},{"id":"W2916496076","doi":"","title":"AUEB at TAC 2009","year":2009,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"WordNet; Computer science; Natural language processing; Parsing; Artificial intelligence; Dependency grammar; Textual entailment; Classifier (UML); Dependency (UML); Similarity measure; Logical consequence","score_opus":0.005567143108380655,"score_gpt":0.25550555075804116,"score_spread":0.2499384076496605,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2916496076","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.134005,0.0040539387,0.11131252,0.041178674,0.02033982,0.0024761609,0.05439827,0.03967096,0.5925645],"genre_scores_gemma":[0.3023616,0.0006742337,0.0951065,0.0047631385,0.0021414652,0.0010225248,0.08039892,0.0064193243,0.5071123],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.996789,0.0008063105,0.000061040424,0.00059926277,0.0009823018,0.00076210534],"domain_scores_gemma":[0.9954072,0.0003602687,0.00007992754,0.000655383,0.0017592413,0.0017378436],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006970697,0.0007953971,0.000936397,0.0011492387,0.002929005,0.004616274,0.0018376522,0.0014672672,0.09122313],"category_scores_gemma":[0.0068263146,0.00032785925,0.00043267728,0.00075213844,0.0005556658,0.0027854936,0.0029663187,0.0024044584,0.051558796],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001078085,0.0009366364,0.0023177653,0.00010125436,0.000037081463,0.00064507464,0.0007005041,0.0014284024,0.006681979,0.0110421805,0.85511476,0.11991631],"study_design_scores_gemma":[0.00026567504,0.00029005238,0.0038535756,0.00006425903,0.000020665639,0.0002527962,0.00057547045,0.011640049,0.0044762683,0.008199653,0.9703086,0.00005304018],"about_ca_topic_score_codex":0.016595917,"about_ca_topic_score_gemma":0.024112025,"teacher_disagreement_score":0.09122313,"about_ca_system_score_codex":0.0026198057,"about_ca_system_score_gemma":0.00306419,"threshold_uncertainty_score":0.30517173},"labels":[],"label_agreement":null},{"id":"W2916862886","doi":"","title":"LIMSI @ WMT'13","year":2013,"lang":"en","type":"article","venue":"Workshop on Statistical Machine Translation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"German; Computer science; Task (project management); Natural language processing; Machine translation; Artificial intelligence; Language model; Space (punctuation); Code (set theory); Open source; Linguistics; Programming language; Set (abstract data type); Engineering","score_opus":0.018850442304512997,"score_gpt":0.29440029693052794,"score_spread":0.27554985462601495,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2916862886","genre_codex":"software","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.026854133,0.0033118553,0.15742408,0.011012764,0.023997497,0.0029056019,0.18641947,0.33279648,0.25527805],"genre_scores_gemma":[0.034548026,0.00073613354,0.10092699,0.0029560588,0.002222224,0.002120373,0.5870414,0.067362584,0.20208623],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9939261,0.0019817282,0.00039404217,0.0011827406,0.0018708776,0.00064441824],"domain_scores_gemma":[0.98962027,0.0013087818,0.00022982206,0.0031839681,0.0029651083,0.0026920899],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0073259026,0.0034327852,0.0024099948,0.0027052527,0.0024792608,0.005684023,0.0037732134,0.002711939,0.20418738],"category_scores_gemma":[0.016803112,0.0011570415,0.0013751981,0.0027704465,0.0010850453,0.0038128311,0.008564118,0.003774239,0.3002972],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000538116,0.00024274067,0.0002962551,0.0003763246,0.000047595004,0.00023411536,0.00017785437,0.00042117442,0.0045413612,0.0017296607,0.9377955,0.053599235],"study_design_scores_gemma":[0.00029695526,0.00026746563,0.00096087565,0.000080121165,0.000022127659,0.00036237034,0.00021105795,0.0061637396,0.008640433,0.0051230765,0.9777999,0.00007196351],"about_ca_topic_score_codex":0.0042737443,"about_ca_topic_score_gemma":0.006185933,"teacher_disagreement_score":0.20418738,"about_ca_system_score_codex":0.001647991,"about_ca_system_score_gemma":0.0031889568,"threshold_uncertainty_score":0.68307483},"labels":[],"label_agreement":null},{"id":"W2917198466","doi":"","title":"Overview of Linguistic Resources for the TAC KBP 2017 Evaluations: Methodologies and Results.","year":2017,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":28,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Natural language processing; Linguistics; Artificial intelligence; Philosophy","score_opus":0.0975557432067578,"score_gpt":0.4224822658712334,"score_spread":0.3249265226644756,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2917198466","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.18670939,0.03809331,0.17353062,0.006335504,0.002043497,0.023710733,0.3616185,0.040851466,0.16710696],"genre_scores_gemma":[0.18256699,0.0067828353,0.25348464,0.0011170858,0.00038051975,0.020712705,0.5110499,0.004243813,0.019661436],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9755078,0.011204929,0.0027259055,0.0016975296,0.007914618,0.00094908953],"domain_scores_gemma":[0.94146377,0.021788387,0.0018474768,0.0067235073,0.025496501,0.0026804628],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.019484915,0.0022721307,0.0017779607,0.014539214,0.003103657,0.0051695826,0.004499431,0.0020746132,0.02063076],"category_scores_gemma":[0.06674006,0.0009044532,0.0012604125,0.009711851,0.0013047508,0.007916971,0.0065743816,0.002987369,0.017179132],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0037529843,0.0031029591,0.009488537,0.016301988,0.00090772647,0.0007354934,0.0035905216,0.011004991,0.012408503,0.008756733,0.3627498,0.5671999],"study_design_scores_gemma":[0.0020468452,0.0019782905,0.024990661,0.0072767623,0.0020007987,0.0013568327,0.00777691,0.04396117,0.044866122,0.014589523,0.8484793,0.0006766998],"about_ca_topic_score_codex":0.022533549,"about_ca_topic_score_gemma":0.02702193,"teacher_disagreement_score":0.022533549,"about_ca_system_score_codex":0.002873591,"about_ca_system_score_gemma":0.00916582,"threshold_uncertainty_score":0.10304737},"labels":[],"label_agreement":null},{"id":"W2917601773","doi":"10.7202/1055247ar","title":"L’indexation automatique : état de la question et perspectives d’avenir","year":2019,"lang":"fr","type":"article","venue":"Documentation et bibliothèques","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Philosophy; Humanities","score_opus":0.010568374895546843,"score_gpt":0.3442220905423182,"score_spread":0.3336537156467714,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2917601773","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010402622,0.036035486,0.87030137,0.023934456,0.0026980713,0.00030539464,0.0012375469,0.004155558,0.050929457],"genre_scores_gemma":[0.06738744,0.028924016,0.83583814,0.0037257825,0.0031702814,0.0005735183,0.0022516323,0.0038564317,0.05427272],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9747515,0.008935161,0.0020686626,0.0029615145,0.010557441,0.0007256857],"domain_scores_gemma":[0.9262687,0.034297638,0.0023821166,0.018867686,0.017453013,0.0007308739],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.021985386,0.0017175776,0.002033518,0.009322604,0.0033701328,0.017416239,0.0036510604,0.0039983676,0.016753491],"category_scores_gemma":[0.072669506,0.001438958,0.0020341421,0.013505472,0.0064224647,0.02279778,0.0042283274,0.005327612,0.010273925],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018838799,0.00012533019,0.00212724,0.0017149847,0.00012863422,0.00016498698,0.0023031184,0.0031009405,0.006741883,0.3768527,0.050269663,0.5562821],"study_design_scores_gemma":[0.00007894111,0.0001739432,0.0031624152,0.0013175831,0.00010846942,0.0011996994,0.0012209843,0.020914737,0.019069849,0.2132622,0.73927355,0.0002175601],"about_ca_topic_score_codex":0.013867613,"about_ca_topic_score_gemma":0.00837577,"teacher_disagreement_score":0.021985386,"about_ca_system_score_codex":0.004857314,"about_ca_system_score_gemma":0.007841188,"threshold_uncertainty_score":0.1162712},"labels":[],"label_agreement":null},{"id":"W2917659727","doi":"10.5539/ijel.v9n2p99","title":"Proposing a Customised Method for Extratextual Documentative Annotation on Written Text Corpus","year":2019,"lang":"en","type":"article","venue":"International Journal of English Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"King Saud University","keywords":"Annotation; Computer science; Natural language processing; Text corpus; Artificial intelligence; Corpus linguistics; Linguistics; Grammar; Sketch; Information retrieval","score_opus":0.013016057471155385,"score_gpt":0.3386101849843603,"score_spread":0.32559412751320493,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2917659727","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0010957742,0.00007483659,0.9888075,0.00014797537,0.00016801678,0.00036109236,0.0003004922,0.0067264773,0.0023178917],"genre_scores_gemma":[0.010444141,0.00010589629,0.97791266,0.00012448298,0.00010131336,0.00085043773,0.0018273174,0.0025380412,0.006095713],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99111253,0.0024428596,0.000989739,0.0024576958,0.002671458,0.0003256542],"domain_scores_gemma":[0.9834184,0.004819469,0.0006302036,0.0060975356,0.004589024,0.00044529547],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007549013,0.0015584722,0.0014240743,0.00796837,0.0028476273,0.005638601,0.004179818,0.0018718917,0.020992069],"category_scores_gemma":[0.019404734,0.0013305164,0.0020846317,0.006552155,0.0027116835,0.008257407,0.008882317,0.0039796666,0.012640121],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002949887,0.00016301143,0.0010463854,0.0010738095,0.0001278779,0.0008290867,0.0054677315,0.0040932344,0.06103562,0.058631565,0.028341593,0.8388951],"study_design_scores_gemma":[0.00016399327,0.0002462608,0.002877033,0.0005194896,0.00019539498,0.0023448495,0.0039220033,0.16726634,0.11381874,0.069768906,0.6384448,0.00043217523],"about_ca_topic_score_codex":0.0027309824,"about_ca_topic_score_gemma":0.0035140738,"teacher_disagreement_score":0.020992069,"about_ca_system_score_codex":0.0011296452,"about_ca_system_score_gemma":0.0028394433,"threshold_uncertainty_score":0.07022548},"labels":[],"label_agreement":null},{"id":"W2921492024","doi":"10.17169/refubium-1361","title":"Multiword expressions at length and in depth","year":2018,"lang":"en","type":"preprint","venue":"BiblioBoard Library Catalog (Open Research Library)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Killam Trusts","keywords":"Computer science; Linguistics; Mathematics; Philosophy","score_opus":0.08918085580013325,"score_gpt":0.37453877823911624,"score_spread":0.285357922438983,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2921492024","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014112983,0.036014087,0.693038,0.02633621,0.008617773,0.0002740056,0.0025406312,0.004081124,0.21498518],"genre_scores_gemma":[0.22314785,0.03164421,0.37739924,0.0065531004,0.0048335595,0.00061695155,0.0074442024,0.008859507,0.33950147],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99804074,0.0006991878,0.0001717746,0.00032149313,0.00065896346,0.00010781957],"domain_scores_gemma":[0.998095,0.0008916689,0.00011678217,0.0003323548,0.0004902171,0.00007390662],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013526677,0.001398525,0.0008631657,0.0017645127,0.0020001992,0.008262293,0.001232108,0.0015677074,0.029427286],"category_scores_gemma":[0.006080576,0.00091548864,0.0010596425,0.003251839,0.002128893,0.015205733,0.0040651,0.0058573084,0.01564088],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001256526,0.000044460685,0.00047803394,0.0010644749,0.00005733172,0.0004248005,0.0072719,0.0011279659,0.009219567,0.55722296,0.13277137,0.29019156],"study_design_scores_gemma":[0.000006301796,0.000018709696,0.0004524164,0.000520928,0.000023796863,0.000556417,0.0017975874,0.003317333,0.0040312344,0.20811863,0.7811233,0.000033291097],"about_ca_topic_score_codex":0.0012320827,"about_ca_topic_score_gemma":0.0020634464,"teacher_disagreement_score":0.029427286,"about_ca_system_score_codex":0.0015433502,"about_ca_system_score_gemma":0.0015281695,"threshold_uncertainty_score":0.098444104},"labels":[],"label_agreement":null},{"id":"W2921737336","doi":"","title":"Compiling and Managing a Bilingual Lexicon in METIS-II","year":2007,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Metis; Lexicon; Computer science; Natural language processing; Artificial intelligence; Linguistics; World Wide Web; Philosophy","score_opus":0.012160493183813877,"score_gpt":0.2918183269246448,"score_spread":0.2796578337408309,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2921737336","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.33635357,0.001445408,0.5551193,0.00093207945,0.00027572596,0.0014369183,0.017317649,0.035342254,0.05177717],"genre_scores_gemma":[0.45292276,0.0007080587,0.4892049,0.00014520102,0.000082548315,0.0006424135,0.037497353,0.0048470786,0.013949686],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99918026,0.00014516857,0.00013796907,0.0002747212,0.0001970083,0.00006493956],"domain_scores_gemma":[0.99900454,0.00023479413,0.0001288684,0.00022022272,0.00031268303,0.00009892525],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008132008,0.00053374044,0.00081509136,0.004097638,0.0013637409,0.0033450446,0.00086792855,0.00027996663,0.0049259826],"category_scores_gemma":[0.0020910485,0.0006054752,0.0004723805,0.0030070823,0.0005839229,0.0026449212,0.0020367254,0.0005774048,0.0029444809],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013297511,0.0003076003,0.0184261,0.001423231,0.00013280332,0.002930107,0.007689311,0.007853377,0.11272358,0.047626626,0.032586537,0.76697105],"study_design_scores_gemma":[0.00041294162,0.00054182095,0.019827569,0.0004013341,0.00035891263,0.003078086,0.0075389035,0.060735375,0.16124457,0.028159436,0.7174907,0.0002103497],"about_ca_topic_score_codex":0.005147512,"about_ca_topic_score_gemma":0.008897198,"teacher_disagreement_score":0.005147512,"about_ca_system_score_codex":0.0015040275,"about_ca_system_score_gemma":0.0026268922,"threshold_uncertainty_score":0.016479075},"labels":[],"label_agreement":null},{"id":"W2921822841","doi":"10.7202/1057966ar","title":"J’ai l’impression que: Lexical Bundles in the Dialogues of Beginner French Textbooks","year":2019,"lang":"en","type":"article","venue":"Canadian Journal of Applied Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Linguistics; Corpus linguistics; Lexical item; Computer science; Psychology; Philosophy","score_opus":0.011652168422202102,"score_gpt":0.24214997687187315,"score_spread":0.23049780844967105,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2921822841","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9858704,0.0003541483,0.005665387,0.000514167,0.000024162346,0.000019984407,0.000111356756,0.00007376486,0.0073666233],"genre_scores_gemma":[0.9946233,0.00012339413,0.0031703357,0.00006368448,0.00001050626,0.000016944678,0.00012582703,0.000025717729,0.0018401925],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9984249,0.00097574876,0.00005265364,0.0002105001,0.0002606215,0.00007559381],"domain_scores_gemma":[0.99512863,0.003203702,0.0006522419,0.00022331193,0.00054052984,0.0002516385],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015735248,0.00033017667,0.00023831506,0.001397627,0.0012956995,0.0028920565,0.0002458343,0.00071193185,0.003288552],"category_scores_gemma":[0.012518049,0.00016494918,0.000146304,0.0007453599,0.0014479093,0.0027527774,0.00090909586,0.0005649237,0.00049536454],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000901703,0.00030717516,0.17973435,0.00057840865,0.00007972199,0.002939411,0.4847999,0.0006704471,0.060071375,0.015787212,0.0058036055,0.24832673],"study_design_scores_gemma":[0.00006577054,0.00064448715,0.63128227,0.00034832917,0.00009250848,0.0038596187,0.25708586,0.006135805,0.016675266,0.008457557,0.07517506,0.00017738412],"about_ca_topic_score_codex":0.00818311,"about_ca_topic_score_gemma":0.013114488,"teacher_disagreement_score":0.00818311,"about_ca_system_score_codex":0.0013331731,"about_ca_system_score_gemma":0.00054748787,"threshold_uncertainty_score":0.016270995},"labels":[],"label_agreement":null},{"id":"W2923275854","doi":"10.26220/pwpl.v5i0.3073","title":"A data archiving and browsing multimodal system for the language of Greek immigrants in Canada","year":2019,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Immigration; Computer science; World Wide Web; Linguistics; History","score_opus":0.013630360848945425,"score_gpt":0.25232832246067377,"score_spread":0.23869796161172835,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2923275854","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9249756,0.0007552594,0.018059347,0.0008729228,0.00007211414,0.0004051072,0.018530514,0.010592165,0.025736988],"genre_scores_gemma":[0.9312765,0.0006473219,0.036579143,0.00019809995,0.000020375533,0.00016202034,0.0091442,0.00051959313,0.02145269],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99982303,0.000025001458,0.000015224896,0.00004071698,0.00005450659,0.000041608087],"domain_scores_gemma":[0.99916625,0.00013968385,0.000024257377,0.000056727204,0.00048869575,0.00012431075],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005212943,0.00037879025,0.00037402663,0.0018040095,0.0024283275,0.0015639472,0.0005092487,0.00051274884,0.005617527],"category_scores_gemma":[0.0012334125,0.00011968677,0.00018934207,0.0019867218,0.0004120733,0.0006490506,0.000886823,0.0003830255,0.001331026],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019081603,0.00026997694,0.07900789,0.0008722637,0.00012117351,0.0042578843,0.039496973,0.0029234833,0.16447102,0.003309182,0.078911774,0.6244502],"study_design_scores_gemma":[0.00023864637,0.0004421422,0.34181544,0.00076974824,0.0006729324,0.0029075616,0.1009832,0.07161402,0.12824911,0.0028070663,0.34866613,0.0008339314],"about_ca_topic_score_codex":0.85703105,"about_ca_topic_score_gemma":0.876501,"teacher_disagreement_score":0.14296895,"about_ca_system_score_codex":0.004474591,"about_ca_system_score_gemma":0.01057614,"threshold_uncertainty_score":0.2876218},"labels":[],"label_agreement":null},{"id":"W2925077947","doi":"10.1371/journal.pone.0212342","title":"Talk2Me: Automated linguistic data collection for personal assessment","year":2019,"lang":"en","type":"article","venue":"PLoS ONE","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada; Vector Institute; St. Michael's Hospital; University of Toronto; Carleton University","funders":"Canadian Institutes of Health Research; Natural Sciences and Engineering Research Council of Canada; Alzheimer Society Research Program; Alzheimer Society; California HIV/AIDS Research Program","keywords":"Computer science; Variety (cybernetics); Natural language processing; Software; Artificial intelligence; Baseline (sea); Linguistics; Programming language","score_opus":0.05615982920436336,"score_gpt":0.31950402849216925,"score_spread":0.2633441992878059,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2925077947","genre_codex":"dataset","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.121406,0.0009188131,0.2836477,0.0009930739,0.00058164244,0.005517691,0.39575008,0.16311535,0.02806969],"genre_scores_gemma":[0.19328171,0.00053483114,0.3846285,0.0007414257,0.00039866945,0.012888092,0.37633726,0.008563276,0.022626165],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99566394,0.0015255492,0.0004892703,0.0008984103,0.0012174436,0.0002053866],"domain_scores_gemma":[0.98397887,0.0061360793,0.0010914146,0.0033913224,0.0042169476,0.00118531],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0046191276,0.0016547927,0.0009973162,0.005407118,0.0007442786,0.0018999885,0.0016048729,0.0010562284,0.03492023],"category_scores_gemma":[0.024591861,0.0007678864,0.00087193126,0.001969941,0.0004509206,0.0023532733,0.0048905383,0.0013553919,0.026716834],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0023640227,0.0008111137,0.047424495,0.0022717656,0.0004970247,0.00057269615,0.0021908658,0.0023151517,0.02406579,0.0024066623,0.34938145,0.565699],"study_design_scores_gemma":[0.000947936,0.001573181,0.3680009,0.0010293552,0.00035255076,0.003134979,0.0028695473,0.06367804,0.07585905,0.025786808,0.4558626,0.0009050398],"about_ca_topic_score_codex":0.0030605432,"about_ca_topic_score_gemma":0.0073602074,"teacher_disagreement_score":0.03492023,"about_ca_system_score_codex":0.0004985121,"about_ca_system_score_gemma":0.0017986515,"threshold_uncertainty_score":0.11681986},"labels":[],"label_agreement":null},{"id":"W2925631053","doi":"10.1007/978-3-030-15719-7_14","title":"Neural Diverse Abstractive Sentence Compression Generation","year":2019,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Lethbridge","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Paraphrase; Sentence; Natural language processing; Machine translation; Artificial intelligence; Encoder; Compression (physics); Speech recognition","score_opus":0.02282935388186901,"score_gpt":0.2712604724152494,"score_spread":0.2484311185333804,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2925631053","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08835626,0.0017151224,0.8667024,0.000761164,0.00092840905,0.00040641858,0.0019734858,0.012372741,0.026784016],"genre_scores_gemma":[0.5239674,0.0006132125,0.43666977,0.00059697893,0.00048095678,0.00045701695,0.006943713,0.0009359778,0.029335042],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99967146,0.00007700103,0.000021260168,0.00008751377,0.000097955264,0.00004482709],"domain_scores_gemma":[0.99934083,0.0002996963,0.000023991402,0.00011252133,0.00018973161,0.00003315923],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00043434493,0.0008524704,0.00064655195,0.0007870563,0.00044285145,0.000710561,0.0013047229,0.0010407433,0.01645932],"category_scores_gemma":[0.0015903929,0.0003055589,0.0006013116,0.0008478646,0.00037466292,0.0010927113,0.0012405873,0.0011046155,0.0043072687],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036856424,0.00020083708,0.00034052684,0.00027648988,0.000050304363,0.00042711513,0.000111585105,0.031710874,0.051136408,0.015999034,0.024837006,0.8745413],"study_design_scores_gemma":[0.000088308836,0.00022206389,0.00073021115,0.000043076274,0.000068412,0.00044722902,0.00007608974,0.89534146,0.06306935,0.029963097,0.009916576,0.000034214856],"about_ca_topic_score_codex":0.0009335571,"about_ca_topic_score_gemma":0.0022968461,"teacher_disagreement_score":0.01645932,"about_ca_system_score_codex":0.0003376896,"about_ca_system_score_gemma":0.00045220574,"threshold_uncertainty_score":0.055061877},"labels":[],"label_agreement":null},{"id":"W2932742427","doi":"10.1075/tis.00031.mos","title":"‘Intersemiotic translating’","year":2019,"lang":"en","type":"article","venue":"Translation and Interpreting Studies","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Linguistics; Variety (cybernetics); Translation (biology); Computer science; Feature (linguistics); Degree (music); Sign (mathematics); Mathematics; Artificial intelligence; Philosophy; Physics","score_opus":0.02457646788039257,"score_gpt":0.3129660369382813,"score_spread":0.2883895690578887,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2932742427","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16801184,0.0011381259,0.40289468,0.004521101,0.0016894193,0.00013914553,0.00036376272,0.0009960514,0.42024592],"genre_scores_gemma":[0.94399077,0.00032520827,0.03918081,0.0008681612,0.00032530128,0.000052315692,0.00031068068,0.0002589376,0.014687754],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9960109,0.0016571328,0.00028287948,0.0007449961,0.001004036,0.00030008386],"domain_scores_gemma":[0.99365145,0.001363249,0.00058733404,0.0029807636,0.0012425588,0.00017470717],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022265848,0.00047105356,0.00041298722,0.0012097665,0.0010700689,0.0028175642,0.0007648067,0.0007891768,0.009598218],"category_scores_gemma":[0.008416375,0.00022304113,0.0006356451,0.0017583793,0.0063404115,0.003763631,0.0018870851,0.0021420524,0.002072338],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013920282,0.0000515188,0.0017197264,0.00011504002,0.00003099534,0.00044080932,0.0032057865,0.0006581414,0.0076535535,0.9039074,0.0037311823,0.07834657],"study_design_scores_gemma":[0.000046144247,0.0002275167,0.009270242,0.00011222033,0.000055854412,0.0023608874,0.00400232,0.008036189,0.02464176,0.83387613,0.11728473,0.00008598194],"about_ca_topic_score_codex":0.00054711284,"about_ca_topic_score_gemma":0.00038605198,"teacher_disagreement_score":0.009598218,"about_ca_system_score_codex":0.00072718435,"about_ca_system_score_gemma":0.00070460606,"threshold_uncertainty_score":0.03210926},"labels":[],"label_agreement":null},{"id":"W2934879341","doi":"10.1007/978-3-030-15712-8_61","title":"Extracting Temporal Event Relations Based on Event Networks","year":2019,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Computer science; Event (particle physics); Traverse; Artificial intelligence; Context (archaeology); Set (abstract data type); Natural language processing; Sentence; Relationship extraction; Relation (database); Data mining; Information extraction","score_opus":0.013095279612134282,"score_gpt":0.2691262062877092,"score_spread":0.2560309266755749,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2934879341","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.059884854,0.003178699,0.90830696,0.0003602185,0.0002455105,0.00048358462,0.011343656,0.0054377336,0.010758826],"genre_scores_gemma":[0.41648385,0.00409049,0.5418327,0.00015764024,0.0003437033,0.00045026466,0.027937362,0.0005563158,0.008147593],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99933547,0.000066132576,0.00007755417,0.00029442183,0.00016099923,0.0000654505],"domain_scores_gemma":[0.99853206,0.0008659991,0.00022489476,0.000121402874,0.00019575135,0.000059958867],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00056949985,0.0012828044,0.0006668936,0.005312946,0.00064689195,0.0013658693,0.0009473441,0.00076014164,0.0048229652],"category_scores_gemma":[0.0031270415,0.00055652764,0.0013231882,0.0041544857,0.00034150627,0.0025318025,0.0008730994,0.0010216209,0.0026447854],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015759438,0.0005203859,0.019124733,0.0014460873,0.00040070165,0.0030360892,0.0004693048,0.043934345,0.06143791,0.036501482,0.019231582,0.81232154],"study_design_scores_gemma":[0.0001176446,0.00026002573,0.027853677,0.00044160613,0.001003245,0.0025179447,0.0006101404,0.7635137,0.04418821,0.09494999,0.064426005,0.00011790911],"about_ca_topic_score_codex":0.00547202,"about_ca_topic_score_gemma":0.008891772,"teacher_disagreement_score":0.00547202,"about_ca_system_score_codex":0.00070472626,"about_ca_system_score_gemma":0.00087801233,"threshold_uncertainty_score":0.01613444},"labels":[],"label_agreement":null},{"id":"W2937821344","doi":"10.1609/aaai.v34i05.6296","title":"One Homonym per Translation","year":2020,"lang":"en","type":"preprint","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Innovates","keywords":"Homonym (biology); Polysemy; Context (archaeology); Computer science; Semantics (computer science); Natural language processing; Linguistics; Lexical semantics; Syntax; Resource (disambiguation); Translation (biology); Artificial intelligence; Lexical item; History; Programming language; Philosophy; Biology","score_opus":0.13614279806786667,"score_gpt":0.3266309128161678,"score_spread":0.19048811474830113,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2937821344","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.49167496,0.0030890654,0.30437487,0.0039109243,0.0012300473,0.0003569114,0.01648586,0.0040154546,0.17486185],"genre_scores_gemma":[0.8693614,0.00057330314,0.109269656,0.0003840483,0.00018488607,0.00012186369,0.005963064,0.00058839243,0.013553382],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9978671,0.0005936562,0.00022082872,0.00065625797,0.0005553031,0.00010692019],"domain_scores_gemma":[0.99559987,0.0017633354,0.00022868314,0.0017207072,0.00051618554,0.0001712529],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001265871,0.00033431174,0.0006333946,0.0020842766,0.0016909224,0.0021129558,0.0005151336,0.0008620889,0.02010652],"category_scores_gemma":[0.0056326226,0.00035026507,0.00039221314,0.002494946,0.0018039775,0.0069657383,0.0025786145,0.0009756962,0.003831719],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00085053686,0.000180306,0.016116384,0.0014691866,0.00012296223,0.0020326884,0.011310211,0.0021086133,0.08942697,0.44527227,0.02338235,0.40772748],"study_design_scores_gemma":[0.00007076331,0.00020022347,0.020724524,0.0004203371,0.0001496233,0.004866252,0.0070210886,0.014884015,0.037744246,0.5180736,0.39562342,0.0002217553],"about_ca_topic_score_codex":0.00053741824,"about_ca_topic_score_gemma":0.00085627876,"teacher_disagreement_score":0.02010652,"about_ca_system_score_codex":0.00052858796,"about_ca_system_score_gemma":0.0009207947,"threshold_uncertainty_score":0.06726301},"labels":[],"label_agreement":null},{"id":"W2938876272","doi":"10.5167/uzh-153213","title":"Exploiting alignment in multiparallel corpora for applications in linguistics and language learning","year":2018,"lang":"en","type":"dissertation","venue":"Zurich Open Repository and Archive (University of Zurich)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Atomic Energy of Canada Limited; University College London; Schweizerischer Nationalfonds zur Förderung der Wissenschaftlichen Forschung; National Science Foundation","keywords":"Linguistics; Applied linguistics; Corpus linguistics; Computer science; Natural language processing; Artificial intelligence; Philosophy","score_opus":0.011344332125427207,"score_gpt":0.25986330059838414,"score_spread":0.24851896847295693,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2938876272","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010688271,0.0013206236,0.97412777,0.0011281644,0.00043780284,0.00025607416,0.0012539212,0.0048653483,0.0059220116],"genre_scores_gemma":[0.05833889,0.0011594297,0.927708,0.00025049856,0.00040119138,0.00056152005,0.0058416417,0.0020951289,0.0036436813],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.991438,0.004071116,0.00074827916,0.0019777159,0.0015184854,0.00024643968],"domain_scores_gemma":[0.9837901,0.008470118,0.0013293659,0.0042424323,0.0018746438,0.00029337063],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0065926537,0.0016329031,0.0014813113,0.005708662,0.0029639138,0.0051296656,0.001958036,0.0016473634,0.009181509],"category_scores_gemma":[0.025378518,0.0016916139,0.0017880057,0.010136701,0.0021121898,0.012326965,0.0060909796,0.0041617956,0.007799119],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000376151,0.0003812181,0.0037179636,0.0017343571,0.0003161232,0.0011988989,0.004534289,0.022568393,0.035270203,0.11173514,0.03062861,0.78753865],"study_design_scores_gemma":[0.0001456157,0.00021004387,0.005251793,0.000563731,0.00018102821,0.0011783191,0.002742865,0.24458282,0.04144469,0.48506302,0.21839239,0.00024371417],"about_ca_topic_score_codex":0.0012274683,"about_ca_topic_score_gemma":0.0024698807,"teacher_disagreement_score":0.009181509,"about_ca_system_score_codex":0.0009846893,"about_ca_system_score_gemma":0.0022662836,"threshold_uncertainty_score":0.034865677},"labels":[],"label_agreement":null},{"id":"W2939226676","doi":"10.46430/phfr0005","title":"Travailler avec des fichiers texte en Python","year":2019,"lang":"fr","type":"article","venue":"The Programming Historian en français","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Python (programming language); Computer science; Humanities; Art; Programming language","score_opus":0.009394658208216198,"score_gpt":0.2370905729434597,"score_spread":0.2276959147352435,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2939226676","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012783374,0.00045581828,0.75060415,0.0021097667,0.00095772854,0.00028519958,0.0031449646,0.19209215,0.037566792],"genre_scores_gemma":[0.14389625,0.0012332606,0.6215983,0.002502209,0.0007756603,0.0007365056,0.010088558,0.120732345,0.09843694],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9943445,0.0013649709,0.0004167786,0.0011744929,0.0022310053,0.0004683433],"domain_scores_gemma":[0.99321026,0.0032902556,0.0003530945,0.0016931695,0.0012168015,0.00023644268],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0038197285,0.0015961789,0.0011316049,0.0012812708,0.0024913526,0.0050904225,0.00241247,0.0015406103,0.04520454],"category_scores_gemma":[0.018469015,0.0011557395,0.0023771543,0.0012126981,0.0027108635,0.009166681,0.0052797436,0.0049878634,0.026376298],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009081666,0.00032198324,0.0036967197,0.0029248828,0.00037727112,0.0023920476,0.010583515,0.0082913255,0.044403467,0.2721949,0.32590464,0.32800105],"study_design_scores_gemma":[0.00009654256,0.00009423244,0.0012555976,0.000368469,0.00010703647,0.0011658774,0.00066240766,0.030624475,0.058514792,0.057922367,0.8489762,0.00021205495],"about_ca_topic_score_codex":0.006231315,"about_ca_topic_score_gemma":0.009638926,"teacher_disagreement_score":0.04520454,"about_ca_system_score_codex":0.0010976724,"about_ca_system_score_gemma":0.0027881186,"threshold_uncertainty_score":0.15122426},"labels":[],"label_agreement":null},{"id":"W2939284995","doi":"10.7939/r32z1352z","title":"Understanding and Fostering SoTL Cultures across a Nation","year":2018,"lang":"en","type":"article","venue":"Open MIND","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Sociology; Political science","score_opus":0.1927821673680027,"score_gpt":0.39930832229010454,"score_spread":0.20652615492210186,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2939284995","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9241178,0.001473581,0.0013803674,0.018165598,0.00024235873,0.00009405781,0.000094361225,0.000032677108,0.054399136],"genre_scores_gemma":[0.99197173,0.00092119625,0.000863051,0.0021436897,0.000011718008,0.000044520275,0.000060647544,0.0000459081,0.003937584],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.99075186,0.0037398192,0.00029396266,0.0007000034,0.0016239852,0.002890381],"domain_scores_gemma":[0.9829123,0.002668482,0.0012664228,0.0010054073,0.005235263,0.006912123],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0104950955,0.00033281173,0.00047911677,0.002914921,0.03098542,0.015858388,0.0027401599,0.0014573393,0.004180342],"category_scores_gemma":[0.014795178,0.00047137405,0.00035410232,0.0028894753,0.021783097,0.007582313,0.017644836,0.0035886203,0.00037812872],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000020560432,0.000026497353,0.0163746,0.000050653165,0.000006871459,0.00026521186,0.9591534,0.00002913153,0.00037155332,0.0107361805,0.0025206816,0.010444687],"study_design_scores_gemma":[0.0000022530114,0.000009621519,0.006143982,0.00015053575,0.000004135028,0.00006759525,0.9699746,0.00006855339,0.00008847405,0.00126347,0.022209479,0.000017319115],"about_ca_topic_score_codex":0.72699,"about_ca_topic_score_gemma":0.81942505,"teacher_disagreement_score":0.98950493,"about_ca_system_score_codex":0.036533885,"about_ca_system_score_gemma":0.07697711,"threshold_uncertainty_score":0.54923564},"labels":[],"label_agreement":null},{"id":"W2942208127","doi":"","title":"Analyzing and modeling free word associations.","year":2018,"lang":"en","type":"article","venue":"eScholarship (California Digital Library)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Word (group theory); Computer science; Natural language processing; Artificial intelligence; Linguistics; Philosophy","score_opus":0.017493919209701014,"score_gpt":0.24267328846059247,"score_spread":0.22517936925089146,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2942208127","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11445848,0.0010216556,0.8780278,0.00058211706,0.000070255,0.0000887221,0.00086983613,0.0012392446,0.003641957],"genre_scores_gemma":[0.75928915,0.00073390716,0.23063098,0.00014495783,0.000120752586,0.00032566718,0.0031459096,0.0002881246,0.0053206286],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99872416,0.0005512856,0.00006682045,0.00037755072,0.00017876318,0.000101434125],"domain_scores_gemma":[0.9865411,0.010826016,0.00081285066,0.0011408894,0.00044966055,0.00022953561],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035383874,0.00088250206,0.0008228653,0.0027617903,0.0007921895,0.0025306516,0.0015234092,0.0013472999,0.0038952464],"category_scores_gemma":[0.022817334,0.0007644741,0.0013851296,0.0023339265,0.00090306683,0.005507237,0.0014454018,0.0015989252,0.0014043736],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00032267056,0.000291824,0.03361161,0.00036528165,0.0004061225,0.0005465739,0.0014328945,0.5454485,0.003752947,0.1716324,0.008328636,0.2338606],"study_design_scores_gemma":[0.000009976759,0.000020526637,0.0011718061,0.000012599459,0.000019083456,0.000108390566,0.00008712426,0.8809799,0.0005925879,0.11528477,0.0017023658,0.00001091263],"about_ca_topic_score_codex":0.007068981,"about_ca_topic_score_gemma":0.010382703,"teacher_disagreement_score":0.007068981,"about_ca_system_score_codex":0.00079445174,"about_ca_system_score_gemma":0.0011884408,"threshold_uncertainty_score":0.018712997},"labels":[],"label_agreement":null},{"id":"W2942742149","doi":"","title":"Harnessing the Redundant Results of Translation Spotting","year":2009,"lang":"en","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; Computer Research Institute of Montréal","funders":"","keywords":"Spotting; Computer science; Translation (biology); Artificial intelligence; Natural language processing","score_opus":0.018622631075136804,"score_gpt":0.2559947502530049,"score_spread":0.23737211917786807,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2942742149","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04771413,0.0013077892,0.91974694,0.0015120601,0.000897124,0.00013079707,0.0005581976,0.012512958,0.0156200165],"genre_scores_gemma":[0.3676932,0.0012385705,0.60053086,0.00071074406,0.0014890328,0.0001636764,0.002880885,0.007777362,0.017515646],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99465877,0.0018312507,0.00043167736,0.000800403,0.0018624247,0.00041539472],"domain_scores_gemma":[0.9731602,0.011400197,0.0010200171,0.009892613,0.004112959,0.00041393982],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0039437227,0.0031235253,0.0027545283,0.0027987252,0.0014425301,0.004481774,0.0030352816,0.0020420349,0.01675167],"category_scores_gemma":[0.024658196,0.0011940598,0.0015912601,0.003617864,0.0021004386,0.0060338797,0.0052390452,0.001829163,0.010663321],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021427553,0.000227998,0.0014622407,0.0014045378,0.00032669827,0.0018923306,0.0012218765,0.028175618,0.13619193,0.035959292,0.018445125,0.7725497],"study_design_scores_gemma":[0.00022735192,0.00085829344,0.0012119737,0.00019677791,0.0006992988,0.0017393102,0.0006074171,0.4681157,0.3138795,0.16701257,0.04525644,0.00019540805],"about_ca_topic_score_codex":0.0007288366,"about_ca_topic_score_gemma":0.001142455,"teacher_disagreement_score":0.01675167,"about_ca_system_score_codex":0.0004485,"about_ca_system_score_gemma":0.0012805077,"threshold_uncertainty_score":0.05603993},"labels":[],"label_agreement":null},{"id":"W2943085977","doi":"","title":"Brief Report: Ordered Neurons: Integrating Tree Structures into Recurrent Neural Networks","year":2019,"lang":"en","type":"article","venue":"International Conference on Learning Representations","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Parsing; Artificial intelligence; Language model; Syntax; Tree structure; Tree (set theory); Recurrent neural network; Natural language processing; Artificial neural network; Machine learning; Algorithm; Binary tree; Mathematics","score_opus":0.024539682217917903,"score_gpt":0.3385625186261699,"score_spread":0.31402283640825196,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2943085977","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010509123,0.00828914,0.9541781,0.006498829,0.0042559197,0.00018268319,0.0011871975,0.0056194854,0.00927952],"genre_scores_gemma":[0.21922538,0.016641518,0.6895276,0.0023401624,0.00784137,0.0006383728,0.0066240104,0.0027900892,0.054371506],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99950945,0.00015961789,0.000033524477,0.00013636927,0.00012340103,0.000037607133],"domain_scores_gemma":[0.9985386,0.0005917597,0.00007602944,0.00016337131,0.00047719633,0.00015295953],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016584488,0.0009426411,0.0005741089,0.00041615934,0.00030892537,0.001765604,0.0020482088,0.0016476661,0.012259628],"category_scores_gemma":[0.005938718,0.0005749714,0.00080955453,0.00083453144,0.0005057819,0.0030927071,0.0011551776,0.0028124496,0.0065994244],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037045073,0.00019476611,0.0028497633,0.0007383519,0.00038735825,0.00043947579,0.0003129715,0.16321978,0.009265589,0.06760303,0.1788181,0.5758004],"study_design_scores_gemma":[0.00008673592,0.00018245251,0.0007715387,0.00012529723,0.0001488427,0.00018315979,0.00004161222,0.8435679,0.005999983,0.08034243,0.068455316,0.00009473332],"about_ca_topic_score_codex":0.0047739013,"about_ca_topic_score_gemma":0.004541619,"teacher_disagreement_score":0.012259628,"about_ca_system_score_codex":0.00071379094,"about_ca_system_score_gemma":0.0008009113,"threshold_uncertainty_score":0.041012585},"labels":[],"label_agreement":null},{"id":"W2943921606","doi":"10.1162/ling_a_00342","title":"Null Objects in Korean: Experimental Evidence for the Argument Ellipsis Analysis","year":2019,"lang":"en","type":"article","venue":"Linguistic Inquiry","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; Simon Fraser University","funders":"","keywords":"Ellipsis (linguistics); Linguistics; Argument (complex analysis); Quantifier (linguistics); Adverb; Pronoun; Verb; Object (grammar); Raising (metalworking); Mathematics; Philosophy","score_opus":0.05239265450870805,"score_gpt":0.3587575137403947,"score_spread":0.30636485923168666,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2943921606","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99532986,0.00012970767,0.0009373488,0.000105369865,0.000013921687,0.00003549275,0.000091051144,0.000016247255,0.0033411111],"genre_scores_gemma":[0.99631953,0.00014212776,0.0022977076,0.00019982782,0.000008706023,0.0000617298,0.00016602669,0.00006599672,0.00073833606],"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9967884,0.0013726612,0.00051095086,0.00070339884,0.0004933878,0.00013131392],"domain_scores_gemma":[0.9517279,0.029690992,0.0068907663,0.00806106,0.002506566,0.0011227287],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0056276764,0.0005823394,0.0004085674,0.00041825807,0.0006586058,0.0015836391,0.0008489152,0.0008597407,0.0071090795],"category_scores_gemma":[0.02355919,0.0006840344,0.00023323389,0.0004060761,0.001864005,0.0029278772,0.0016752533,0.0010453924,0.0008556703],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0111039225,0.011562819,0.1728467,0.0026015285,0.0005605533,0.0020208335,0.07042001,0.0004931853,0.64564264,0.018815763,0.001915779,0.062016334],"study_design_scores_gemma":[0.0019728355,0.013681682,0.64087313,0.00042076924,0.0013799346,0.004834757,0.04064221,0.005062167,0.24164909,0.025275305,0.023716824,0.00049119],"about_ca_topic_score_codex":0.00081378064,"about_ca_topic_score_gemma":0.0010475682,"teacher_disagreement_score":0.0071090795,"about_ca_system_score_codex":0.00027843524,"about_ca_system_score_gemma":0.0003390258,"threshold_uncertainty_score":0.029762328},"labels":[],"label_agreement":null},{"id":"W2944326605","doi":"10.1515/9783110540253-009","title":"9. Usage-based Grammar","year":2019,"lang":"en","type":"book-chapter","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Grammar; Computer science; Linguistics; Natural language processing; Philosophy","score_opus":0.014863048297651813,"score_gpt":0.2420390182138993,"score_spread":0.2271759699162475,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2944326605","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013383703,0.0026231816,0.31722042,0.0058990642,0.0003043221,0.000106006286,0.0004207472,0.00054325623,0.65949935],"genre_scores_gemma":[0.6349619,0.004967819,0.18662986,0.0020858617,0.0010088829,0.00059722824,0.0012651293,0.0013343556,0.16714893],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.997227,0.0010376704,0.00018902955,0.0005386532,0.00084115623,0.00016643453],"domain_scores_gemma":[0.99797684,0.00085384055,0.00010947675,0.0006423313,0.0003563967,0.00006117789],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023030743,0.00049067655,0.00046283865,0.0017073187,0.0013164349,0.0043962686,0.0014213489,0.001474378,0.008293401],"category_scores_gemma":[0.0037291169,0.0003825571,0.0008701967,0.0012677001,0.009454014,0.0069480874,0.0023277903,0.0021637825,0.0027680025],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000020831806,0.0000032320181,0.000118948534,0.000017556124,0.0000018415449,0.000027047114,0.0006159198,0.00017637642,0.000120364435,0.9897314,0.0013737172,0.0078115854],"study_design_scores_gemma":[0.0000032052346,0.000007907727,0.00027383844,0.00004315886,0.0000039010474,0.0002303021,0.0002656693,0.0013239316,0.00036922895,0.8866873,0.11078267,0.000008773633],"about_ca_topic_score_codex":0.0021063257,"about_ca_topic_score_gemma":0.0015778901,"teacher_disagreement_score":0.008293401,"about_ca_system_score_codex":0.0022891827,"about_ca_system_score_gemma":0.0014896543,"threshold_uncertainty_score":0.027744174},"labels":[],"label_agreement":null},{"id":"W294494887","doi":"10.63317/3f5rfz78d85e","title":"Definition patterns for predicative terms in specialized lexical resources","year":2014,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Predicative expression; Computer science; Relation (database); Set (abstract data type); Natural language processing; Lexical item; Function (biology); Information retrieval; Artificial intelligence; Linguistics; Data mining; Programming language","score_opus":0.021231998813719163,"score_gpt":0.2798542439077047,"score_spread":0.2586222450939855,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W294494887","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.049583036,0.00036562397,0.9334031,0.00040014644,0.000087721055,0.00060038,0.0013164685,0.0023351992,0.011908414],"genre_scores_gemma":[0.19490789,0.00033433654,0.7941069,0.00023003848,0.000036845704,0.0006432127,0.0032999262,0.0013955581,0.005045289],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9945952,0.0018438321,0.0011179436,0.0010161509,0.0011629841,0.0002638325],"domain_scores_gemma":[0.9921948,0.0044211484,0.0008485258,0.0013444899,0.0010219924,0.00016899042],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0038042683,0.00075844844,0.00092056976,0.0041660722,0.0013071402,0.0031586902,0.0019325946,0.0012091323,0.00581335],"category_scores_gemma":[0.010317873,0.0009731213,0.001444231,0.0038097948,0.0023321584,0.0067102453,0.0027593945,0.0018348554,0.0018944994],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022371157,0.00016840061,0.010489301,0.001117523,0.00012845584,0.0039725285,0.011846141,0.004089515,0.022615923,0.77202,0.009809699,0.16351876],"study_design_scores_gemma":[0.00009600543,0.00018320716,0.00629466,0.000964951,0.00022434874,0.004835269,0.005337213,0.05181736,0.044751126,0.6166374,0.26863295,0.00022561956],"about_ca_topic_score_codex":0.0012223696,"about_ca_topic_score_gemma":0.0031867067,"teacher_disagreement_score":0.00581335,"about_ca_system_score_codex":0.0011993605,"about_ca_system_score_gemma":0.001279748,"threshold_uncertainty_score":0.02011913},"labels":[],"label_agreement":null},{"id":"W2945187830","doi":"10.4000/geolinguistique.306","title":"Automatic Documentation of Faetar’s [i]: A Methodology for Discovering Vowel Space Using Artificial Neural Networks","year":2018,"lang":"fr","type":"article","venue":"Géolinguistique","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; Wilfrid Laurier University","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Vowel; Space (punctuation); Categorization; Artificial neural network; Measure (data warehouse); Heuristic; Natural language; Phone; Variation (astronomy); Spoken language; Speech recognition; Linguistics","score_opus":0.09338856956461265,"score_gpt":0.39688587731680086,"score_spread":0.3034973077521882,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2945187830","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.21827629,0.00069590285,0.75476074,0.0003097964,0.00016124532,0.00023581444,0.0056158146,0.009353918,0.0105904555],"genre_scores_gemma":[0.4632094,0.00030803448,0.5211092,0.000067057255,0.000048842405,0.00021785841,0.008380755,0.00078190677,0.0058769006],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99924064,0.00012098434,0.00005614615,0.000365508,0.00014819311,0.00006860667],"domain_scores_gemma":[0.9987685,0.00042875088,0.00017089266,0.00026769497,0.00031239577,0.00005171022],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008659612,0.00061525166,0.00037285697,0.0020880806,0.00069812324,0.0011547545,0.0010550573,0.0006083809,0.0020251116],"category_scores_gemma":[0.0038843013,0.0003955863,0.00037389508,0.0015865624,0.0006261745,0.001084279,0.0008951059,0.0010784669,0.0018363417],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025716613,0.0001191955,0.026733538,0.00033486978,0.00013120976,0.0003610487,0.0020029207,0.019047216,0.058605466,0.006611782,0.007271494,0.87852407],"study_design_scores_gemma":[0.00004854135,0.00029727299,0.11029705,0.00021733715,0.000108975306,0.0014176489,0.002492395,0.7077816,0.09081093,0.019221652,0.06707765,0.0002289379],"about_ca_topic_score_codex":0.008380664,"about_ca_topic_score_gemma":0.013970239,"teacher_disagreement_score":0.008380664,"about_ca_system_score_codex":0.00044439902,"about_ca_system_score_gemma":0.0007421379,"threshold_uncertainty_score":0.01666373},"labels":[],"label_agreement":null},{"id":"W2945479974","doi":"10.1007/978-3-030-18305-9_2","title":"Weakly Supervised, Data-Driven Acquisition of Rules for Open Information Extraction","year":2019,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Relationship extraction; Lemma (botany); Rank (graph theory); Natural language processing; Information extraction; Sentence; Relation (database); Recall; Artificial intelligence; Precision and recall; Quality (philosophy); Sequence (biology); Data mining; Information retrieval","score_opus":0.028140804625840177,"score_gpt":0.30427004187589934,"score_spread":0.27612923725005917,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2945479974","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.036108147,0.00037869983,0.9522317,0.00032229495,0.00006435131,0.00030161903,0.0015852782,0.0057262653,0.0032815991],"genre_scores_gemma":[0.19848657,0.0002104475,0.7882479,0.00022146034,0.00008303274,0.00036801992,0.008600136,0.00053887995,0.0032436352],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99595416,0.001087719,0.00048433148,0.0009776085,0.0012303317,0.0002658429],"domain_scores_gemma":[0.9799645,0.011748418,0.0008574306,0.0033144874,0.0037188123,0.00039634656],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003502177,0.0007938516,0.0012541385,0.0023812866,0.00088668143,0.0031440642,0.0030510575,0.0013914753,0.0030513564],"category_scores_gemma":[0.017893648,0.00072595844,0.0012981279,0.0018338024,0.0010783781,0.0039006283,0.003218316,0.0028552604,0.0039167036],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00046086547,0.00059949956,0.0059566563,0.0005133061,0.0001431706,0.000379139,0.00049428374,0.019978499,0.046880253,0.011667345,0.012709072,0.90021783],"study_design_scores_gemma":[0.00004941704,0.0002055612,0.0026436814,0.00013114265,0.00008575041,0.00043990317,0.00033520625,0.85596174,0.08002192,0.04836732,0.011695793,0.00006249462],"about_ca_topic_score_codex":0.0015934212,"about_ca_topic_score_gemma":0.0055693095,"teacher_disagreement_score":0.003502177,"about_ca_system_score_codex":0.0008235035,"about_ca_system_score_gemma":0.0023242591,"threshold_uncertainty_score":0.018521488},"labels":[],"label_agreement":null},{"id":"W2946211941","doi":"10.5539/mas.v13n6p44","title":"Retrieving Arabic Textual Documents Based on Queries Written in Bahraini Slang Language","year":2019,"lang":"en","type":"article","venue":"Modern Applied Science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Slang; Computer science; Natural language processing; Arabic; Linguistics; Artificial intelligence; Process (computing); Programming language","score_opus":0.006947019329891353,"score_gpt":0.25599663932482297,"score_spread":0.2490496199949316,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2946211941","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7685405,0.006938279,0.13630164,0.0030665644,0.00045008629,0.0018418523,0.01724114,0.024609609,0.04101028],"genre_scores_gemma":[0.7436868,0.0025443349,0.21859676,0.0006098983,0.00022118352,0.00036678696,0.015968096,0.0003426955,0.017663425],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9993406,0.00014006482,0.00011641585,0.00012543098,0.00021206334,0.00006541414],"domain_scores_gemma":[0.9987633,0.000514936,0.00009006875,0.00009363489,0.00047659955,0.00006152075],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005390843,0.0009036082,0.0008664428,0.0032840518,0.00077598385,0.0018656339,0.00060548075,0.00067981216,0.005626177],"category_scores_gemma":[0.0035702733,0.00016917057,0.0004459369,0.0023429953,0.00027128364,0.0020516124,0.00048187407,0.00032894753,0.004575082],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0025414156,0.0007671677,0.01309849,0.0037601457,0.00014033899,0.0028286208,0.003980072,0.004795155,0.24328764,0.0032002733,0.05952075,0.66207993],"study_design_scores_gemma":[0.00041689098,0.0019137248,0.06156351,0.00056034967,0.00066728226,0.006766764,0.014361034,0.33312634,0.38884264,0.005752632,0.18567124,0.00035765956],"about_ca_topic_score_codex":0.0067720027,"about_ca_topic_score_gemma":0.0061472673,"teacher_disagreement_score":0.0067720027,"about_ca_system_score_codex":0.00072432036,"about_ca_system_score_gemma":0.00084252754,"threshold_uncertainty_score":0.018821418},"labels":[],"label_agreement":null},{"id":"W2946378017","doi":"10.1007/978-3-030-18305-9_16","title":"Identifying Misaligned Spans in Parallel Corpora Using Change Point Detection","year":2019,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada; Carleton University","funders":"","keywords":"Computer science; Parallel corpora; Machine translation; Artificial intelligence; Change detection; Identification (biology); Point (geometry); Natural language processing; Translation (biology); Quality (philosophy); German; Speech recognition; Linguistics","score_opus":0.04631436178743084,"score_gpt":0.2887615899639254,"score_spread":0.24244722817649458,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2946378017","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.30359,0.006108159,0.63958234,0.0007439799,0.0015951936,0.00045998587,0.012642536,0.02253887,0.012739017],"genre_scores_gemma":[0.46932912,0.0021168087,0.47177002,0.00024327343,0.0006893263,0.0005210665,0.04223698,0.0032470413,0.009846328],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99661714,0.00050675625,0.00033831358,0.0014705337,0.0008642337,0.00020306453],"domain_scores_gemma":[0.99217635,0.0031173944,0.00081289065,0.0016744985,0.0020203183,0.00019857408],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002445788,0.001413729,0.0012910499,0.0049468847,0.0013887944,0.0025598006,0.0015325046,0.0017899437,0.00516665],"category_scores_gemma":[0.009459597,0.000921885,0.0008538305,0.0076848506,0.0007698628,0.004229848,0.002295698,0.0017441806,0.0064761057],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001339386,0.0003701081,0.012635335,0.001239795,0.00029427133,0.0020982225,0.0013704315,0.01070247,0.13158982,0.005211669,0.0272516,0.80589694],"study_design_scores_gemma":[0.00025937357,0.0012292147,0.07444388,0.0004157145,0.00093846035,0.0072196787,0.0032486524,0.5257438,0.19608001,0.037925154,0.15214433,0.0003517389],"about_ca_topic_score_codex":0.002554127,"about_ca_topic_score_gemma":0.0040473486,"teacher_disagreement_score":0.00516665,"about_ca_system_score_codex":0.0005154624,"about_ca_system_score_gemma":0.00095384795,"threshold_uncertainty_score":0.017284155},"labels":[],"label_agreement":null},{"id":"W2946534924","doi":"10.1017/s1351324919000135","title":"NLP commercialisation in the last 25 years","year":2019,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Focus (optics); Quarter (Canadian coin); Point (geometry); Artificial intelligence; Natural (archaeology); Natural language processing; History","score_opus":0.004465130453185606,"score_gpt":0.23091334490317533,"score_spread":0.22644821444998972,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2946534924","genre_codex":"other","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09056501,0.28123158,0.019730018,0.24481165,0.030755844,0.00026436866,0.0043603573,0.0039471723,0.32433403],"genre_scores_gemma":[0.5040691,0.17101409,0.043579295,0.052458603,0.01940721,0.00040454612,0.013507564,0.0044859066,0.19107372],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9913591,0.0014154625,0.0006569955,0.0014739536,0.0044990126,0.0005955768],"domain_scores_gemma":[0.94728225,0.01798301,0.00483877,0.003242474,0.021938076,0.004715478],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.017618125,0.00053683383,0.0005158945,0.004283735,0.0018050934,0.012217952,0.0016752691,0.0032287717,0.017646514],"category_scores_gemma":[0.041836906,0.00040417176,0.0006015472,0.0052010226,0.0035602094,0.011602894,0.004659221,0.003371062,0.005631595],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00034321073,0.00018818326,0.003363188,0.0018992937,0.00006387653,0.0011780526,0.0018700064,0.0007152286,0.005005743,0.13024175,0.29538655,0.55974483],"study_design_scores_gemma":[0.000012347139,0.00004129281,0.003075311,0.0005107394,0.00001249894,0.00043042089,0.0006200393,0.0006243361,0.0015141311,0.006982847,0.9861501,0.000025917629],"about_ca_topic_score_codex":0.0025811594,"about_ca_topic_score_gemma":0.0022252537,"teacher_disagreement_score":0.017646514,"about_ca_system_score_codex":0.005496271,"about_ca_system_score_gemma":0.004207455,"threshold_uncertainty_score":0.09317464},"labels":[],"label_agreement":null},{"id":"W2946640399","doi":"10.5430/ijhe.v8n3p103","title":"Building a Knowledge Base for QA System by Linking Korean Vocabulary and Wikipedia","year":2019,"lang":"en","type":"article","venue":"International Journal of Higher Education","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Workbench; Vocabulary; Computer science; Knowledge base; Process (computing); Word (group theory); Controlled vocabulary; Base (topology); World Wide Web; Natural language processing; Information retrieval; Knowledge management; Artificial intelligence; Linguistics; Programming language; Visualization","score_opus":0.008150376768133984,"score_gpt":0.3103966235613997,"score_spread":0.3022462467932657,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2946640399","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07254997,0.00201604,0.81971633,0.0020186615,0.0005390304,0.0023196603,0.045393765,0.026069751,0.029376848],"genre_scores_gemma":[0.15049212,0.0015434031,0.7499249,0.0005193296,0.00007149986,0.0011110256,0.08841135,0.0014626777,0.0064637447],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9990688,0.0002246985,0.00018074653,0.0002834018,0.00017346058,0.00006888503],"domain_scores_gemma":[0.9969798,0.001057232,0.00021091773,0.00044091197,0.0011949177,0.000116207535],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012754258,0.000911402,0.0007111101,0.0074454932,0.0013894255,0.00238221,0.0016846675,0.0009079121,0.010693071],"category_scores_gemma":[0.006747795,0.0009848722,0.001014268,0.0043162717,0.00041438424,0.0061471015,0.002105994,0.0014780245,0.005866153],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037812055,0.00081801176,0.01589145,0.003287196,0.00037367293,0.004614467,0.0046241363,0.02022072,0.052685574,0.022554135,0.11797088,0.7565817],"study_design_scores_gemma":[0.0003050755,0.00034927396,0.01692712,0.0021411302,0.0014359234,0.0036991474,0.006786351,0.19078016,0.081873424,0.041326445,0.6539152,0.00046077417],"about_ca_topic_score_codex":0.0150646875,"about_ca_topic_score_gemma":0.014487571,"teacher_disagreement_score":0.0150646875,"about_ca_system_score_codex":0.00088067044,"about_ca_system_score_gemma":0.0020746873,"threshold_uncertainty_score":0.035771906},"labels":[],"label_agreement":null},{"id":"W2946836897","doi":"10.5539/ijel.v9n3p357","title":"Analysis of Lecxico-Semantic Relations of Punjabi Shahmukhi Nouns: A Corpus Based Study","year":2019,"lang":"en","type":"article","venue":"International Journal of English Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"WordNet; Noun; Computer science; Natural language processing; Artificial intelligence; Linguistics; Part of speech; Proper noun","score_opus":0.011242682127095189,"score_gpt":0.29335794845679714,"score_spread":0.28211526632970196,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2946836897","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.97906643,0.0015368517,0.002800656,0.00032891883,0.0000628828,0.00036700265,0.0042338176,0.00007815569,0.011525273],"genre_scores_gemma":[0.97489035,0.0015082391,0.010470163,0.00015041015,0.000042055934,0.00079981587,0.0096425135,0.000069462476,0.0024269575],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99805945,0.0005388271,0.00034615168,0.00041913684,0.0005367343,0.0000997104],"domain_scores_gemma":[0.9915086,0.004858728,0.0011470803,0.00067188474,0.0016216073,0.00019201197],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001757407,0.0002664141,0.00044820277,0.0067953044,0.0024779143,0.0016042825,0.0005679116,0.0004301202,0.0033872358],"category_scores_gemma":[0.008751144,0.00028944798,0.00026735457,0.009776118,0.0020290588,0.001807862,0.0018056502,0.00072062376,0.00066679827],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005856385,0.0008391281,0.34614587,0.010962386,0.00022880825,0.01126324,0.22542492,0.0010871973,0.048016697,0.023143614,0.023031317,0.30927122],"study_design_scores_gemma":[0.000035419995,0.00014358402,0.76107615,0.000814315,0.00015576315,0.004486156,0.093421094,0.0026508602,0.0081308745,0.0021414626,0.12684068,0.00010364531],"about_ca_topic_score_codex":0.010011699,"about_ca_topic_score_gemma":0.018954538,"teacher_disagreement_score":0.010011699,"about_ca_system_score_codex":0.0016662048,"about_ca_system_score_gemma":0.0020285635,"threshold_uncertainty_score":0.019906878},"labels":[],"label_agreement":null},{"id":"W2947684374","doi":"","title":"MRG_UWaterloo Participation in the TREC 2018 Common Core Track.","year":2018,"lang":"en","type":"article","venue":"Text REtrieval Conference","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Track (disk drive); Core (optical fiber); Computer science; Artificial intelligence; Telecommunications","score_opus":0.05982771642909706,"score_gpt":0.3453437846347422,"score_spread":0.2855160682056451,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2947684374","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023023982,0.010369802,0.026930664,0.088229455,0.04369895,0.006032956,0.32329115,0.016033303,0.46238974],"genre_scores_gemma":[0.025256116,0.0015643587,0.01768599,0.007899152,0.004423045,0.0013983299,0.16101788,0.0022941763,0.7784609],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99408543,0.0013152816,0.00015628838,0.0009038065,0.0024494152,0.0010898152],"domain_scores_gemma":[0.98279285,0.0008798252,0.00029601195,0.0010071706,0.009616914,0.005407167],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009450942,0.0019130756,0.0029234777,0.0035478869,0.004633829,0.00504433,0.0031975126,0.0027150204,0.1578649],"category_scores_gemma":[0.010087335,0.000532568,0.00079963007,0.0033580384,0.0011785137,0.0044510053,0.0045098206,0.0020051568,0.087831736],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011348235,0.00009309315,0.00014939225,0.00007708998,0.000008310027,0.00002335787,0.00006105366,0.00007658349,0.0009973526,0.0005171809,0.98388827,0.013994716],"study_design_scores_gemma":[0.00014040062,0.000101703794,0.0016921221,0.00005126684,0.000021126849,0.000031951837,0.0003382834,0.0011581855,0.002017072,0.0012794435,0.9931356,0.000032803204],"about_ca_topic_score_codex":0.20594339,"about_ca_topic_score_gemma":0.39558187,"teacher_disagreement_score":0.20594339,"about_ca_system_score_codex":0.008546774,"about_ca_system_score_gemma":0.015822342,"threshold_uncertainty_score":0.5281107},"labels":[],"label_agreement":null},{"id":"W2949008176","doi":"","title":"Demonstration of the German to English METIS-II MT system.","year":2007,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Metis; German; Computer science; History; World Wide Web; Archaeology","score_opus":0.00698526190228118,"score_gpt":0.26354323462715346,"score_spread":0.2565579727248723,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2949008176","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.36625904,0.0021337473,0.20989925,0.009141765,0.0029083735,0.001424184,0.03477233,0.107538216,0.26592308],"genre_scores_gemma":[0.799487,0.0004911806,0.12808637,0.0008838684,0.00020428991,0.0004064623,0.016270569,0.0015989966,0.05257114],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9996488,0.00010995821,0.000031685056,0.00006380507,0.00010006082,0.000045585864],"domain_scores_gemma":[0.99921477,0.00027669195,0.000023272516,0.00012279466,0.00025015898,0.000112294794],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007732158,0.00051696115,0.0003616547,0.0003902445,0.00074682134,0.0010295371,0.00070926914,0.00088682165,0.025938073],"category_scores_gemma":[0.0018583551,0.00024132367,0.00019478657,0.0004808531,0.00032878984,0.0012066563,0.000999916,0.00066469464,0.01112151],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0029520881,0.0007071757,0.005852341,0.001599807,0.00019044879,0.0082867695,0.0031391133,0.0049070055,0.17883345,0.029927408,0.45330352,0.31030095],"study_design_scores_gemma":[0.0015366118,0.001367837,0.02828102,0.00032993034,0.00021921362,0.0091080405,0.0032762033,0.12357373,0.23493811,0.019020144,0.5780875,0.0002616506],"about_ca_topic_score_codex":0.009794752,"about_ca_topic_score_gemma":0.013776214,"teacher_disagreement_score":0.025938073,"about_ca_system_score_codex":0.0003433543,"about_ca_system_score_gemma":0.0005978987,"threshold_uncertainty_score":0.08677149},"labels":[],"label_agreement":null},{"id":"W2949037946","doi":"","title":"Text linguistics and text revision: an alliance approach","year":2018,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Linguistic Association","funders":"","keywords":"Alliance; Linguistics; Text linguistics; Computer science; Corpus linguistics; Natural language processing; History; Philosophy","score_opus":0.022307410652867107,"score_gpt":0.3007855133585948,"score_spread":0.2784781027057277,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2949037946","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00827576,0.0013313548,0.9571739,0.0064711994,0.0004240348,0.0002054174,0.00014748509,0.0012226413,0.024748165],"genre_scores_gemma":[0.3935978,0.0024492536,0.56245047,0.0020322162,0.001720736,0.0007694441,0.0007367141,0.0011226924,0.035120647],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9891922,0.006102546,0.0008441004,0.00138869,0.0021355434,0.00033697026],"domain_scores_gemma":[0.95820546,0.02382186,0.0021344125,0.008581429,0.005928853,0.0013279552],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011308811,0.00064578245,0.0010926176,0.00447279,0.0021234127,0.007251108,0.0036188029,0.0027969915,0.010995988],"category_scores_gemma":[0.034802284,0.00076392613,0.0016438384,0.0038732118,0.0045572384,0.018482067,0.00857136,0.0033014384,0.0029230968],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019538231,0.00023499937,0.0017450561,0.00044869818,0.00012842283,0.00037673156,0.00453334,0.0031540412,0.003927867,0.73039204,0.010661852,0.24420154],"study_design_scores_gemma":[0.00008065679,0.000115260955,0.0007366496,0.0001797507,0.00014611677,0.0004361723,0.0016862351,0.046125,0.005729106,0.8791963,0.065496095,0.000072553645],"about_ca_topic_score_codex":0.0011403533,"about_ca_topic_score_gemma":0.00083768804,"teacher_disagreement_score":0.011308811,"about_ca_system_score_codex":0.0012019291,"about_ca_system_score_gemma":0.0031529283,"threshold_uncertainty_score":0.05980736},"labels":[],"label_agreement":null},{"id":"W2949266915","doi":"10.48550/arxiv.1808.03967","title":"Augmenting word2vec with latent Dirichlet allocation within a clinical application","year":2018,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Latent Dirichlet allocation; Word2vec; Dirichlet distribution; Word (group theory); Embedding; Topic model; Computer science; Word embedding; Artificial intelligence; Natural language processing; Mathematics","score_opus":0.06291036633874258,"score_gpt":0.23809619448385674,"score_spread":0.17518582814511416,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2949266915","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08065404,0.0024553258,0.8997421,0.0022971018,0.0005910584,0.00024171155,0.0031401848,0.006240377,0.0046381713],"genre_scores_gemma":[0.67822707,0.0011976699,0.30471238,0.0009605755,0.0005121872,0.00037328355,0.0072614755,0.00053696986,0.0062183673],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.998005,0.0011832373,0.00009298081,0.00040528312,0.00019550107,0.00011805181],"domain_scores_gemma":[0.99759465,0.0016488913,0.000083547115,0.00027187605,0.0003030371,0.00009799754],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020317826,0.0012212913,0.00074376207,0.001355645,0.00039598075,0.0008949242,0.00080543826,0.0010415855,0.0025541638],"category_scores_gemma":[0.006207993,0.00043322556,0.0008203478,0.0012141834,0.00051085406,0.0012989653,0.0013675146,0.0014546306,0.002787998],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001082523,0.00067206507,0.01436635,0.00040590356,0.00043479042,0.00041254537,0.0004702418,0.08504161,0.016292773,0.007430492,0.024998805,0.84839195],"study_design_scores_gemma":[0.000078360834,0.00027464266,0.0034642264,0.00006268155,0.00013936317,0.0005561944,0.00015998543,0.9388571,0.011003149,0.031971835,0.013356261,0.00007625656],"about_ca_topic_score_codex":0.0049522673,"about_ca_topic_score_gemma":0.009686399,"teacher_disagreement_score":0.0049522673,"about_ca_system_score_codex":0.00047344776,"about_ca_system_score_gemma":0.0013809031,"threshold_uncertainty_score":0.010745227},"labels":[],"label_agreement":null},{"id":"W2949495625","doi":"10.48550/arxiv.1302.6777","title":"Ending-based Strategies for Part-of-speech Tagging","year":2013,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Saskatchewan","funders":"","keywords":"Word (group theory); Punctuation; Computer science; Part of speech; Natural language processing; Artificial intelligence; Part-of-speech tagging; Set (abstract data type); Word error rate; Speech recognition; Mathematics","score_opus":0.07083707437817484,"score_gpt":0.22118117372502302,"score_spread":0.15034409934684817,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2949495625","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023218317,0.00019368304,0.9695386,0.00006494756,0.00011264919,0.00009833237,0.00033197986,0.0044590067,0.0019824253],"genre_scores_gemma":[0.23411451,0.00020813578,0.7575959,0.00018199378,0.00008613633,0.0002800076,0.0018755709,0.0018065698,0.003851179],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99652904,0.0014494495,0.00043635897,0.000814285,0.00058839936,0.00018245676],"domain_scores_gemma":[0.9836162,0.0081004,0.00088186114,0.0046031456,0.0025032347,0.0002951995],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0052730665,0.0014040819,0.0009765215,0.002656875,0.00091780606,0.0017966289,0.0018748402,0.0013210915,0.0034836824],"category_scores_gemma":[0.014162629,0.0007727992,0.0009400151,0.0018633796,0.0010666106,0.0036821598,0.002465805,0.0018133103,0.008434398],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009626666,0.00042129366,0.0145486295,0.0006992657,0.00023368359,0.0008469358,0.001807027,0.040967345,0.18071511,0.03785898,0.007628218,0.7133109],"study_design_scores_gemma":[0.00008447823,0.0007888352,0.005340042,0.00013440376,0.00021174514,0.0021709725,0.00052184495,0.5624519,0.2703042,0.12055956,0.03714457,0.00028738377],"about_ca_topic_score_codex":0.00033010394,"about_ca_topic_score_gemma":0.0010025086,"teacher_disagreement_score":0.0052730665,"about_ca_system_score_codex":0.00034855626,"about_ca_system_score_gemma":0.00078654464,"threshold_uncertainty_score":0.027886927},"labels":[],"label_agreement":null},{"id":"W2949637934","doi":"10.5087/dad.2019.104","title":"How compatible are our discourse annotation frameworks? Insights from mapping RST-DT and PDTB annotations","year":2019,"lang":"en","type":"article","venue":"Dialogue & Discourse","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Treebank; Annotation; Computer science; Rhetorical question; Relation (database); Natural language processing; Resource (disambiguation); Artificial intelligence; Linguistics; Data mining","score_opus":0.014211910558336502,"score_gpt":0.2761876015369167,"score_spread":0.2619756909785802,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2949637934","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13718052,0.0020885887,0.78051645,0.010407435,0.0003319677,0.00045412005,0.0032125236,0.0024190592,0.06338934],"genre_scores_gemma":[0.64071625,0.00092358066,0.3487906,0.0005578442,0.00008878385,0.0006479772,0.0032396424,0.0013768767,0.0036584628],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.94099754,0.042870305,0.0027900527,0.0053045335,0.006510457,0.0015271623],"domain_scores_gemma":[0.87580585,0.08424418,0.005309265,0.021572884,0.011502049,0.0015657567],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04872491,0.0010551382,0.0011421856,0.007927607,0.004958651,0.014604604,0.0030317097,0.002958879,0.00668972],"category_scores_gemma":[0.156378,0.0014054384,0.0009851481,0.0076516974,0.0055778516,0.020276941,0.0081335055,0.004760646,0.0020106654],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008775321,0.00022683134,0.024021203,0.0011699607,0.0002074431,0.0007145589,0.07549348,0.012425014,0.018535644,0.56073034,0.009149878,0.29644817],"study_design_scores_gemma":[0.00011726269,0.00017986495,0.027751474,0.0015058027,0.0002448594,0.0008237995,0.041020457,0.15014313,0.023222199,0.62093675,0.1336604,0.00039404354],"about_ca_topic_score_codex":0.015680067,"about_ca_topic_score_gemma":0.014994276,"teacher_disagreement_score":0.04872491,"about_ca_system_score_codex":0.005067472,"about_ca_system_score_gemma":0.0057305642,"threshold_uncertainty_score":0.257685},"labels":[],"label_agreement":null},{"id":"W2950333634","doi":"10.18653/v1/p19-1173","title":"Adversarial Multitask Learning for Joint Multi-Feature and Multi-Dialect Morphological Modeling","year":2019,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"York University; New York University Abu Dhabi","keywords":"Adversarial system; Computer science; Artificial intelligence; Transfer of learning; Modern Standard Arabic; Multi-task learning; Natural language processing; Context (archaeology); Feature (linguistics); Scheme (mathematics); Machine learning; Arabic; Linguistics; Task (project management); Geography; Engineering","score_opus":0.03411566176462708,"score_gpt":0.28490114798101907,"score_spread":0.25078548621639196,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2950333634","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.030005403,0.0005319394,0.9651733,0.00069508137,0.00009203617,0.000059669022,0.00027783003,0.0009197682,0.0022448825],"genre_scores_gemma":[0.8397198,0.00035649678,0.14660256,0.00065489014,0.00018384977,0.00026875304,0.001204826,0.00021979355,0.010789018],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989992,0.00050982455,0.000043943106,0.00022988407,0.000108746855,0.00010843638],"domain_scores_gemma":[0.99610585,0.0028407644,0.00024042197,0.00041818758,0.00026667086,0.00012800378],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029717234,0.0013449704,0.0012237448,0.000646907,0.00054261636,0.0010319882,0.0020973093,0.001664745,0.0030331826],"category_scores_gemma":[0.005867062,0.00058990106,0.0011778533,0.0007632585,0.0012975668,0.0015575968,0.0020871835,0.0031914162,0.0012152548],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017085526,0.000099786346,0.0012992562,0.00007227358,0.00008677627,0.00014571678,0.00008493394,0.93530226,0.0015476237,0.012500928,0.003360138,0.045329507],"study_design_scores_gemma":[0.0000039682154,0.000010027753,0.000075162185,0.0000032398814,0.0000037033542,0.000009780039,0.000004331791,0.99387795,0.00022156359,0.0055374736,0.00024896624,0.0000038120195],"about_ca_topic_score_codex":0.004014541,"about_ca_topic_score_gemma":0.004863488,"teacher_disagreement_score":0.004014541,"about_ca_system_score_codex":0.001108285,"about_ca_system_score_gemma":0.00077832775,"threshold_uncertainty_score":0.015716136},"labels":[],"label_agreement":null},{"id":"W2950582891","doi":"10.48550/arxiv.1703.07713","title":"Hierarchical RNN with Static Sentence-Level Attention for Text-Based Speaker Change Detection","year":2017,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"National Key Research and Development Program of China; National Natural Science Foundation of China","keywords":"Computer science; Dialog box; Recurrent neural network; Sentence; Feature (linguistics); Speech recognition; Task (project management); Artificial intelligence; Matching (statistics); Point (geometry); Artificial neural network; Natural language processing; Linguistics","score_opus":0.1273654846581534,"score_gpt":0.23454132298214256,"score_spread":0.10717583832398916,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2950582891","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11263324,0.0014713906,0.87551117,0.0004175081,0.0002518949,0.00013246766,0.0005023647,0.0055099637,0.0035699292],"genre_scores_gemma":[0.7965047,0.00035698968,0.19574219,0.00032928918,0.00015822265,0.00015156975,0.0011104093,0.00020625009,0.005440487],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995394,0.00011059311,0.000026644198,0.0001939619,0.00007226472,0.000057175475],"domain_scores_gemma":[0.99927527,0.00035453163,0.000073558775,0.00007094711,0.00018927026,0.000036370122],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00086040533,0.0009762783,0.0005671962,0.00061094365,0.00029750145,0.0003751926,0.001065869,0.00071645185,0.001924702],"category_scores_gemma":[0.002284735,0.00029621794,0.0005505266,0.00046733537,0.0002795159,0.0009995411,0.0007185425,0.0010845719,0.00096004223],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005365667,0.0003545105,0.0027117364,0.00023373462,0.00015498557,0.0003543138,0.00030857915,0.19781965,0.07819709,0.0038755191,0.0077738273,0.7076795],"study_design_scores_gemma":[0.00000677836,0.000045126082,0.0006057123,0.0000061632613,0.000028205886,0.000028939114,0.000014110403,0.9919133,0.0053749215,0.0014438549,0.0005240819,0.000008687282],"about_ca_topic_score_codex":0.010504108,"about_ca_topic_score_gemma":0.015184071,"teacher_disagreement_score":0.010504108,"about_ca_system_score_codex":0.0007021275,"about_ca_system_score_gemma":0.0007771401,"threshold_uncertainty_score":0.020885885},"labels":[],"label_agreement":null},{"id":"W2950682695","doi":"","title":"An Autoencoder Approach to Learning Bilingual Word Representations","year":2014,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":114,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Sherbrooke","funders":"","keywords":"Computer science; Autoencoder; Artificial intelligence; Natural language processing; Word (group theory); Classifier (UML); German; Task (project management); Point (geometry); Language model; Deep learning; Linguistics; Mathematics","score_opus":0.05856459895297245,"score_gpt":0.23931282255367603,"score_spread":0.18074822360070358,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2950682695","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012935359,0.0002908827,0.9840385,0.00015489201,0.000053939257,0.000038607846,0.00009967921,0.0011272305,0.0012610215],"genre_scores_gemma":[0.40460172,0.0007700683,0.5813785,0.00045747994,0.00019302328,0.00032722487,0.0017503264,0.00031942865,0.010202156],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992853,0.00023684102,0.000047101617,0.00023585469,0.0001265455,0.00006844127],"domain_scores_gemma":[0.9988937,0.0005290575,0.00009675627,0.00022622623,0.00021926557,0.000035025154],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014650564,0.0011777355,0.0008971483,0.0010522447,0.00039122172,0.0007811241,0.0011641785,0.0010848085,0.0031067228],"category_scores_gemma":[0.0034219138,0.0006644191,0.0009584287,0.0011630706,0.00056754175,0.0020602352,0.0015178612,0.0022111011,0.0015735308],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002479068,0.0003867817,0.0019208764,0.00021124845,0.00039525243,0.00016453443,0.00023967898,0.29031488,0.018578935,0.03039545,0.005611009,0.6515335],"study_design_scores_gemma":[0.0000151641325,0.00006211344,0.00034118618,0.000015949905,0.000030013902,0.000049717055,0.000025969966,0.9778595,0.0050892136,0.014528894,0.001970071,0.000012199256],"about_ca_topic_score_codex":0.0026801315,"about_ca_topic_score_gemma":0.004364404,"teacher_disagreement_score":0.0031067228,"about_ca_system_score_codex":0.0006615956,"about_ca_system_score_gemma":0.0009874405,"threshold_uncertainty_score":0.010392964},"labels":[],"label_agreement":null},{"id":"W2950992105","doi":"10.48550/arxiv.1809.01446","title":"Free as in Free Word Order: An Energy Based Model for Word Segmentation and Morphological Tagging in Sanskrit","year":2018,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Natural language processing; Parsing; Sanskrit; Segmentation; Artificial intelligence; Sentence; Word (group theory); Text segmentation; Task (project management); Graph; Context (archaeology); Mathematics; Linguistics; Theoretical computer science","score_opus":0.055034760049493585,"score_gpt":0.2268021652373592,"score_spread":0.1717674051878656,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2950992105","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.43174896,0.0010708679,0.5527831,0.002258275,0.00020723957,0.00008087625,0.0013146235,0.003485185,0.007050877],"genre_scores_gemma":[0.9212789,0.0002930642,0.06531496,0.00027343538,0.00006914703,0.0001116803,0.0015344882,0.0003641369,0.010760142],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99980813,0.00005660801,0.000010047989,0.00008082904,0.000018933823,0.000025518952],"domain_scores_gemma":[0.9996401,0.00020804642,0.000034440312,0.000036035595,0.000056779874,0.000024677805],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004997791,0.00080049696,0.0007140029,0.0005670781,0.00049320736,0.0009399416,0.0011981067,0.0014763015,0.0026433745],"category_scores_gemma":[0.0010441183,0.0005966031,0.00082482945,0.000537596,0.0005797835,0.0018037811,0.0005360439,0.0012576758,0.001089386],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035330624,0.00014755859,0.0035192913,0.00007980393,0.00007120431,0.0002711374,0.00022553492,0.90191317,0.0074331374,0.015903862,0.00400872,0.066073276],"study_design_scores_gemma":[0.0000036257093,0.000009655059,0.0002561295,0.0000026997227,0.0000052298815,0.000013713697,0.0000065283125,0.99451387,0.00032713258,0.0046483586,0.00020818156,0.000005016881],"about_ca_topic_score_codex":0.017802706,"about_ca_topic_score_gemma":0.04099388,"teacher_disagreement_score":0.017802706,"about_ca_system_score_codex":0.00082906603,"about_ca_system_score_gemma":0.00078660535,"threshold_uncertainty_score":0.035398185},"labels":[],"label_agreement":null},{"id":"W2951075934","doi":"10.48550/arxiv.cs/0607120","title":"Expressing Implicit Semantic Relations without Supervision","year":2006,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Ranking (information retrieval); Analogy; Natural language processing; Similarity (geometry); Word (group theory); Computer science; Artificial intelligence; Semantic similarity; Noun; Mathematics; Linguistics; Image (mathematics)","score_opus":0.0283576828757911,"score_gpt":0.29469778644696276,"score_spread":0.26634010357117166,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2951075934","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02255176,0.00033770208,0.96687984,0.00036303452,0.000055670676,0.00011668123,0.0016656572,0.005487608,0.0025420217],"genre_scores_gemma":[0.18407804,0.00033021162,0.8028974,0.00023703772,0.000121980265,0.00036110793,0.007838155,0.00038934214,0.0037467834],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9982414,0.00033244005,0.00013159329,0.0006787162,0.0005230904,0.00009264397],"domain_scores_gemma":[0.99568236,0.0023150272,0.00038446853,0.0009000885,0.0006290103,0.00008908502],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012895794,0.001132445,0.00094229,0.0021392044,0.0008191043,0.0012409384,0.002230961,0.001203341,0.0035884993],"category_scores_gemma":[0.008748601,0.000675923,0.0012179246,0.0028943976,0.00094547356,0.005081522,0.0015117034,0.0016704394,0.002814557],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00038298024,0.000402405,0.005788383,0.00069876347,0.00014764303,0.00034176433,0.00059590035,0.032224555,0.02672295,0.024650391,0.020864049,0.8871802],"study_design_scores_gemma":[0.000052443356,0.00011576074,0.0021042773,0.00007786639,0.000072458985,0.00042046752,0.000191888,0.8527544,0.019139268,0.10365726,0.021372188,0.000041788764],"about_ca_topic_score_codex":0.0028802317,"about_ca_topic_score_gemma":0.006812263,"teacher_disagreement_score":0.0035884993,"about_ca_system_score_codex":0.00065950834,"about_ca_system_score_gemma":0.0015344745,"threshold_uncertainty_score":0.012004733},"labels":[],"label_agreement":null},{"id":"W2951417328","doi":"10.48550/arxiv.1809.01074","title":"A Novel Neural Sequence Model with Multiple Attentions for Word Sense Disambiguation","year":2018,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Computer science; Bridging (networking); Natural language processing; Word (group theory); Sentence; Artificial intelligence; Word-sense disambiguation; Encoder; Sequence (biology); SemEval; Linguistics","score_opus":0.12058664178523112,"score_gpt":0.23494811575296173,"score_spread":0.11436147396773061,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2951417328","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023573857,0.0011703196,0.9679183,0.0011344722,0.00037948525,0.00007201193,0.0003702274,0.0017756185,0.0036056307],"genre_scores_gemma":[0.68797135,0.0013095752,0.28635156,0.0012321066,0.0003715128,0.00033940436,0.0010501123,0.0004443827,0.020929987],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99951386,0.00012660705,0.000027418426,0.00019070442,0.0000807143,0.000060725866],"domain_scores_gemma":[0.9992304,0.00042424395,0.000061440136,0.000074731826,0.00014650717,0.000062710824],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009703578,0.0009893993,0.0011037419,0.0010071213,0.00058345194,0.0010792331,0.0027711408,0.0018510618,0.0045987023],"category_scores_gemma":[0.0028879947,0.0007450595,0.0010556213,0.001356179,0.00082317815,0.0028545193,0.001636074,0.0021836513,0.0013637202],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036379945,0.00023089396,0.001430678,0.00021279564,0.00018196732,0.00044501395,0.00032149185,0.62134385,0.010389343,0.051031016,0.008223366,0.30582583],"study_design_scores_gemma":[0.0000114051545,0.00002064352,0.000071997325,0.000007338385,0.000015915848,0.000036735135,0.000007366608,0.9834715,0.000685769,0.0148711335,0.0007922671,0.000007903751],"about_ca_topic_score_codex":0.011702,"about_ca_topic_score_gemma":0.01740402,"teacher_disagreement_score":0.011702,"about_ca_system_score_codex":0.0013053468,"about_ca_system_score_gemma":0.0019648045,"threshold_uncertainty_score":0.023267746},"labels":[],"label_agreement":null},{"id":"W2951890879","doi":"10.48550/arxiv.1906.11483","title":"Morphological Irregularity Correlates with Frequency","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Predictability; Correlation; Computer science; Measure (data warehouse); Code (set theory); Natural language processing; Artificial intelligence; Linguistics; Mathematics; Statistics; Data mining","score_opus":0.041401415334015944,"score_gpt":0.1879616409647252,"score_spread":0.14656022563070925,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2951890879","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98019904,0.00024713832,0.014840603,0.00014162083,0.000010572466,0.0000091173215,0.00030481737,0.00010630373,0.0041408082],"genre_scores_gemma":[0.996874,0.000086122134,0.0022944228,0.000015188808,0.000018672044,0.0000064202236,0.00025801416,0.00004752135,0.00039969085],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99926966,0.0001666977,0.00007788867,0.00022084762,0.00019811383,0.000066794935],"domain_scores_gemma":[0.9787011,0.01062539,0.005146509,0.0035970162,0.0014913299,0.00043856143],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009386891,0.00024444924,0.00035290915,0.0027875658,0.00051071896,0.0011880293,0.00042438894,0.0004336255,0.004748908],"category_scores_gemma":[0.017834183,0.00023209676,0.00038048273,0.002951159,0.0013524592,0.0019883616,0.0012560742,0.0008291237,0.00088301266],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006554331,0.00014932164,0.7647668,0.00023255423,0.00027395025,0.00088919676,0.0030629772,0.006357421,0.040190298,0.028765522,0.001372513,0.15328401],"study_design_scores_gemma":[0.000017737464,0.00022258535,0.8603729,0.000037767135,0.00012102091,0.003639934,0.0009584545,0.035017744,0.009194585,0.08510881,0.005214385,0.00009412937],"about_ca_topic_score_codex":0.00049389113,"about_ca_topic_score_gemma":0.00045876636,"teacher_disagreement_score":0.004748908,"about_ca_system_score_codex":0.0002419817,"about_ca_system_score_gemma":0.00019690358,"threshold_uncertainty_score":0.015886664},"labels":[],"label_agreement":null},{"id":"W2952153923","doi":"10.18653/v1/p19-1121","title":"Improved Zero-shot Neural Machine Translation via Ignoring Spurious Correlations","year":2019,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":87,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Tencent; Samsung Advanced Institute of Technology; Canadian Institute for Advanced Research; Samsung; Nvidia","keywords":"Machine translation; Computer science; Spurious relationship; Translation (biology); Zero (linguistics); Artificial intelligence; Shot (pellet); Degeneracy (biology); Natural language processing; Language model; Speech recognition; Algorithm; Machine learning","score_opus":0.023876708353676494,"score_gpt":0.2818461076736029,"score_spread":0.2579693993199264,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2952153923","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05074909,0.000726602,0.93808913,0.00034835242,0.00017979463,0.000048764792,0.00018617876,0.004347478,0.005324516],"genre_scores_gemma":[0.64437586,0.00038110957,0.33905676,0.00041741272,0.00016440672,0.000120745346,0.001570526,0.00085000385,0.013063261],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9982559,0.0006201051,0.0000865746,0.00046257698,0.00042958176,0.00014514591],"domain_scores_gemma":[0.9968534,0.0013943131,0.00017818908,0.0009027105,0.00056712946,0.00010434008],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018855131,0.0011010724,0.0013319617,0.0008172667,0.00090816384,0.001435915,0.0016415297,0.0013914893,0.0041610324],"category_scores_gemma":[0.008832741,0.0006757826,0.000730362,0.0010661057,0.0011008583,0.002597222,0.0026011236,0.0016300986,0.002698575],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010820894,0.0003410922,0.0018895294,0.0004203209,0.0003139841,0.00058563484,0.00047921584,0.2110077,0.04313495,0.05187049,0.014818825,0.6740562],"study_design_scores_gemma":[0.000044977784,0.0001238013,0.00040531618,0.000027008593,0.000052611544,0.0001858074,0.000041753254,0.95747226,0.012946755,0.025711006,0.0029599094,0.000028851742],"about_ca_topic_score_codex":0.0038783078,"about_ca_topic_score_gemma":0.008113934,"teacher_disagreement_score":0.0041610324,"about_ca_system_score_codex":0.0006087297,"about_ca_system_score_gemma":0.0019547914,"threshold_uncertainty_score":0.013920009},"labels":[],"label_agreement":null},{"id":"W2952174764","doi":"10.1007/978-3-030-21920-8_22","title":"Integrating Antonyms in Fuzzy Inferential Systems via Anti-membership","year":2019,"lang":"en","type":"book-chapter","venue":"Advances in intelligent systems and computing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Intersection (aeronautics); Fuzzy logic; Computation; Computer science; Fuzzy set operations; Artificial intelligence; Fuzzy set; Mathematics; Theoretical computer science; Algorithm; Engineering","score_opus":0.015691227846161355,"score_gpt":0.2761488838056367,"score_spread":0.26045765595947534,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2952174764","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022946022,0.00030999826,0.96313566,0.00040179567,0.00016427168,0.00007056819,0.00008625541,0.00055865553,0.012326795],"genre_scores_gemma":[0.43156922,0.00035490264,0.5619555,0.00022111749,0.00024157544,0.00010307032,0.00026774846,0.00020135434,0.005085528],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99734396,0.0010376942,0.000262835,0.0004169348,0.00082895526,0.00010962613],"domain_scores_gemma":[0.9954144,0.0030866093,0.00016108702,0.0005398094,0.000708318,0.00008974663],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003108757,0.00040125664,0.0006939414,0.0014057202,0.0011875558,0.0026397752,0.0015198202,0.0008559842,0.0055041555],"category_scores_gemma":[0.012885345,0.00043799612,0.00090228324,0.0015435221,0.0017277638,0.008146033,0.002838744,0.0021551198,0.0014543107],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023572797,0.00013809124,0.0009686877,0.00026441048,0.00006184763,0.00028882475,0.0011378261,0.0070924894,0.0066849496,0.6845258,0.0032808275,0.29532063],"study_design_scores_gemma":[0.000018581522,0.00005575337,0.00030513172,0.000056821456,0.00006317305,0.0002386303,0.00022256198,0.18892668,0.007898967,0.790842,0.011338933,0.000032927288],"about_ca_topic_score_codex":0.00054231496,"about_ca_topic_score_gemma":0.0008193998,"teacher_disagreement_score":0.0055041555,"about_ca_system_score_codex":0.0005279409,"about_ca_system_score_gemma":0.00057925744,"threshold_uncertainty_score":0.018413246},"labels":[],"label_agreement":null},{"id":"W2952197960","doi":"","title":"Word Sense Disambiguation by Web Mining for Word Co-occurrence Probabilities","year":2004,"lang":"en","type":"preprint","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"University of Waterloo","keywords":"Computer science; Natural language processing; SemEval; Word (group theory); Artificial intelligence; Feature (linguistics); Task (project management); Novelty; Sample (material); Linguistics","score_opus":0.025886442496460285,"score_gpt":0.3028768655376259,"score_spread":0.2769904230411656,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2952197960","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.046011664,0.00086198474,0.93595016,0.000276797,0.00009678563,0.0002908487,0.0034442218,0.010753785,0.0023138039],"genre_scores_gemma":[0.24704275,0.00055809104,0.7432746,0.000077790566,0.00011567317,0.00038440415,0.006560951,0.0007409043,0.0012447659],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9956345,0.0009840883,0.0005195124,0.0015295232,0.0011575589,0.00017478292],"domain_scores_gemma":[0.9928543,0.0046404507,0.00067801273,0.0010269799,0.00066895306,0.00013124748],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029010908,0.0011669225,0.0017502527,0.01571041,0.0014575059,0.00257552,0.0013087696,0.0010310531,0.002819404],"category_scores_gemma":[0.015372304,0.0009895641,0.0017724675,0.010547552,0.000871575,0.0039112126,0.0019321174,0.0011671609,0.00409258],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005839277,0.00035152334,0.02747904,0.00068248215,0.00051423186,0.0013233965,0.00084632175,0.037210874,0.023127764,0.021013362,0.016566971,0.8703001],"study_design_scores_gemma":[0.00012410767,0.00012699279,0.01833693,0.00012755877,0.00025508556,0.002581104,0.0004926088,0.8065167,0.035805926,0.103253044,0.032200325,0.0001795571],"about_ca_topic_score_codex":0.00241998,"about_ca_topic_score_gemma":0.003277832,"teacher_disagreement_score":0.01571041,"about_ca_system_score_codex":0.0005536339,"about_ca_system_score_gemma":0.001390068,"threshold_uncertainty_score":0.015342653},"labels":[],"label_agreement":null},{"id":"W2952904369","doi":"10.48550/arxiv.1204.0257","title":"Not As Easy As It Seems: Automating the Construction of Lexical Chains Using Roget's Thesaurus","year":2012,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Thesaurus; Automatic summarization; Computer science; Natural language processing; Information retrieval; WordNet; Artificial intelligence; Lexical database","score_opus":0.07862772650852944,"score_gpt":0.23503616264642976,"score_spread":0.15640843613790034,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2952904369","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017924737,0.00037591535,0.96532947,0.00072941196,0.0001634916,0.0002411592,0.00028689482,0.008114713,0.006834183],"genre_scores_gemma":[0.047046553,0.0002403699,0.9462887,0.00021619341,0.000033740827,0.00017425454,0.00070353213,0.0010924515,0.0042041726],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99826705,0.0005501942,0.0002441749,0.0004896166,0.00036097277,0.000087984896],"domain_scores_gemma":[0.99471056,0.0017562523,0.00031207033,0.0020037757,0.0010890693,0.00012833714],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024662649,0.00089120015,0.0010164329,0.003150828,0.0017253195,0.0032528099,0.0015223919,0.0013738549,0.011044413],"category_scores_gemma":[0.012015594,0.0011328345,0.0012273623,0.003664135,0.0018701339,0.0064626504,0.0030446055,0.0015342517,0.007910962],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018257672,0.00015038546,0.0029316489,0.00068201724,0.00013478936,0.00040721847,0.0031413538,0.0035415858,0.030947326,0.07145916,0.01948797,0.866934],"study_design_scores_gemma":[0.00028741086,0.0005087393,0.0062378487,0.00093817286,0.00037899465,0.0028830457,0.0041049873,0.096208625,0.08713516,0.3417684,0.45909542,0.00045319338],"about_ca_topic_score_codex":0.002745398,"about_ca_topic_score_gemma":0.0048835143,"teacher_disagreement_score":0.011044413,"about_ca_system_score_codex":0.0006327258,"about_ca_system_score_gemma":0.0015292815,"threshold_uncertainty_score":0.03694725},"labels":[],"label_agreement":null},{"id":"W2953129863","doi":"10.48550/arxiv.cs/0508103","title":"Corpus-based Learning of Analogies and Semantic Relations","year":2005,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Natural language processing; Computer science; Artificial intelligence; Linguistics; Corpus linguistics; Philosophy","score_opus":0.02907103913200881,"score_gpt":0.28522808181434006,"score_spread":0.25615704268233125,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2953129863","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09823782,0.002049471,0.8724402,0.0013393587,0.00042710363,0.00052596955,0.003623769,0.0076812073,0.013675081],"genre_scores_gemma":[0.41597897,0.00057237543,0.5652172,0.00039280736,0.0002222651,0.0009857208,0.011128323,0.0004267581,0.0050754696],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99599105,0.0014837864,0.00026282936,0.0013771842,0.0007787571,0.00010637075],"domain_scores_gemma":[0.98577166,0.009997031,0.00047345285,0.0019408307,0.0016568303,0.0001602545],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037772611,0.0013553255,0.001132658,0.005097138,0.0014227477,0.0024045259,0.0031756249,0.0022569667,0.005601095],"category_scores_gemma":[0.028344737,0.0008881815,0.0013960333,0.0035303799,0.0018896016,0.0061482848,0.001994071,0.0028466214,0.0020122621],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000446253,0.00068018923,0.008442196,0.0007783455,0.0002998954,0.00040232178,0.00084415753,0.11050929,0.0062627825,0.042343892,0.027309764,0.801681],"study_design_scores_gemma":[0.00011156916,0.00010617199,0.002112492,0.000104978586,0.000078512785,0.0001984059,0.00026183485,0.90766716,0.0049086055,0.06820342,0.016196253,0.000050657254],"about_ca_topic_score_codex":0.0071199927,"about_ca_topic_score_gemma":0.010341784,"teacher_disagreement_score":0.0071199927,"about_ca_system_score_codex":0.0017633858,"about_ca_system_score_gemma":0.0019171058,"threshold_uncertainty_score":0.019976318},"labels":[],"label_agreement":null},{"id":"W2953152789","doi":"10.48550/arxiv.1809.01854","title":"Top-down Tree Structured Decoding with Syntactic Connections for Neural Machine Translation and Parsing","year":2018,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Parsing; Syntax; Artificial intelligence; Abstract syntax tree; Machine translation; Natural language processing; Sentence; Tree (set theory); Recurrent neural network; Decoding methods; Artificial neural network; Neural decoding; Tree structure; Dependency grammar; Programming language; Data structure; Algorithm","score_opus":0.05470065587620791,"score_gpt":0.2180873751482772,"score_spread":0.16338671927206927,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2953152789","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009537557,0.0005600483,0.98281485,0.0003846147,0.00011034328,0.00004605973,0.00028397838,0.0033426518,0.002919824],"genre_scores_gemma":[0.3518134,0.00096374395,0.6362052,0.00044407984,0.0001502044,0.0002912716,0.0018374072,0.00080220174,0.0074924217],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994616,0.0002195428,0.000041852756,0.00014841826,0.000097505275,0.00003116105],"domain_scores_gemma":[0.9988636,0.000623735,0.00006737572,0.0002113116,0.00020676668,0.000027230066],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009141846,0.00089649676,0.00059655873,0.0007147602,0.00050223566,0.0010060017,0.0011575299,0.001197591,0.0052223247],"category_scores_gemma":[0.0042008692,0.0005559916,0.0009895161,0.0011794588,0.0006730933,0.0022189885,0.0009129729,0.0019151698,0.0024296679],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021552482,0.00012995537,0.0007044421,0.00044993573,0.0001404113,0.00035461233,0.0002384944,0.38487875,0.023379888,0.06434069,0.0138057,0.51136166],"study_design_scores_gemma":[0.000010648175,0.000030373078,0.000109521396,0.000016937456,0.000021845553,0.000057189598,0.00001161845,0.94663554,0.0056754346,0.04492325,0.002494366,0.000013293182],"about_ca_topic_score_codex":0.004297804,"about_ca_topic_score_gemma":0.008205637,"teacher_disagreement_score":0.0052223247,"about_ca_system_score_codex":0.0009495133,"about_ca_system_score_gemma":0.001414249,"threshold_uncertainty_score":0.01747042},"labels":[],"label_agreement":null},{"id":"W2953445427","doi":"10.17345/triangle1.65-78","title":"An Introduction to Natural Language Processing: the Main Problems","year":2018,"lang":"en","type":"article","venue":"Triangle lenguaje literatura computación","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Simon Fraser University; Universitat Rovira i Virgili; European Commission","keywords":"Computer science; Natural language; Natural language processing; Natural (archaeology); Language technology; Natural language programming; Artificial intelligence; Universal Networking Language; Language identification; Process (computing); Human language; Linguistics; Comprehension approach; Programming language; History","score_opus":0.008686147077507509,"score_gpt":0.2808684325673573,"score_spread":0.27218228548984974,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2953445427","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0016000232,0.44581962,0.38720357,0.04780726,0.012117178,0.00019454368,0.002312597,0.0018287143,0.10111654],"genre_scores_gemma":[0.040696885,0.42108247,0.41660973,0.017049301,0.029818758,0.00085671316,0.0038533658,0.0010212858,0.06901142],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9983442,0.00033828052,0.00017324521,0.0004218988,0.00064666074,0.00007559996],"domain_scores_gemma":[0.99480766,0.0040983013,0.00015200979,0.00032326605,0.00051397743,0.00010478461],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017359862,0.0012909416,0.0012152207,0.003205101,0.0014790313,0.0059495266,0.0024621333,0.0034894696,0.019628003],"category_scores_gemma":[0.006402835,0.00080623303,0.0011567792,0.0062324074,0.0042763483,0.012899193,0.0023183688,0.0056486595,0.014722485],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000039717008,0.000068986774,0.00033299136,0.0024618953,0.00003656446,0.00024741943,0.00050290284,0.0014801533,0.00092261314,0.5139675,0.17241788,0.3075214],"study_design_scores_gemma":[0.000006200528,0.000014045278,0.00025713112,0.00060196046,0.0000093548615,0.00038359006,0.00010986308,0.0016476932,0.0002592697,0.39171642,0.60496986,0.000024644049],"about_ca_topic_score_codex":0.0014701497,"about_ca_topic_score_gemma":0.0012785443,"teacher_disagreement_score":0.019628003,"about_ca_system_score_codex":0.0016052657,"about_ca_system_score_gemma":0.0016127287,"threshold_uncertainty_score":0.065662265},"labels":[],"label_agreement":null},{"id":"W2953772283","doi":"10.22215/etd/2019-13514","title":"Representation Learning for Information Extraction","year":2019,"lang":"en","type":"dissertation","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Terminology; Information extraction; Bridging (networking); Sentence; Task (project management); Domain (mathematical analysis); Information retrieval","score_opus":0.012856257512250622,"score_gpt":0.33506640696819756,"score_spread":0.32221014945594695,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2953772283","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0037560794,0.0020138845,0.98707294,0.0009117494,0.00011161781,0.00010767532,0.000873664,0.0013902532,0.0037620838],"genre_scores_gemma":[0.17673066,0.0043189093,0.8020564,0.00056916545,0.000346698,0.00065760355,0.007123805,0.0003454175,0.007851458],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99800986,0.000769439,0.00017035085,0.00053414586,0.00041466634,0.000101592785],"domain_scores_gemma":[0.9970806,0.0013919044,0.00016893384,0.0009416955,0.0003673657,0.000049432732],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021492222,0.0009517809,0.0007831801,0.0021912635,0.00053699355,0.0026097526,0.00125125,0.0011005045,0.0054698503],"category_scores_gemma":[0.009975241,0.00039450132,0.0014806625,0.003260766,0.0010030484,0.0048992643,0.002559916,0.0026089721,0.003341538],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008255614,0.00008957137,0.0006150085,0.0006558609,0.00015278494,0.000073514384,0.00018445699,0.026817087,0.004608123,0.15303537,0.017304705,0.79638106],"study_design_scores_gemma":[0.000037387133,0.00009524713,0.0006507874,0.00025399175,0.000084701795,0.00017188555,0.00012682256,0.35171518,0.009600873,0.5893851,0.047833372,0.000044692635],"about_ca_topic_score_codex":0.001618039,"about_ca_topic_score_gemma":0.002017989,"teacher_disagreement_score":0.0054698503,"about_ca_system_score_codex":0.0013477273,"about_ca_system_score_gemma":0.0014253278,"threshold_uncertainty_score":0.018298507},"labels":[],"label_agreement":null},{"id":"W2953930904","doi":"10.7202/1060166ar","title":"Register, Source Language, and Cognateness Effects on Lexical Choice in Translated Dutch","year":2019,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Lexeme; Linguistics; Register (sociolinguistics); Multinomial logistic regression; Deviance (statistics); Natural language processing; Artificial intelligence","score_opus":0.017787090537982286,"score_gpt":0.2800686278559145,"score_spread":0.26228153731793225,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2953930904","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9949602,0.00018492853,0.0007209577,0.00009556226,0.000014042248,0.000011411771,0.0011441768,0.000026747844,0.0028419492],"genre_scores_gemma":[0.99583817,0.00008577458,0.0005615019,0.000020307043,0.0000097787815,0.000028632252,0.0024336241,0.00009678446,0.0009254948],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9964888,0.0015210288,0.00038470578,0.0007955459,0.0006559768,0.0001538956],"domain_scores_gemma":[0.9819356,0.013384003,0.0022203766,0.00085199624,0.0013138539,0.000294089],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023332227,0.000435028,0.0006517504,0.0009782931,0.000598945,0.0020191206,0.00041085793,0.00037305482,0.0057265013],"category_scores_gemma":[0.023343116,0.00021144378,0.00043322906,0.0021615566,0.0008376952,0.0014947449,0.0013844399,0.00072815374,0.0011648888],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004054355,0.00056576746,0.7074287,0.0020405443,0.000565398,0.004709202,0.088785335,0.003964604,0.04868713,0.004817042,0.006665477,0.12771651],"study_design_scores_gemma":[0.00009930964,0.00016364931,0.948804,0.00017283212,0.00011224232,0.0010817513,0.018950708,0.008526752,0.006310771,0.0018773358,0.013774965,0.00012566205],"about_ca_topic_score_codex":0.019141313,"about_ca_topic_score_gemma":0.024312112,"teacher_disagreement_score":0.019141313,"about_ca_system_score_codex":0.00093701907,"about_ca_system_score_gemma":0.00047413103,"threshold_uncertainty_score":0.03805983},"labels":[],"label_agreement":null},{"id":"W2954097401","doi":"10.7152/acro.v29i1.15455","title":"Machine translation and author keywords: A viable search strategy for scholars with limited English proficiency?","year":2019,"lang":"en","type":"article","venue":"Advances in Classification Research Online","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Search engine indexing; Information retrieval; Machine translation; Domain (mathematical analysis); Field (mathematics); Natural language processing; Artificial intelligence; Variation (astronomy)","score_opus":0.10458370749946665,"score_gpt":0.43296020380894934,"score_spread":0.3283764963094827,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2954097401","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.43290657,0.016203204,0.37303388,0.025901048,0.0011845445,0.003044359,0.015007562,0.01183782,0.120881006],"genre_scores_gemma":[0.7045714,0.004108623,0.26266438,0.0024176624,0.0005324326,0.001164364,0.007653303,0.00086311845,0.016024642],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9936069,0.0031975845,0.0011775692,0.0008308362,0.0008479651,0.00033911032],"domain_scores_gemma":[0.9751099,0.014029974,0.002947719,0.0026473375,0.004546451,0.00071869884],"candidate_categories":["metaresearch","scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.007010448,0.0009406301,0.0017359481,0.008532696,0.0014182162,0.0051107826,0.0010015649,0.0015472221,0.019699283],"category_scores_gemma":[0.040604424,0.0005687366,0.0007087097,0.010322817,0.0008542039,0.009621863,0.0027249074,0.0006017447,0.015526976],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001100773,0.00034167233,0.021886861,0.0034461413,0.00014106558,0.0024035813,0.007837975,0.0011845411,0.031478688,0.015147304,0.029522562,0.8855089],"study_design_scores_gemma":[0.0010143783,0.0027961645,0.041482612,0.0033417388,0.0008552934,0.014020322,0.040429957,0.04996808,0.07287763,0.12757242,0.6448987,0.0007427145],"about_ca_topic_score_codex":0.0022560155,"about_ca_topic_score_gemma":0.004660071,"teacher_disagreement_score":0.9948892,"about_ca_system_score_codex":0.0010813212,"about_ca_system_score_gemma":0.0041595334,"threshold_uncertainty_score":0.065900624},"labels":[],"label_agreement":null},{"id":"W2955002705","doi":"10.3758/s13428-019-01282-6","title":"LADEC: The Large Database of English Compounds","year":2019,"lang":"en","type":"article","venue":"Behavior Research Methods","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":60,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University; University of Alberta","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Bigram; Lexicon; Computer science; Natural language processing; Lexical database; Morpheme; Compound; WordNet; Parsing; Database; Artificial intelligence; Linguistics; Information retrieval; Trigram","score_opus":0.15682359605961013,"score_gpt":0.5424968809298324,"score_spread":0.38567328487022223,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2955002705","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.062402353,0.0043435446,0.0049047684,0.00046119562,0.0001932975,0.0005077796,0.90130347,0.0038284187,0.022055073],"genre_scores_gemma":[0.050745215,0.0016352145,0.01714166,0.0002584519,0.000080728954,0.00095072755,0.9231363,0.0007269762,0.0053247106],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.998579,0.00025741127,0.0003685938,0.00033720362,0.0003713558,0.00008639375],"domain_scores_gemma":[0.995672,0.0014782245,0.000414208,0.00084074895,0.0012265043,0.00036830342],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011429812,0.0013645483,0.0012814465,0.0064954353,0.0011662851,0.0024114419,0.0023745708,0.0015516629,0.035250716],"category_scores_gemma":[0.0072111944,0.0007201092,0.00068732724,0.0068577407,0.0005815396,0.005956205,0.0025841505,0.0010587046,0.02419136],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0024008916,0.000620743,0.01738151,0.015210469,0.00028757623,0.0028593338,0.0029729134,0.0016396675,0.019033097,0.012659094,0.68718946,0.2377453],"study_design_scores_gemma":[0.00033761785,0.00015254012,0.022741778,0.0005054148,0.00010469332,0.00094027136,0.0013700203,0.0015228242,0.004826951,0.0028278346,0.96452254,0.00014745607],"about_ca_topic_score_codex":0.007225091,"about_ca_topic_score_gemma":0.0135695385,"teacher_disagreement_score":0.035250716,"about_ca_system_score_codex":0.0011268556,"about_ca_system_score_gemma":0.0018594265,"threshold_uncertainty_score":0.117925406},"labels":[],"label_agreement":null},{"id":"W2955033484","doi":"10.18653/v1/w19-2516","title":"Sign Clustering and Topic Extraction in Proto-Elamite","year":2019,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada; Ministère de la Défense Nationale","keywords":"Chen; Sign (mathematics); Cluster analysis; Computer science; Library science; Linguistics; Natural language processing; Data science; Sociology; Artificial intelligence; Philosophy; Geology; Mathematics","score_opus":0.009330401100224761,"score_gpt":0.27077287905692754,"score_spread":0.2614424779567028,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2955033484","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07654139,0.0011418868,0.867197,0.0005685884,0.0004731439,0.00040836574,0.004351988,0.035841182,0.013476398],"genre_scores_gemma":[0.2935152,0.0004287365,0.6567989,0.00024852515,0.00019342185,0.0003332905,0.016731067,0.0031248517,0.028626021],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99901116,0.0002641168,0.00008553636,0.00027921822,0.00019720671,0.00016270166],"domain_scores_gemma":[0.99847347,0.00037401228,0.00006622758,0.00041915197,0.00055038114,0.00011685565],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015254958,0.0007034912,0.0009517364,0.004092483,0.0014011707,0.0020327836,0.001316633,0.0011341431,0.009642568],"category_scores_gemma":[0.0028966363,0.00045284172,0.0010884731,0.0022587948,0.00053272437,0.002206095,0.0030017716,0.00085920305,0.010580433],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010982405,0.00024986197,0.0066812607,0.0004194511,0.00014987265,0.0005938306,0.0014764332,0.005126094,0.062240247,0.01126849,0.04649014,0.8642061],"study_design_scores_gemma":[0.00032895416,0.00039747127,0.016677877,0.00022488888,0.0003723588,0.0018904476,0.0034133494,0.5851958,0.14023767,0.046463735,0.20460314,0.00019436885],"about_ca_topic_score_codex":0.003116334,"about_ca_topic_score_gemma":0.012279711,"teacher_disagreement_score":0.009642568,"about_ca_system_score_codex":0.00043099688,"about_ca_system_score_gemma":0.0011792802,"threshold_uncertainty_score":0.032257617},"labels":[],"label_agreement":null},{"id":"W2955265407","doi":"10.1075/slcs.180.09tor","title":"The organizational structure of lexical compound verbs in Japanese","year":2017,"lang":"en","type":"book-chapter","venue":"Studies in language companion series","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Linguistics; Organizational structure; Psychology; Natural language processing; Computer science; Political science; Philosophy","score_opus":0.025651778610635386,"score_gpt":0.3174149762868722,"score_spread":0.2917631976762368,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2955265407","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9348027,0.0004024201,0.025953408,0.00017315714,0.000013945204,0.000058094494,0.00025054,0.00010306321,0.038242836],"genre_scores_gemma":[0.99197394,0.00009603909,0.006039969,0.000012687168,0.000008313635,0.000032267508,0.0002171241,0.000049597245,0.0015700365],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.9996474,0.00006898371,0.00003694549,0.0001236156,0.00006606029,0.000057023182],"domain_scores_gemma":[0.9994735,0.00015533116,0.00012719493,0.00008822306,0.000105116866,0.000050602514],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00034232723,0.00025623475,0.00032203406,0.0016956425,0.0013745407,0.0019498207,0.00040783777,0.00038895625,0.0042279866],"category_scores_gemma":[0.00093626947,0.00037898766,0.0003478167,0.002269188,0.0024718458,0.00264533,0.00097088446,0.0003605347,0.0003898052],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004494773,0.00012166061,0.09905207,0.0007470712,0.00008575376,0.003809805,0.04231909,0.0023907328,0.14103532,0.60343957,0.0024060563,0.10414351],"study_design_scores_gemma":[0.00016525896,0.00053748034,0.46613488,0.000368222,0.0004968767,0.0058577326,0.048731625,0.03455949,0.036428284,0.29450747,0.11197648,0.0002361039],"about_ca_topic_score_codex":0.00399232,"about_ca_topic_score_gemma":0.006247831,"teacher_disagreement_score":0.0042279866,"about_ca_system_score_codex":0.0012435658,"about_ca_system_score_gemma":0.0007283679,"threshold_uncertainty_score":0.014144063},"labels":[],"label_agreement":null},{"id":"W2957637517","doi":"10.3968/11087","title":"A Comparative Study of Productivity and Quality Gain Between Post-Editing and Translating From Scratch","year":2019,"lang":"en","type":"article","venue":"Studies in literature and language","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Productivity; Scratch; Quality (philosophy); Computer science; Machine translation; Translation (biology); Mode (computer interface); Linguistics; Natural language processing; Human–computer interaction","score_opus":0.0359773795693733,"score_gpt":0.36584830630730464,"score_spread":0.32987092673793134,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2957637517","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9948826,0.00032567536,0.0015535412,0.000072579336,0.000021441729,0.000042320782,0.00009019251,0.000069214424,0.002942435],"genre_scores_gemma":[0.99557364,0.0002144436,0.0016849426,0.000024365767,0.000039502727,0.000038881317,0.00022862523,0.000042242238,0.0021532604],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99555826,0.0019605367,0.0004002976,0.00052542065,0.0012115525,0.00034378588],"domain_scores_gemma":[0.94120723,0.039610296,0.0057113506,0.0042114765,0.0067344527,0.002525214],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005348969,0.0004539722,0.00074305147,0.0014210308,0.0005130229,0.0015664897,0.0005368195,0.00063354796,0.0042478815],"category_scores_gemma":[0.043162543,0.00016288683,0.00059241283,0.0014231788,0.0008925267,0.0014398065,0.0014041448,0.00065279007,0.0010242494],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.01641309,0.0066804616,0.16680579,0.0024773523,0.00062751235,0.0013856621,0.043265,0.0028580278,0.099943236,0.0011026185,0.0020686225,0.6563727],"study_design_scores_gemma":[0.00037521656,0.02875135,0.90360403,0.0001691482,0.0005194394,0.001358436,0.012682299,0.0034630387,0.040673606,0.0011665385,0.007117644,0.00011924853],"about_ca_topic_score_codex":0.0006661177,"about_ca_topic_score_gemma":0.00083852216,"teacher_disagreement_score":0.005348969,"about_ca_system_score_codex":0.0006144198,"about_ca_system_score_gemma":0.00062157173,"threshold_uncertainty_score":0.028288364},"labels":[],"label_agreement":null},{"id":"W2957991775","doi":"10.1515/tlr-2019-2031","title":"Constraining long-distance allomorphy","year":2019,"lang":"en","type":"article","venue":"The Linguistic Review","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Allomorph; Linguistics; Computer science; Vocabulary; Natural language processing; Morpheme; Philosophy","score_opus":0.0187757938722218,"score_gpt":0.2996946338603698,"score_spread":0.28091883998814804,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2957991775","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3657159,0.0094888015,0.49250105,0.003975851,0.00030266776,0.000060018956,0.0003721858,0.0008961901,0.12668735],"genre_scores_gemma":[0.9666021,0.0016198577,0.027299391,0.0003477854,0.00009713378,0.000036913454,0.00020006292,0.00023171237,0.003565064],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99869764,0.00049823563,0.00007495551,0.00034893586,0.000244174,0.00013593453],"domain_scores_gemma":[0.9958444,0.0023400746,0.00043796175,0.00092026475,0.0003445245,0.00011275948],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015510995,0.00031005603,0.00047052067,0.0006198983,0.0006528803,0.0021999269,0.0015064513,0.001054478,0.0071661905],"category_scores_gemma":[0.0062921764,0.00036996364,0.00046051675,0.0007861389,0.0031010793,0.005916734,0.0026213625,0.0016486872,0.001109046],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005904355,0.000041754145,0.0032543105,0.0004557507,0.00004336526,0.00043915262,0.0008789195,0.00704521,0.013410418,0.8979642,0.0017454398,0.074662484],"study_design_scores_gemma":[0.000016869499,0.00004058172,0.004481568,0.000121576864,0.000024028057,0.00052111957,0.0004993537,0.029444853,0.0042512915,0.9324753,0.028088845,0.000034506415],"about_ca_topic_score_codex":0.0011246771,"about_ca_topic_score_gemma":0.0011425735,"teacher_disagreement_score":0.0071661905,"about_ca_system_score_codex":0.0009422106,"about_ca_system_score_gemma":0.0005936173,"threshold_uncertainty_score":0.023973227},"labels":[],"label_agreement":null},{"id":"W2961885508","doi":"10.7202/1060173ar","title":"Traduction automatique et usage linguistique : une analyse de traductions anglais-français réunies en corpus","year":2019,"lang":"fr","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Philosophy","score_opus":0.023203655751475243,"score_gpt":0.30413366911468226,"score_spread":0.280930013363207,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2961885508","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9100982,0.0033196223,0.02664003,0.00093600946,0.00010570741,0.00017088733,0.0042385818,0.00020454683,0.05428645],"genre_scores_gemma":[0.967562,0.0011852945,0.013928659,0.00010493253,0.00003204152,0.00016299268,0.0029500825,0.0002163699,0.013857562],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9962262,0.0013762158,0.00041389407,0.0006329792,0.0010641551,0.00028658396],"domain_scores_gemma":[0.99006945,0.004554069,0.0010613225,0.0011142296,0.0030831038,0.00011790723],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033194574,0.0004896991,0.00037978587,0.005543899,0.0026170136,0.003274811,0.0005200231,0.0003746647,0.0046952404],"category_scores_gemma":[0.0074374597,0.000304415,0.00036844067,0.011099662,0.0023446365,0.0018066439,0.0014065318,0.00064601423,0.0007562029],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005149007,0.00010271268,0.10917669,0.0017763678,0.00024603188,0.001434557,0.34567824,0.0017799408,0.036774985,0.04038662,0.00829081,0.4538382],"study_design_scores_gemma":[0.000033768927,0.00012598957,0.35922837,0.0007719733,0.0002465855,0.0022240707,0.15126052,0.0034768379,0.023719117,0.004595343,0.4541018,0.0002156876],"about_ca_topic_score_codex":0.16840246,"about_ca_topic_score_gemma":0.19446726,"teacher_disagreement_score":0.16840246,"about_ca_system_score_codex":0.0046077413,"about_ca_system_score_gemma":0.0036256744,"threshold_uncertainty_score":0.33484453},"labels":[],"label_agreement":null},{"id":"W2962355539","doi":"10.7202/1060174ar","title":"Teaching Specialised Translation Through Corpus Linguistics: Translation Quality Assessment and Methodology Evaluation and Enhancement by Experimental Approach","year":2019,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Terminology; Computer science; Quality (philosophy); Corpus linguistics; Context (archaeology); Applied linguistics; Translation (biology); Machine translation; Linguistics; Natural language processing; Artificial intelligence","score_opus":0.21068210889984973,"score_gpt":0.4369700918338444,"score_spread":0.22628798293399466,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2962355539","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.33947566,0.0021250385,0.6158934,0.0006751137,0.00024164947,0.02693574,0.00046370755,0.0009453086,0.013244384],"genre_scores_gemma":[0.41332898,0.000641583,0.56099766,0.00015604055,0.00008243039,0.022395078,0.00031957772,0.00034865385,0.0017299562],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.830045,0.14017332,0.010949474,0.006985687,0.010554985,0.001291534],"domain_scores_gemma":[0.68813276,0.20063953,0.014161883,0.03592549,0.058849458,0.0022909346],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.13708062,0.0014227349,0.0014435098,0.0046948413,0.0028215125,0.0041577877,0.0023919584,0.0014561283,0.004839201],"category_scores_gemma":[0.2325737,0.000900804,0.0011242084,0.0046253824,0.0046837847,0.0034027903,0.0066790725,0.0012909841,0.0007367816],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.003933835,0.0064584143,0.025368294,0.008528953,0.00071848114,0.0005168889,0.07135125,0.008318561,0.04456078,0.032568578,0.0024106107,0.7952654],"study_design_scores_gemma":[0.00742087,0.040254764,0.108086474,0.008670439,0.004971635,0.0033081518,0.07583271,0.12033304,0.37650943,0.12292149,0.13023908,0.0014519454],"about_ca_topic_score_codex":0.0015993142,"about_ca_topic_score_gemma":0.0021319645,"teacher_disagreement_score":0.13708062,"about_ca_system_score_codex":0.004178135,"about_ca_system_score_gemma":0.0064278343,"threshold_uncertainty_score":0.7249603},"labels":[],"label_agreement":null},{"id":"W2962568471","doi":"10.7202/1060172ar","title":"Assessing the Status of Technical Documents as Textual Materials for Translation Training in Terms of Technical Terms","year":2019,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"Japan Society for the Promotion of Science","keywords":"Terminology; Computer science; Technical documentation; Translation (biology); Domain (mathematical analysis); Natural language processing; Information retrieval; Artificial intelligence; Linguistics; Documentation; Mathematics","score_opus":0.04976673191036504,"score_gpt":0.3549762152903598,"score_spread":0.30520948337999476,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2962568471","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.777961,0.004178115,0.1557524,0.0013644267,0.00060203334,0.0017096183,0.0032477777,0.0014275115,0.05375713],"genre_scores_gemma":[0.8088174,0.0013037233,0.17801848,0.00016949502,0.00022144002,0.0015085763,0.004128264,0.0007377788,0.0050948695],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.97116244,0.015494625,0.0039405585,0.001987609,0.007019632,0.00039522725],"domain_scores_gemma":[0.78589624,0.13835722,0.019756583,0.014354784,0.039441247,0.0021938798],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.026162222,0.00072156213,0.0007221485,0.012273397,0.0020854839,0.006895196,0.000854655,0.000936766,0.0052497545],"category_scores_gemma":[0.14481765,0.0003931886,0.00047859826,0.008302745,0.0019100033,0.0067220754,0.00252242,0.0011848689,0.0021143358],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017069684,0.0006982879,0.08475445,0.0053962776,0.00027783553,0.0007281294,0.046970986,0.0038750903,0.081848636,0.021645272,0.008816431,0.7432816],"study_design_scores_gemma":[0.00030913446,0.0033206665,0.4374361,0.0049601947,0.0011537116,0.0031461248,0.056295063,0.046714973,0.1571829,0.029662859,0.2590191,0.0007991741],"about_ca_topic_score_codex":0.0010938774,"about_ca_topic_score_gemma":0.0017122844,"teacher_disagreement_score":0.026162222,"about_ca_system_score_codex":0.0017262273,"about_ca_system_score_gemma":0.0018565562,"threshold_uncertainty_score":0.13836068},"labels":[],"label_agreement":null},{"id":"W2962696263","doi":"","title":"Extracting Parallel Sentences with Bidirectional Recurrent Neural Networks to Improve Machine Translation","year":2018,"lang":"en","type":"article","venue":"International Conference on Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Machine translation; Sentence; Artificial intelligence; Task (project management); Natural language processing; Translation (biology); Feature engineering; Recurrent neural network; Parallel corpora; Baseline (sea); Artificial neural network; Feature (linguistics); Feature extraction; Speech recognition; Deep learning","score_opus":0.03912998338648059,"score_gpt":0.3326886547462771,"score_spread":0.2935586713597965,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2962696263","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09780084,0.001507756,0.8837446,0.00057620637,0.00043487025,0.00021876744,0.00092305016,0.009709714,0.0050841286],"genre_scores_gemma":[0.3819616,0.0008031826,0.6018493,0.00047564175,0.00034040058,0.0003577566,0.0060969265,0.00092613854,0.007189017],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992028,0.00025052196,0.0000871623,0.00019941757,0.00018923,0.00007087732],"domain_scores_gemma":[0.998403,0.0005940583,0.00016211273,0.00024098033,0.00055700826,0.000042816253],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010589669,0.0016512768,0.0010800065,0.0012739786,0.00067139394,0.0008562485,0.0009095888,0.0008769784,0.0033750131],"category_scores_gemma":[0.004423319,0.0005207679,0.0010096051,0.0016446022,0.00037775672,0.0023014315,0.0010618935,0.0012875298,0.0031914392],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00057037285,0.00047930502,0.002073808,0.0007271477,0.0003263101,0.0011002667,0.0005443455,0.06104802,0.14057729,0.007311102,0.019637784,0.76560426],"study_design_scores_gemma":[0.00007344663,0.00033904292,0.0013617589,0.00004061663,0.00025553795,0.00036327512,0.0001688803,0.90057826,0.07185357,0.0139908455,0.010914855,0.000059925213],"about_ca_topic_score_codex":0.0024753662,"about_ca_topic_score_gemma":0.0060492824,"teacher_disagreement_score":0.0033750131,"about_ca_system_score_codex":0.00040877933,"about_ca_system_score_gemma":0.0010915883,"threshold_uncertainty_score":0.01129055},"labels":[],"label_agreement":null},{"id":"W2962724530","doi":"10.18653/v1/p17-2095","title":"Challenging Language-Dependent Segmentation for Arabic: An\\n Application to Machine Translation and Part-of-Speech Tagging","year":2017,"lang":"en","type":"article","venue":"Figshare","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Arabic; Computer science; Machine translation; Natural language processing; Artificial intelligence; Computational linguistics; Speech recognition; Segmentation; Volume (thermodynamics); Speech translation; Linguistics; Translation (biology); Philosophy","score_opus":0.04363071102748368,"score_gpt":0.33506845696126925,"score_spread":0.2914377459337856,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2962724530","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.46455503,0.0036669453,0.47372913,0.005623164,0.001285461,0.00040043858,0.003772247,0.020818114,0.026149489],"genre_scores_gemma":[0.6863377,0.0012246426,0.29238504,0.00074274867,0.00029107308,0.00019102189,0.0057492265,0.0011434794,0.011934991],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99925655,0.0002779569,0.000049558286,0.00020712952,0.00014743987,0.00006137095],"domain_scores_gemma":[0.9972204,0.0015958734,0.00012953965,0.00039563654,0.000524339,0.00013420267],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011974302,0.0014932702,0.00068852305,0.001021348,0.0014156207,0.0017924451,0.0009254722,0.0022221643,0.007649945],"category_scores_gemma":[0.0049885,0.00032415698,0.00057951343,0.0018343847,0.0006848762,0.001524242,0.0011218772,0.0012366629,0.0047209654],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012500287,0.0004764952,0.0078832805,0.0011026795,0.00014986595,0.0029328568,0.0009866705,0.10517644,0.09546546,0.012351504,0.053816415,0.7184082],"study_design_scores_gemma":[0.00005809442,0.00018391725,0.00411982,0.00004728911,0.000047606678,0.00094450143,0.0005393993,0.88955927,0.07160914,0.013903266,0.018920744,0.00006694545],"about_ca_topic_score_codex":0.007461717,"about_ca_topic_score_gemma":0.010390418,"teacher_disagreement_score":0.007649945,"about_ca_system_score_codex":0.0008333971,"about_ca_system_score_gemma":0.0009723693,"threshold_uncertainty_score":0.025591671},"labels":[],"label_agreement":null},{"id":"W2962725091","doi":"","title":"Structured Generative Models of Natural Source Code","year":2014,"lang":"en","type":"article","venue":"International Conference on Machine Learning","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":48,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Generative grammar; Source code; Probabilistic logic; Code (set theory); Context (archaeology); Generative model; Natural language processing; Artificial intelligence; Natural language; Rule-based machine translation; Language model; Statistical model; Programming language","score_opus":0.023374194294264492,"score_gpt":0.29857618009771947,"score_spread":0.27520198580345495,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2962725091","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.026416508,0.00034654155,0.968351,0.0005041801,0.000030663898,0.000059511414,0.0008158217,0.0011273929,0.0023484428],"genre_scores_gemma":[0.69397765,0.00082117313,0.29109147,0.0004271396,0.00012387022,0.0004996026,0.0047398396,0.0011556138,0.0071636974],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9984428,0.0006249373,0.00006299357,0.00039677392,0.00035987646,0.000112555725],"domain_scores_gemma":[0.9831366,0.013593438,0.0009715857,0.0011753901,0.00090538943,0.00021765122],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020280394,0.0008506885,0.0005751484,0.0019798195,0.0005110956,0.0015232858,0.0019254916,0.0014760997,0.0032325964],"category_scores_gemma":[0.016193168,0.0008996109,0.0014526463,0.0014048591,0.0019037622,0.0028822236,0.001316102,0.0020026783,0.0009999388],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006589059,0.00007663778,0.004550778,0.00019921293,0.000083645515,0.00038849784,0.0008438428,0.7320086,0.0024620437,0.20916997,0.0044449354,0.04570605],"study_design_scores_gemma":[0.000009641111,0.000009400772,0.0003024798,0.000019994439,0.000009482818,0.000086534405,0.000025556943,0.88699555,0.0004883753,0.11025917,0.0017830015,0.000010872104],"about_ca_topic_score_codex":0.006691869,"about_ca_topic_score_gemma":0.01157548,"teacher_disagreement_score":0.006691869,"about_ca_system_score_codex":0.00140358,"about_ca_system_score_gemma":0.0014210301,"threshold_uncertainty_score":0.013305843},"labels":[],"label_agreement":null},{"id":"W2963099470","doi":"10.18653/v1/p19-1632","title":"Matching Article Pairs with Graphical Decomposition and Convolutions","year":2019,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":55,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Sentence; Matching (statistics); Graph; Information retrieval; Theoretical computer science; Natural language processing; Artificial intelligence; Mathematics","score_opus":0.008370593085022299,"score_gpt":0.26696449777092185,"score_spread":0.25859390468589954,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2963099470","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0455908,0.00052410254,0.94263524,0.00030048954,0.00012630776,0.00014809209,0.0013134961,0.006930612,0.0024308872],"genre_scores_gemma":[0.3673307,0.0004816391,0.6157625,0.00034181547,0.00018438259,0.00023493123,0.008096151,0.000840512,0.0067273895],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99854076,0.0002481501,0.000117717886,0.0005143734,0.00040272705,0.0001763256],"domain_scores_gemma":[0.9979505,0.0006395796,0.00032178365,0.0005354192,0.00043592253,0.00011683031],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010489373,0.001099633,0.0010425474,0.005611276,0.00056498323,0.0017867805,0.0014635413,0.0015566325,0.004415273],"category_scores_gemma":[0.006061336,0.0006029256,0.0020669883,0.0040635075,0.0006624221,0.0030065104,0.0021831568,0.0013384834,0.0029191915],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000746566,0.00034320194,0.006444194,0.00039496293,0.00027924043,0.0005433271,0.0003311674,0.09891803,0.044645824,0.03350797,0.019796317,0.7940492],"study_design_scores_gemma":[0.000031257343,0.00006348077,0.001400406,0.000017440321,0.000051281713,0.00023994822,0.00007066632,0.9460449,0.011760605,0.03519455,0.005100388,0.000024988778],"about_ca_topic_score_codex":0.005375181,"about_ca_topic_score_gemma":0.0078116786,"teacher_disagreement_score":0.005611276,"about_ca_system_score_codex":0.000976363,"about_ca_system_score_gemma":0.0011114373,"threshold_uncertainty_score":0.014770567},"labels":[],"label_agreement":null},{"id":"W2963167206","doi":"10.26615/978-954-452-049-6_042","title":"Argument Labeling of Explicit Discourse Relations using LSTM Neural Networks","year":2017,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Argument (complex analysis); Computer science; Feature (linguistics); Feature engineering; Task (project management); Artificial intelligence; Measure (data warehouse); Recurrent neural network; Natural language processing; State (computer science); Term (time); Artificial neural network; Deep learning; Pattern recognition (psychology); Algorithm; Linguistics; Data mining","score_opus":0.030334449820018435,"score_gpt":0.3313397831565521,"score_spread":0.3010053333365337,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2963167206","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14927202,0.003294833,0.77795273,0.002313647,0.00092611794,0.00031228663,0.007436122,0.033179365,0.025312923],"genre_scores_gemma":[0.58872896,0.0007770042,0.3770672,0.0005647613,0.0002645128,0.00023522353,0.016809063,0.0008495521,0.01470367],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99887544,0.00037286693,0.00007747102,0.00042137824,0.00016525085,0.000087459935],"domain_scores_gemma":[0.9975986,0.001310324,0.00024016145,0.00035234718,0.00041985643,0.00007877997],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012703411,0.0017337821,0.0007028238,0.0017425406,0.0007738217,0.0021445171,0.0015898129,0.0019741699,0.0055976696],"category_scores_gemma":[0.0063759624,0.0005320787,0.00091547944,0.0013251752,0.00046273763,0.0050930777,0.0012675384,0.0029118862,0.005399351],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004886655,0.0003665031,0.0035572113,0.00083075935,0.0001234927,0.0004452805,0.00070105854,0.04500875,0.048775062,0.009611054,0.03843944,0.8516527],"study_design_scores_gemma":[0.000057575584,0.00011408188,0.0019190719,0.00020798523,0.000074688905,0.00016266065,0.0004155821,0.90852684,0.03397647,0.024727296,0.02977248,0.000045377037],"about_ca_topic_score_codex":0.0036344135,"about_ca_topic_score_gemma":0.0092563955,"teacher_disagreement_score":0.0055976696,"about_ca_system_score_codex":0.0013974113,"about_ca_system_score_gemma":0.001359195,"threshold_uncertainty_score":0.01872611},"labels":[],"label_agreement":null},{"id":"W2963258372","doi":"","title":"Bayesian Optimisation for Machine Translation","year":2014,"lang":"en","type":"article","venue":"Neural Information Processing Systems","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Machine translation; Bayesian probability; Artificial intelligence; Translation (biology); Machine learning","score_opus":0.016432709502656333,"score_gpt":0.26443310301270806,"score_spread":0.24800039351005174,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2963258372","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0056666713,0.002007395,0.9813984,0.0011761059,0.0001909164,0.00006406625,0.0002643359,0.0011864465,0.008045745],"genre_scores_gemma":[0.34749475,0.0024760375,0.61484545,0.00072775834,0.00069281115,0.00061176764,0.0017858414,0.001640681,0.029724905],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99736387,0.0014976036,0.00014187404,0.00037971287,0.00047421985,0.00014266833],"domain_scores_gemma":[0.992159,0.006498594,0.00022327747,0.00041884417,0.0005958522,0.00010437336],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036290505,0.0010509604,0.0025494704,0.001820314,0.001019976,0.0021016172,0.0018758124,0.0031321784,0.011847638],"category_scores_gemma":[0.019460078,0.001645278,0.0015316557,0.0023811439,0.0017556705,0.0038851278,0.0025374817,0.0032932628,0.0028797511],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002452227,0.00011033449,0.00027100096,0.00042483065,0.00015639172,0.00008685033,0.00015240221,0.60350543,0.0014254005,0.16352737,0.014942129,0.21515262],"study_design_scores_gemma":[0.000024827927,0.000013453251,0.000096971184,0.00002797779,0.000012910146,0.000012614067,0.000008535804,0.82463145,0.0002991935,0.17286803,0.001990754,0.000013348622],"about_ca_topic_score_codex":0.01012243,"about_ca_topic_score_gemma":0.008820158,"teacher_disagreement_score":0.011847638,"about_ca_system_score_codex":0.0022432457,"about_ca_system_score_gemma":0.00208296,"threshold_uncertainty_score":0.039634347},"labels":[],"label_agreement":null},{"id":"W2963318887","doi":"10.1142/s2425038416300032","title":"Open information extraction","year":2017,"lang":"en","type":"article","venue":"Encyclopedia with Semantic Computing and Robotic Intelligence","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Computer science; Natural language processing; Relationship extraction; Verb; Infinitive; Artificial intelligence; Verb phrase; Noun; Noun phrase; Variety (cybernetics); Tuple; Relation (database); Linguistics; Phrase; Information extraction; Mathematics; Data mining","score_opus":0.015175564793776212,"score_gpt":0.3063630140236197,"score_spread":0.2911874492298435,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2963318887","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0052008606,0.004781847,0.8750032,0.0018218855,0.00063861004,0.0008708901,0.023014808,0.04029794,0.04836982],"genre_scores_gemma":[0.046370562,0.0048698983,0.8272506,0.0013052244,0.0005098591,0.00073248503,0.09024716,0.004311108,0.024403052],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99479467,0.0009002879,0.0007577769,0.0011820176,0.0020234662,0.0003417821],"domain_scores_gemma":[0.98206997,0.005272519,0.00096339773,0.007379206,0.0038842482,0.0004307131],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0044590123,0.0019096023,0.0017635445,0.0114887245,0.0020930176,0.007578087,0.0034459927,0.002185363,0.028421177],"category_scores_gemma":[0.024916481,0.0011155141,0.0025777526,0.010112119,0.0015162627,0.01638305,0.010038546,0.002763285,0.026367482],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024718032,0.00013338422,0.0020374428,0.0015077853,0.00018338006,0.00049582554,0.00063131534,0.0014995298,0.0049705505,0.0813326,0.117545635,0.78941536],"study_design_scores_gemma":[0.000053829128,0.000082960076,0.0018325456,0.00074514403,0.00017028829,0.0012164647,0.00062775967,0.026806172,0.026801411,0.23723339,0.70427126,0.00015884622],"about_ca_topic_score_codex":0.0015912153,"about_ca_topic_score_gemma":0.0017173347,"teacher_disagreement_score":0.028421177,"about_ca_system_score_codex":0.0013368438,"about_ca_system_score_gemma":0.003033043,"threshold_uncertainty_score":0.09507829},"labels":[],"label_agreement":null},{"id":"W2963418282","doi":"10.1609/icwsm.v11i1.14859","title":"Data Sets: Word Embeddings Learned from Tweets and General Data","year":2017,"lang":"en","type":"article","venue":"Proceedings of the International AAAI Conference on Web and Social Media","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":32,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Thomson Reuters (Canada)","funders":"","keywords":"Computer science; Word (group theory); Word embedding; Natural language processing; Embedding; Artificial intelligence; Representation (politics); Information retrieval; Sentiment analysis; Linguistics","score_opus":0.12453903940145157,"score_gpt":0.3492956645279873,"score_spread":0.22475662512653577,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2963418282","genre_codex":"empirical","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6759546,0.002594461,0.10777132,0.002082671,0.0011976359,0.0017724825,0.19535834,0.005601882,0.0076666195],"genre_scores_gemma":[0.55810094,0.00084082177,0.11826679,0.0005723351,0.00023916113,0.0019116065,0.31509992,0.00028644787,0.004681952],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9984445,0.0004265759,0.00020719344,0.00047638448,0.000316431,0.00012883694],"domain_scores_gemma":[0.9946398,0.002116408,0.00041267183,0.0015452902,0.0010426214,0.00024329578],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015092137,0.0014880934,0.0006909809,0.0018867376,0.0005465838,0.00086653855,0.001265566,0.0019028137,0.0031765576],"category_scores_gemma":[0.013130464,0.00045316303,0.0015403184,0.0025693267,0.0008969465,0.002765133,0.0016371764,0.0022382534,0.0023925332],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004989858,0.0045445757,0.1445396,0.0038479543,0.0017269799,0.0029556202,0.001343517,0.12521708,0.023577558,0.007783481,0.16019629,0.5192775],"study_design_scores_gemma":[0.00094247377,0.0022613343,0.1340108,0.00038777606,0.0008245008,0.0040410026,0.0018312257,0.63649005,0.04859147,0.026004948,0.14406492,0.0005495427],"about_ca_topic_score_codex":0.004067454,"about_ca_topic_score_gemma":0.0056598685,"teacher_disagreement_score":0.004067454,"about_ca_system_score_codex":0.00076315703,"about_ca_system_score_gemma":0.00063137914,"threshold_uncertainty_score":0.010626674},"labels":[],"label_agreement":null},{"id":"W2963461220","doi":"10.7202/1061923ar","title":"Stringer, Gary A., gen. ed. DigitalDonne: The Online Variorum, vol. 6","year":2018,"lang":"en","type":"article","venue":"Renaissance and Reformation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Stringer; History; Genealogy; Engineering; Structural engineering","score_opus":0.010229110385176772,"score_gpt":0.2506358901954879,"score_spread":0.2404067798103111,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2963461220","genre_codex":"review","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00042903598,0.6505067,0.005897352,0.04902754,0.026092181,0.000036759753,0.00084633235,0.0007343038,0.2664298],"genre_scores_gemma":[0.0042465134,0.28144765,0.0036772725,0.009096597,0.008347392,0.00006447074,0.00072534743,0.0006163078,0.6917784],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99846196,0.0002879781,0.00012892179,0.00024224067,0.0007914117,0.00008748144],"domain_scores_gemma":[0.99820316,0.00067411736,0.00016089874,0.0001700783,0.00053549913,0.00025627014],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019703815,0.0014110215,0.0014025411,0.004574303,0.0019183297,0.0070386096,0.0015290843,0.003697665,0.20094913],"category_scores_gemma":[0.004867893,0.0008348194,0.00062209857,0.0060199657,0.00218546,0.013250176,0.0029385777,0.004168678,0.17658707],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000020717885,0.000009821695,0.00007996645,0.00027190987,0.0000031672287,0.000034729994,0.00014776013,0.000058840207,0.00013930617,0.0097240405,0.8429967,0.146513],"study_design_scores_gemma":[0.0000012251867,0.0000036215915,0.000071329974,0.00016202476,0.0000012549747,0.000075981276,0.000066804765,0.000023932249,0.000045443503,0.0019278084,0.99761677,0.000003779546],"about_ca_topic_score_codex":0.003850455,"about_ca_topic_score_gemma":0.012393988,"teacher_disagreement_score":0.20094913,"about_ca_system_score_codex":0.0022282011,"about_ca_system_score_gemma":0.0023533572,"threshold_uncertainty_score":0.6722418},"labels":[],"label_agreement":null},{"id":"W2963590687","doi":"10.1093/acrefore/9780199384655.013.631","title":"Morphology in Dene-Yeniseian Languages","year":2019,"lang":"en","type":"reference-entry","venue":"Oxford Research Encyclopedia of Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Prefix; Verb; Linguistics; Morpheme; Language family; Suffix; History; Philosophy","score_opus":0.04058193197752658,"score_gpt":0.3745562926024947,"score_spread":0.3339743606249681,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2963590687","genre_codex":"empirical","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.96109605,0.00030945978,0.0005305593,0.00016348042,0.000009604475,0.000014879491,0.00022364932,0.000023594212,0.03762867],"genre_scores_gemma":[0.997015,0.00010651617,0.00038512595,0.000020751977,0.0000015424948,0.000013526188,0.00022065255,0.0000150176,0.0022218328],"study_design_codex":"observational","study_design_gemma":"not_applicable","domain_scores_codex":[0.99977475,0.000033535263,0.000031160143,0.000056206274,0.000033752898,0.00007069182],"domain_scores_gemma":[0.999858,0.00003262773,0.00002803522,0.000016490301,0.00004834014,0.000016398246],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002678697,0.0002881852,0.00025405682,0.0013355623,0.0018590037,0.0015245171,0.0003042322,0.00033744096,0.003992159],"category_scores_gemma":[0.00048441923,0.00019013701,0.0001229349,0.002200065,0.0018860974,0.0010137685,0.0015866085,0.0003642847,0.00036697523],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010793121,0.00012343664,0.35152248,0.00048941385,0.000107220294,0.009552241,0.20660812,0.0013124563,0.092397176,0.138387,0.003742479,0.19467874],"study_design_scores_gemma":[0.000070250884,0.00010739998,0.7919527,0.00018671872,0.00004278959,0.0053317375,0.07892032,0.001042697,0.004653907,0.009348263,0.10826131,0.00008190696],"about_ca_topic_score_codex":0.014488851,"about_ca_topic_score_gemma":0.029571472,"teacher_disagreement_score":0.014488851,"about_ca_system_score_codex":0.0017332012,"about_ca_system_score_gemma":0.00072868314,"threshold_uncertainty_score":0.028809011},"labels":[],"label_agreement":null},{"id":"W2963723885","doi":"","title":"Understanding the Origins of Bias in Word Embeddings","year":2019,"lang":"en","type":"article","venue":"International Conference on Machine Learning","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":58,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Word (group theory); Word embedding; Natural language processing; Embedding; Artificial intelligence; TRACE (psycholinguistics); Harm; Machine learning; Linguistics; Psychology","score_opus":0.10216070678590511,"score_gpt":0.34135641771307296,"score_spread":0.23919571092716785,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2963723885","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3268034,0.0015525637,0.6644776,0.0014992214,0.0001608544,0.000068123416,0.0004973945,0.0007151815,0.0042257025],"genre_scores_gemma":[0.8819226,0.0010121188,0.11390493,0.00024392405,0.000099842975,0.00010762242,0.00069889805,0.0003318803,0.0016779953],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9978573,0.0009006187,0.00015743263,0.0004023217,0.0005519471,0.0001304162],"domain_scores_gemma":[0.98407215,0.009261594,0.00166935,0.0025589839,0.0022950496,0.00014282049],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003172817,0.0005175489,0.00047329796,0.0016221004,0.00054456404,0.0017597676,0.00052507233,0.0008547837,0.0009082171],"category_scores_gemma":[0.04818755,0.00047701556,0.00035981173,0.0016086907,0.0012911009,0.005480413,0.0015580286,0.0015430981,0.00052935735],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003789664,0.00013762452,0.09534712,0.000577245,0.000242481,0.0004587252,0.0047169724,0.20142433,0.03820248,0.13347045,0.006925682,0.5181179],"study_design_scores_gemma":[0.000032886768,0.000079487865,0.019615978,0.00015576318,0.000057123103,0.00040670176,0.00093548145,0.7184461,0.030379986,0.21633318,0.013477878,0.000079504316],"about_ca_topic_score_codex":0.004065749,"about_ca_topic_score_gemma":0.005401797,"teacher_disagreement_score":0.004065749,"about_ca_system_score_codex":0.00092335127,"about_ca_system_score_gemma":0.0010997421,"threshold_uncertainty_score":0.016779661},"labels":[],"label_agreement":null},{"id":"W2963793321","doi":"","title":"Effective Slot Filling Based on Shallow Distant Supervision Methods","year":2014,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Pipeline (software); Computer science; Modular design; Relation (database); Scheme (mathematics); Feature (linguistics); Relationship extraction; Artificial intelligence; Translation (biology); Representation (politics); Data mining; Mathematics; Programming language","score_opus":0.007045751759885142,"score_gpt":0.29659222993453005,"score_spread":0.2895464781746449,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2963793321","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.054074105,0.00043208658,0.9246741,0.00019916521,0.000079294485,0.00012000066,0.00044455833,0.016559143,0.00341757],"genre_scores_gemma":[0.40502205,0.00015445687,0.5844057,0.00020662937,0.00008041066,0.00013648917,0.0019770828,0.00093160616,0.0070856367],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99859256,0.00037185985,0.00011105035,0.00039350608,0.00038675452,0.0001441866],"domain_scores_gemma":[0.9977264,0.0012138698,0.00012168287,0.00049173145,0.00036074457,0.000085625856],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015281868,0.00095930125,0.001147876,0.0010228335,0.0006959312,0.0011796795,0.0018985423,0.0008965078,0.006488635],"category_scores_gemma":[0.0038163897,0.00058789813,0.0009425836,0.00092889584,0.0008138391,0.0034075493,0.0020842894,0.0013187841,0.0034562272],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006487466,0.00022344712,0.0013626073,0.00032364394,0.00005180604,0.00024501298,0.00063875935,0.024025569,0.04466523,0.010820258,0.01121318,0.90578175],"study_design_scores_gemma":[0.00011663814,0.00027305592,0.0010778387,0.000030332718,0.00005512865,0.00027664375,0.0002457296,0.9084795,0.051883988,0.027908085,0.009596122,0.000056963905],"about_ca_topic_score_codex":0.0022606095,"about_ca_topic_score_gemma":0.0049651633,"teacher_disagreement_score":0.006488635,"about_ca_system_score_codex":0.00054701685,"about_ca_system_score_gemma":0.0014161512,"threshold_uncertainty_score":0.021706581},"labels":[],"label_agreement":null},{"id":"W2964083623","doi":"10.3390/info10080246","title":"Text Filtering through Multi-Pattern Matching: A Case Study of Wu–Manber–Uy on the Language of Uyghur","year":2019,"lang":"en","type":"article","venue":"Information","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"Natural Sciences and Engineering Research Council of Canada; National Natural Science Foundation of China; York University","keywords":"Computer science; Spelling; Word (group theory); Word2vec; Artificial intelligence; Vowel; Natural language processing; Matching (statistics); Population; Field (mathematics); Linguistics; Speech recognition; Mathematics; Statistics; Sociology","score_opus":0.019728807507551432,"score_gpt":0.2912128139284301,"score_spread":0.27148400642087867,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2964083623","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8106694,0.0018693715,0.17232178,0.0016169687,0.00012617566,0.00032543985,0.00061759073,0.0029833326,0.009469865],"genre_scores_gemma":[0.7455184,0.0005599078,0.2413029,0.00048889924,0.00009953234,0.00013199695,0.0010654372,0.00048246526,0.010350437],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9991904,0.00031670302,0.00006779816,0.00017426834,0.00017491687,0.0000760083],"domain_scores_gemma":[0.99858785,0.0008129322,0.00010276406,0.00022304346,0.00021130827,0.00006223368],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010300056,0.00040960708,0.00070009637,0.000999107,0.0012128697,0.0009895698,0.0005499704,0.00087215647,0.0014819722],"category_scores_gemma":[0.0037803468,0.00015688375,0.00051667844,0.0016646595,0.0008958414,0.0024302006,0.0010437763,0.00057490397,0.0006333215],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012355465,0.0007010256,0.039548766,0.0010448201,0.00024292721,0.01633137,0.016324824,0.02452381,0.0850091,0.05785019,0.026059143,0.7311285],"study_design_scores_gemma":[0.0002457219,0.001073605,0.041361105,0.00017874093,0.00022443885,0.010456535,0.011674178,0.61783713,0.10307547,0.043017585,0.170595,0.0002605312],"about_ca_topic_score_codex":0.01186669,"about_ca_topic_score_gemma":0.0135346195,"teacher_disagreement_score":0.01186669,"about_ca_system_score_codex":0.00045652717,"about_ca_system_score_gemma":0.0007022339,"threshold_uncertainty_score":0.023595214},"labels":[],"label_agreement":null},{"id":"W2964199361","doi":"10.3115/v1/w14-4012","title":"On the Properties of Neural Machine Translation: Encoder–Decoder Approaches","year":2014,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6616,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; Canadian Institute for Advanced Research","funders":"Compute Canada; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs; Canadian Institute for Advanced Research","keywords":"Computer science; Machine translation; Encoder; Translation (biology); Artificial intelligence; Speech recognition; Operating system","score_opus":0.08784431085000213,"score_gpt":0.257640388060015,"score_spread":0.16979607721001286,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2964199361","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01045157,0.0021616127,0.97808206,0.0011342373,0.00009474363,0.00004670056,0.0001514331,0.0003033968,0.0075742267],"genre_scores_gemma":[0.60801893,0.0063699964,0.3739262,0.00097588584,0.0009539538,0.0005574742,0.0008245112,0.00072552706,0.007647597],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9965501,0.0016036971,0.00024365615,0.0007067264,0.000748071,0.00014778],"domain_scores_gemma":[0.9666987,0.027877856,0.0011021083,0.0022470402,0.0018673282,0.00020698384],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0060308264,0.0011943931,0.0014969064,0.0019402363,0.0009969762,0.002478506,0.001890862,0.002433633,0.0041174367],"category_scores_gemma":[0.034294847,0.0012061731,0.0010608601,0.0020711184,0.003895187,0.009783267,0.0020072008,0.004066052,0.0011308183],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016788018,0.000055447654,0.0010494603,0.00031988756,0.00009975925,0.00027140297,0.0002360922,0.22756892,0.0028593955,0.6900561,0.0020330003,0.07528273],"study_design_scores_gemma":[0.000015002329,0.000033483037,0.00019208346,0.0000347195,0.000022319151,0.00010451826,0.000017867245,0.6669067,0.0019612731,0.3292256,0.0014664097,0.000019964065],"about_ca_topic_score_codex":0.0034659768,"about_ca_topic_score_gemma":0.0024548955,"teacher_disagreement_score":0.0060308264,"about_ca_system_score_codex":0.0023374078,"about_ca_system_score_gemma":0.0015485198,"threshold_uncertainty_score":0.031894445},"labels":[],"label_agreement":null},{"id":"W2964227218","doi":"10.33011/computel.v2i.437","title":"Towards a General-Purpose Linguistic Annotation Backend","year":2019,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"National Science Foundation","keywords":"Computer science; Documentation; Annotation; Natural language processing; Upload; Transcription (linguistics); Process (computing); Artificial intelligence; Natural language; Linguistics; World Wide Web; Programming language","score_opus":0.009956162096114869,"score_gpt":0.27364523070711605,"score_spread":0.26368906861100117,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2964227218","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0048350366,0.00023209359,0.80698466,0.0006209088,0.0005885291,0.00034899317,0.0038756002,0.17486128,0.00765289],"genre_scores_gemma":[0.07723614,0.00048090165,0.8181943,0.0025250788,0.00056099426,0.0010018413,0.03766362,0.026272822,0.03606421],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9973557,0.00032736597,0.00023222981,0.000805807,0.0009864417,0.00029251742],"domain_scores_gemma":[0.9917462,0.0016802482,0.00023544634,0.0033464967,0.0025950426,0.00039646841],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004113893,0.0025533997,0.0019663966,0.0022910382,0.0016769712,0.009395057,0.00585702,0.002656582,0.028420437],"category_scores_gemma":[0.011104683,0.0017859864,0.0019207553,0.0020326974,0.0012356133,0.007935234,0.006993513,0.006179754,0.037513565],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0023522049,0.00073128624,0.0030756898,0.0010989466,0.0002622862,0.0014429622,0.0018703705,0.008854385,0.087301835,0.036173474,0.2680476,0.588789],"study_design_scores_gemma":[0.00026858106,0.00037582678,0.0022946855,0.00050930073,0.00022151295,0.0009367291,0.00083110586,0.2680045,0.23082808,0.06871634,0.4266862,0.00032712368],"about_ca_topic_score_codex":0.004864386,"about_ca_topic_score_gemma":0.0052995468,"teacher_disagreement_score":0.028420437,"about_ca_system_score_codex":0.0016283147,"about_ca_system_score_gemma":0.0021193277,"threshold_uncertainty_score":0.095075846},"labels":[],"label_agreement":null},{"id":"W2964270503","doi":"10.26615/978-954-452-049-6_054","title":"Improving Discourse Relation Projection to Build Discourse Annotated Corpora","year":2017,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Annotation; Natural language processing; Classifier (UML); Artificial intelligence; Parallel corpora; Intersection (aeronautics); Relation (database); Machine translation; Linguistics","score_opus":0.01701369306538897,"score_gpt":0.3263320920225469,"score_spread":0.30931839895715796,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2964270503","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07504673,0.0014330031,0.8854815,0.0008449946,0.0003800161,0.0005296487,0.004383409,0.022311073,0.009589642],"genre_scores_gemma":[0.23118505,0.00059734273,0.73585254,0.00022886512,0.00019566334,0.0011753456,0.022930212,0.0022471293,0.005587858],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9897267,0.004695376,0.00063794164,0.0027544787,0.0018416971,0.0003437957],"domain_scores_gemma":[0.97788405,0.011533942,0.0009133967,0.0036472543,0.005639854,0.0003815652],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006901233,0.001916003,0.0017581418,0.006029483,0.0019381221,0.0026998778,0.0017973448,0.0014142987,0.0065503335],"category_scores_gemma":[0.024264727,0.0010974105,0.0010574547,0.0043722163,0.0010090667,0.005974397,0.004764744,0.0034792887,0.0065752384],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007082188,0.0005534503,0.0067114574,0.001487789,0.00023454102,0.0006437919,0.0032568246,0.022035467,0.07634627,0.016982473,0.025754841,0.8452848],"study_design_scores_gemma":[0.0001603795,0.00032005025,0.007893538,0.00037756658,0.00022447977,0.0005624789,0.0026743633,0.7364541,0.1129534,0.036296614,0.101911634,0.00017144065],"about_ca_topic_score_codex":0.004919361,"about_ca_topic_score_gemma":0.0077173067,"teacher_disagreement_score":0.006901233,"about_ca_system_score_codex":0.001245928,"about_ca_system_score_gemma":0.0035792158,"threshold_uncertainty_score":0.036497653},"labels":[],"label_agreement":null},{"id":"W2964769234","doi":"10.3233/jifs-179350","title":"Summarizing videos into a target language: Methodology, architectures and evaluation","year":2019,"lang":"en","type":"article","venue":"Journal of Intelligent & Fuzzy Systems","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Agence Nationale de la Recherche","keywords":"Automatic summarization; Computer science; Focus (optics); Component (thermodynamics); Machine translation; Natural language processing; Quality (philosophy); Segmentation; Multimedia; Artificial intelligence; Information retrieval","score_opus":0.03387011312488569,"score_gpt":0.3418592066533649,"score_spread":0.3079890935284792,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2964769234","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.31700674,0.0014841108,0.66501135,0.00039597112,0.00012122257,0.0058233775,0.0009865679,0.002955356,0.0062153237],"genre_scores_gemma":[0.43839037,0.0008077682,0.5521719,0.000058353122,0.00006330397,0.002741126,0.002114182,0.00022602757,0.0034269388],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99319094,0.0037294682,0.0006102766,0.0007133553,0.0015044871,0.00025155552],"domain_scores_gemma":[0.9898741,0.0039974377,0.00078271056,0.00090950273,0.003976841,0.00045945233],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00903773,0.0009976056,0.0006085372,0.0027730206,0.00068699656,0.0022667192,0.0014277147,0.0012545126,0.0034211786],"category_scores_gemma":[0.016031284,0.00031840455,0.00049705047,0.0016257345,0.0007337435,0.0018809688,0.001292972,0.00064573693,0.000977072],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015720275,0.0022915814,0.011615267,0.0026384066,0.00034400003,0.00026411895,0.0031721957,0.062147908,0.07144074,0.008713466,0.0042554988,0.83154476],"study_design_scores_gemma":[0.00049038295,0.008354923,0.024524458,0.0003990296,0.00057991926,0.0006812963,0.004141631,0.7421314,0.19052471,0.0078730285,0.020049678,0.0002495559],"about_ca_topic_score_codex":0.0037611,"about_ca_topic_score_gemma":0.0036564956,"teacher_disagreement_score":0.00903773,"about_ca_system_score_codex":0.002153199,"about_ca_system_score_gemma":0.0014097044,"threshold_uncertainty_score":0.047796667},"labels":[],"label_agreement":null},{"id":"W2965122730","doi":"10.29173/cais510","title":"Analyse de la consistance dans l'agrégation des transcriptions pinyin polysyllabiques dans les bases bibliographiques","year":2013,"lang":"fr","type":"article","venue":"Proceedings of the Annual Conference of CAIS / Actes du congrès annuel de l ACSI","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Pinyin; Humanities; Philosophy; Linguistics","score_opus":0.02685372436193631,"score_gpt":0.2719849818114393,"score_spread":0.245131257449503,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2965122730","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6912391,0.008471423,0.13204117,0.0021555268,0.0009794022,0.00037495786,0.098240115,0.0066873278,0.05981084],"genre_scores_gemma":[0.84233516,0.0033198455,0.07190818,0.00014045982,0.0002518732,0.00032993974,0.06027513,0.0016966189,0.019742813],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99814415,0.00037436932,0.00024502538,0.00040720523,0.00072502514,0.00010425005],"domain_scores_gemma":[0.98609364,0.009576213,0.00074512686,0.000587794,0.0028689925,0.00012822663],"candidate_categories":["bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.0013594319,0.0005451823,0.00047227603,0.0061650844,0.0007362612,0.002328046,0.0005351614,0.00058026065,0.013063086],"category_scores_gemma":[0.014536352,0.00036624147,0.00029762543,0.0073259543,0.0004995003,0.0013557367,0.00083671237,0.0009551852,0.0055048033],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00154054,0.000099506244,0.039131645,0.0038162963,0.00032041417,0.0015927962,0.011924023,0.0038269898,0.22395849,0.008429955,0.02229253,0.6830668],"study_design_scores_gemma":[0.00022920818,0.0003490341,0.3701767,0.0016319554,0.0010317737,0.0030015996,0.014130384,0.060330585,0.20785515,0.004531744,0.33644634,0.0002854937],"about_ca_topic_score_codex":0.022789316,"about_ca_topic_score_gemma":0.029247442,"teacher_disagreement_score":0.9938349,"about_ca_system_score_codex":0.00080311875,"about_ca_system_score_gemma":0.0014363787,"threshold_uncertainty_score":0.04531336},"labels":[],"label_agreement":null},{"id":"W2965362018","doi":"10.29173/cais921","title":"Different Views of Textual « Aboutness »: A Recipient Evaluation of the Content Descriptors Proposed by Professional Indexers, Authors, Readers and Corpus Analysis Tools","year":2016,"lang":"fr","type":"article","venue":"Proceedings of the Annual Conference of CAIS / Actes du congrès annuel de l ACSI","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Identifier; Computer science; Artificial intelligence; Natural language processing; Linguistics; Information retrieval; Humanities; Art; Philosophy","score_opus":0.08332343261633626,"score_gpt":0.2983843559193685,"score_spread":0.21506092330303225,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2965362018","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.76314956,0.0032645147,0.12687384,0.0068460256,0.0004738186,0.0008389482,0.0010510447,0.0017887756,0.095713444],"genre_scores_gemma":[0.9414958,0.0008552939,0.045554344,0.0006074096,0.00021876446,0.0003757591,0.0010635593,0.0007174221,0.009111557],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.95922625,0.026616598,0.0023042466,0.0013689237,0.009945929,0.0005380896],"domain_scores_gemma":[0.8827937,0.06860123,0.007489614,0.009549982,0.029068014,0.002497476],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.038822096,0.00063688343,0.00059468084,0.008262235,0.0015714136,0.009202637,0.0010249326,0.0012398332,0.005947172],"category_scores_gemma":[0.11043524,0.0002792466,0.00060061796,0.004242405,0.0034730367,0.010701267,0.00491814,0.0017723697,0.0012558994],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0040056207,0.000829118,0.05962222,0.002887585,0.00039648006,0.0003996888,0.14377064,0.002369057,0.060186468,0.07108538,0.020631615,0.6338161],"study_design_scores_gemma":[0.00058373075,0.006209455,0.21511145,0.0029189575,0.0017028311,0.002814134,0.2336439,0.07507553,0.13562848,0.06434228,0.26118213,0.0007871345],"about_ca_topic_score_codex":0.0013007747,"about_ca_topic_score_gemma":0.0012776497,"teacher_disagreement_score":0.038822096,"about_ca_system_score_codex":0.002296713,"about_ca_system_score_gemma":0.0011211775,"threshold_uncertainty_score":0.20531332},"labels":[],"label_agreement":null},{"id":"W2965664538","doi":"10.29173/cais916","title":"Sounds of Yesterday: Case Study Taxonomy of Topoi from Dutch Silent Film Music","year":2016,"lang":"fr","type":"article","venue":"Proceedings of the Annual Conference of CAIS / Actes du congrès annuel de l ACSI","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Yesterday; Topos theory; Nomination; Taxonomy (biology); Art; Art history; Humanities; Literature; Physics; Zoology; Biology","score_opus":0.04852684936611298,"score_gpt":0.26847498809379516,"score_spread":0.21994813872768218,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2965664538","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.96645546,0.0018662966,0.0064157196,0.0009017439,0.00006479866,0.000719458,0.0010628727,0.00003326711,0.022480609],"genre_scores_gemma":[0.9658865,0.002988258,0.013547279,0.00040430337,0.000054647728,0.0008739101,0.0018243039,0.00006185163,0.014358933],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.99714285,0.0011443368,0.00034460344,0.00037177032,0.000541571,0.00045484552],"domain_scores_gemma":[0.99776304,0.0012384405,0.00024282048,0.00022989453,0.00024983435,0.00027604817],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017474539,0.0006208565,0.000549541,0.004081604,0.0069671595,0.0036787903,0.0019990315,0.00214515,0.00553109],"category_scores_gemma":[0.004906309,0.0004379769,0.0005260991,0.006402856,0.003018552,0.0029912286,0.0033351998,0.0015058811,0.0006707592],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002670422,0.00064725184,0.057971388,0.0017516136,0.000037647977,0.040382646,0.7796121,0.0003351295,0.009393948,0.012782344,0.004501724,0.09231716],"study_design_scores_gemma":[0.00003153667,0.00018407157,0.077719115,0.0005186355,0.000042297354,0.023425076,0.76872945,0.00072823593,0.004626939,0.00264745,0.121274754,0.000072430215],"about_ca_topic_score_codex":0.05790517,"about_ca_topic_score_gemma":0.1364991,"teacher_disagreement_score":0.05790517,"about_ca_system_score_codex":0.0040909266,"about_ca_system_score_gemma":0.0023571395,"threshold_uncertainty_score":0.115136266},"labels":[],"label_agreement":null},{"id":"W2967837635","doi":"10.1016/j.joi.2019.07.004","title":"Analyzing linguistic complexity and scientific impact","year":2019,"lang":"en","type":"article","venue":"Journal of Informetrics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":74,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"Qinglan Project of Jiangsu Province of China; Major Program of National Fund of Philosophy and Social Science of China","keywords":"Computer science; Linguistics; Linguistic sequence complexity; Natural language processing; Data science; Philosophy","score_opus":0.024388254781145414,"score_gpt":0.3143265421653923,"score_spread":0.2899382873842469,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2967837635","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.95397645,0.0016352346,0.014774297,0.0031327787,0.00009560063,0.000088671055,0.0009902067,0.00014320688,0.025163446],"genre_scores_gemma":[0.99472106,0.0003065239,0.003753889,0.000067314846,0.0001385776,0.000028671055,0.00035966616,0.000034530804,0.0005896469],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99238247,0.0026450949,0.00055962993,0.00055210246,0.003431752,0.00042880094],"domain_scores_gemma":[0.78886086,0.17190936,0.015675677,0.006105574,0.013803421,0.00364516],"candidate_categories":["bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.008062338,0.00039741764,0.00070931576,0.019047104,0.0017605476,0.0071885386,0.0008551857,0.0009950238,0.005903327],"category_scores_gemma":[0.117697276,0.00041266414,0.0007490084,0.012455265,0.0036302423,0.008554831,0.0040215286,0.0019456651,0.0005009656],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001654333,0.0008975413,0.6350241,0.0008065751,0.0010387937,0.0010538781,0.0058903415,0.028868537,0.006414833,0.11973263,0.0052089514,0.19340953],"study_design_scores_gemma":[0.00015054172,0.00040209174,0.41017583,0.00025444105,0.0007799708,0.00063744315,0.011313086,0.14065903,0.004838499,0.41904002,0.011514251,0.00023470406],"about_ca_topic_score_codex":0.0044750813,"about_ca_topic_score_gemma":0.0045965835,"teacher_disagreement_score":0.9809529,"about_ca_system_score_codex":0.0023549956,"about_ca_system_score_gemma":0.0020259493,"threshold_uncertainty_score":0.042638183},"labels":[],"label_agreement":null},{"id":"W2969615198","doi":"","title":"The First “Shares: The First Ten Foruns for Linguistics Sharing","year":2018,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Linguistic Association","funders":"","keywords":"Linguistics; Computer science; Philosophy","score_opus":0.01990896633683319,"score_gpt":0.2900070156759816,"score_spread":0.2700980493391484,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2969615198","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03034112,0.0045765894,0.7276325,0.04424091,0.007143131,0.00042630627,0.0011892072,0.005401867,0.17904837],"genre_scores_gemma":[0.4704901,0.0026639076,0.38110304,0.011443702,0.0038361314,0.0009461699,0.0019847464,0.0058598956,0.12167231],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.98779196,0.0052000056,0.0008278618,0.0021453819,0.0027747198,0.0012600207],"domain_scores_gemma":[0.97657996,0.0032949075,0.0006724566,0.014907348,0.0031635333,0.0013818233],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007415551,0.0011507377,0.0015625048,0.0016379553,0.008284755,0.010351941,0.0035038115,0.004214509,0.034371726],"category_scores_gemma":[0.030967265,0.0011044771,0.0014829597,0.0022397107,0.011603835,0.046185113,0.020256294,0.007895629,0.011367635],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012134867,0.00004663245,0.00093914487,0.00014871891,0.00004869913,0.00012507859,0.0023973123,0.00030651377,0.0015748047,0.87093407,0.033675842,0.089681745],"study_design_scores_gemma":[0.000014335313,0.00003868029,0.00025338007,0.0001367707,0.000030840292,0.00023152884,0.0013479269,0.0018742994,0.0023041903,0.7605515,0.23316604,0.000050373033],"about_ca_topic_score_codex":0.0018533187,"about_ca_topic_score_gemma":0.0029706385,"teacher_disagreement_score":0.034371726,"about_ca_system_score_codex":0.0025696883,"about_ca_system_score_gemma":0.0043701306,"threshold_uncertainty_score":0.11498493},"labels":[],"label_agreement":null},{"id":"W2970009562","doi":"10.18653/v1/w19-5431","title":"Neural Machine Translation of Low-Resource and Similar Languages with Backtranslation","year":2019,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada; Compute Canada","keywords":"Machine translation; Nepali; Computer science; Czech; Natural language processing; Artificial intelligence; Example-based machine translation; Transformer; Transfer-based machine translation; Translation (biology); Hindi; Portuguese; Task (project management); Linguistics; Engineering","score_opus":0.005545113999298296,"score_gpt":0.23247263247754477,"score_spread":0.22692751847824646,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2970009562","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.27157155,0.0016203922,0.6843951,0.001193922,0.0005335426,0.00021976585,0.0030526093,0.0074773245,0.029935751],"genre_scores_gemma":[0.79875386,0.0004183474,0.18349655,0.000385601,0.0001364395,0.00010417414,0.00557069,0.0007013291,0.010432969],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99881834,0.00042715296,0.000084258245,0.00033084973,0.00020785892,0.00013144751],"domain_scores_gemma":[0.9981528,0.00069810025,0.00012967312,0.0005707583,0.0004001801,0.00004842916],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013997082,0.0010317449,0.0007390305,0.0008031815,0.00061282323,0.0017927794,0.0012261979,0.0008816583,0.0063250787],"category_scores_gemma":[0.0062815445,0.00035281043,0.0010081495,0.0015042896,0.00063319964,0.0029245606,0.0019373761,0.0016095005,0.002840514],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000580288,0.00049128174,0.0055987663,0.0008096785,0.00053252967,0.001243049,0.0010602572,0.14123863,0.037883095,0.023910012,0.016677935,0.76997447],"study_design_scores_gemma":[0.00009202764,0.00030999558,0.0026768271,0.000087226304,0.00019945935,0.0007382594,0.0003529276,0.8913879,0.04762138,0.03603078,0.020422544,0.000080670216],"about_ca_topic_score_codex":0.007299488,"about_ca_topic_score_gemma":0.01322955,"teacher_disagreement_score":0.007299488,"about_ca_system_score_codex":0.00076987274,"about_ca_system_score_gemma":0.0014797027,"threshold_uncertainty_score":0.02115947},"labels":[],"label_agreement":null},{"id":"W2970352405","doi":"10.18653/v1/w19-5326","title":"Multi-Source Transformer for Kazakh-Russian-English Neural Machine Translation","year":2019,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Kazakh; Machine translation; Computer science; Transformer; Sentence; Natural language processing; Example-based machine translation; Artificial intelligence; Linguistics; Engineering; Voltage; Electrical engineering","score_opus":0.015863482647465038,"score_gpt":0.26790794981513927,"score_spread":0.25204446716767426,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2970352405","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024744002,0.0008428863,0.9403916,0.0006727751,0.0005329035,0.00018950686,0.0023454316,0.016869199,0.013411714],"genre_scores_gemma":[0.45065176,0.0007845724,0.51571107,0.00029042747,0.00019453747,0.0002370927,0.008620324,0.0020923577,0.021417974],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99945575,0.00014690624,0.00006587074,0.00013150104,0.00013680664,0.000063192456],"domain_scores_gemma":[0.99949944,0.00011454529,0.000026637645,0.00015846493,0.00017314884,0.000027704791],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00071470405,0.00056731026,0.0004977949,0.0007148801,0.0008378146,0.0010231097,0.0009136747,0.0006881475,0.013710174],"category_scores_gemma":[0.001939567,0.0002722571,0.00075553305,0.00094439136,0.00034744068,0.0019585132,0.0017298696,0.0009845406,0.006088607],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000635981,0.0002518091,0.0012466037,0.0010469307,0.00016401133,0.0011667694,0.00062966597,0.039425895,0.08058026,0.10543168,0.072805494,0.696615],"study_design_scores_gemma":[0.00009232004,0.00023556007,0.0015711503,0.00007308335,0.00014717135,0.0016404239,0.0002014316,0.65077615,0.17196552,0.075539485,0.09766353,0.00009421194],"about_ca_topic_score_codex":0.0023679074,"about_ca_topic_score_gemma":0.0057355086,"teacher_disagreement_score":0.013710174,"about_ca_system_score_codex":0.0007203944,"about_ca_system_score_gemma":0.0011032594,"threshold_uncertainty_score":0.04586512},"labels":[],"label_agreement":null},{"id":"W2970485137","doi":"10.18653/v1/w19-4637","title":"No Army, No Navy: BERT Semi-Supervised Learning of Arabic Dialects","year":2019,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":36,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Compute Canada","keywords":"Porting; Computer science; Arabic; Artificial intelligence; Natural language processing; Identification (biology); Navy; Macro; Supervised learning; Task (project management); F1 score; Set (abstract data type); Machine learning; Engineering; Linguistics; History; Artificial neural network; Programming language","score_opus":0.007084348724772917,"score_gpt":0.23562531512277335,"score_spread":0.22854096639800042,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2970485137","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20567356,0.0011824027,0.6220889,0.0013293743,0.00094850204,0.00032407042,0.01527142,0.12939553,0.023786243],"genre_scores_gemma":[0.63792217,0.00027846263,0.29997486,0.0006200073,0.00019620096,0.00024139955,0.021709593,0.0015764018,0.03748085],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995555,0.00009123632,0.000026136453,0.00018821514,0.0000913536,0.00004752474],"domain_scores_gemma":[0.99919015,0.00016283312,0.0000657581,0.00023717716,0.00025566653,0.0000884699],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00064073637,0.00096961344,0.00046314142,0.0007906028,0.0006201179,0.00096259813,0.0009271389,0.0006616104,0.00772749],"category_scores_gemma":[0.001917074,0.0002706662,0.00045145006,0.00040168987,0.0002337798,0.0014456407,0.0013958765,0.0012256509,0.010073664],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007024259,0.00024131827,0.008959425,0.00029780445,0.000118797376,0.00021685023,0.00035451382,0.013059416,0.05571292,0.0027779324,0.07455615,0.8430025],"study_design_scores_gemma":[0.000064315806,0.00029575033,0.0071220244,0.00007469015,0.00006531032,0.00045304565,0.000351977,0.82392305,0.09482054,0.009824036,0.062909864,0.00009538714],"about_ca_topic_score_codex":0.0041814875,"about_ca_topic_score_gemma":0.012009308,"teacher_disagreement_score":0.00772749,"about_ca_system_score_codex":0.000397682,"about_ca_system_score_gemma":0.000794359,"threshold_uncertainty_score":0.02585107},"labels":[],"label_agreement":null},{"id":"W2970776494","doi":"","title":"A Symbolic Summarizer for the Update Task of TAC 2008.","year":2008,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Task (project management); Programming language; Natural language processing; Systems engineering; Engineering","score_opus":0.008506072012932964,"score_gpt":0.25370875295127565,"score_spread":0.2452026809383427,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2970776494","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025086388,0.0015322912,0.7477488,0.0012712897,0.00089190004,0.00048057237,0.013213901,0.19387138,0.015903478],"genre_scores_gemma":[0.20610027,0.00056344894,0.7372751,0.0005248207,0.00041078468,0.00061468035,0.026075386,0.005722718,0.022712762],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991731,0.0002225799,0.00010192996,0.00018036041,0.00025107086,0.00007085737],"domain_scores_gemma":[0.99823135,0.0007067824,0.00008856953,0.0004956307,0.00037854037,0.00009916972],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012139035,0.00082420796,0.00096352666,0.0012212761,0.00064139167,0.0014610743,0.0017044913,0.0011618835,0.02053356],"category_scores_gemma":[0.0071797064,0.00038857234,0.00053729664,0.0011176148,0.00027762834,0.002877854,0.0013690144,0.001253061,0.012336309],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012771939,0.0002404634,0.0009836708,0.000787843,0.00012708557,0.00036635064,0.0004845414,0.004232053,0.02631241,0.015209986,0.23465946,0.7153189],"study_design_scores_gemma":[0.00080051535,0.0011963113,0.0030540926,0.00025073608,0.00041798496,0.0010340591,0.0006302284,0.41346672,0.092806056,0.09000059,0.3961754,0.00016724301],"about_ca_topic_score_codex":0.0014761876,"about_ca_topic_score_gemma":0.0037940391,"teacher_disagreement_score":0.02053356,"about_ca_system_score_codex":0.00039143045,"about_ca_system_score_gemma":0.00090195617,"threshold_uncertainty_score":0.06869161},"labels":[],"label_agreement":null},{"id":"W2971045947","doi":"10.18653/v1/w19-5434","title":"NRC Parallel Corpus Filtering System for WMT 2019","year":2019,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Task (project management); Machine translation; Natural language processing; Artificial intelligence; Translation (biology); Information retrieval; Engineering","score_opus":0.010627906549785397,"score_gpt":0.2485762465397344,"score_spread":0.237948339989949,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2971045947","genre_codex":"software","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":"software","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015271694,0.0015394717,0.23634274,0.004240956,0.0059611723,0.0022145645,0.31240228,0.36477178,0.0572553],"genre_scores_gemma":[0.025519667,0.00036283524,0.26068032,0.0010310223,0.0006750843,0.00181055,0.62878627,0.039362162,0.04177207],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99505544,0.0009750543,0.00050236174,0.0011796628,0.001787309,0.0005001551],"domain_scores_gemma":[0.98682225,0.0022859261,0.0003976106,0.003151403,0.0064037945,0.0009389667],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0071862857,0.003312396,0.002974785,0.0065441616,0.0049584983,0.0052784416,0.004385267,0.0030825709,0.116244726],"category_scores_gemma":[0.022804782,0.0022290715,0.0019952348,0.006745255,0.0009886972,0.00445897,0.004883227,0.00429399,0.109312385],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006010677,0.00012504151,0.0005151796,0.00066117,0.00013610357,0.0005315904,0.00028433205,0.0014525454,0.0136288265,0.0036974421,0.8963407,0.08202604],"study_design_scores_gemma":[0.00041148718,0.0002037773,0.0024374058,0.00015973368,0.00012688237,0.00079007307,0.0003927809,0.031751376,0.035190143,0.009857608,0.9184605,0.00021828488],"about_ca_topic_score_codex":0.038862947,"about_ca_topic_score_gemma":0.047623966,"teacher_disagreement_score":0.116244726,"about_ca_system_score_codex":0.0032815228,"about_ca_system_score_gemma":0.010540345,"threshold_uncertainty_score":0.3888774},"labels":[],"label_agreement":null},{"id":"W2971882750","doi":"10.33011/computel.v1i.4277","title":"Bootstrapping a Neural Morphological Analyzer for St. Lawrence Island Yupik from a Finite-State Transducer","year":2019,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Spectrum analyzer; Computer science; Bootstrapping (finance); State (computer science); Set (abstract data type); Artificial neural network; Artificial intelligence; Algorithm; Mathematics; Telecommunications; Programming language","score_opus":0.02030717725948598,"score_gpt":0.2697912964862553,"score_spread":0.24948411922676936,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2971882750","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5013819,0.00009642209,0.4658416,0.00039364022,0.000103520906,0.00012624587,0.0014935122,0.023317331,0.007245913],"genre_scores_gemma":[0.742246,0.000064376036,0.24989554,0.00011100387,0.000015049525,0.00008777746,0.0023897155,0.0008661268,0.0043243878],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99972755,0.00004230642,0.000027561993,0.00010980485,0.00006881373,0.000024040064],"domain_scores_gemma":[0.9988103,0.00055457995,0.00009085652,0.00013956366,0.00036900598,0.000035726884],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00036660163,0.0004250754,0.00025021634,0.00061611686,0.00050371594,0.0008243462,0.0006396122,0.00037136473,0.0034444488],"category_scores_gemma":[0.0017341387,0.00028152094,0.00037529835,0.0003840469,0.0005370424,0.0016136204,0.00066796667,0.00069412816,0.0015723236],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00051614357,0.00015906796,0.016911007,0.00036175517,0.00009670704,0.0012885039,0.0017204686,0.024604255,0.33420336,0.016280657,0.009739281,0.5941188],"study_design_scores_gemma":[0.000027038595,0.0001529832,0.008541486,0.00003708519,0.000075797856,0.00063267705,0.0006818373,0.81816864,0.14967704,0.011162755,0.0107747335,0.00006800329],"about_ca_topic_score_codex":0.011440553,"about_ca_topic_score_gemma":0.029575061,"teacher_disagreement_score":0.011440553,"about_ca_system_score_codex":0.0006749031,"about_ca_system_score_gemma":0.001196408,"threshold_uncertainty_score":0.022747934},"labels":[],"label_agreement":null},{"id":"W2972368819","doi":"10.1016/b978-0-444-64193-9.00023-3","title":"Representing complementary user perspectives in a language atlas","year":2019,"lang":"en","type":"book-chapter","venue":"Modern cartography","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"Social Sciences and Humanities Research Council of Canada; Faculty of Arts and Social Sciences, Carleton University; Carleton University; Alexander von Humboldt-Stiftung","keywords":"Atlas (anatomy); Computer science; Linguistics; Relation (database); Cartography; Representation (politics); Artificial intelligence; Natural language processing; Geography; Data mining","score_opus":0.01768767379351895,"score_gpt":0.27055057039734143,"score_spread":0.2528628966038225,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2972368819","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025101103,0.0011069791,0.7248253,0.0009983531,0.000333016,0.00011490739,0.0057797413,0.0082325665,0.233508],"genre_scores_gemma":[0.32455266,0.0029870425,0.5810163,0.00030326136,0.00014352093,0.0002669359,0.00655707,0.0027684784,0.08140461],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9996648,0.000102043065,0.000019227873,0.00008568755,0.00009129417,0.000036994752],"domain_scores_gemma":[0.9993818,0.0003188997,0.000027438344,0.000101778525,0.000114513685,0.000055679582],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00044055987,0.00061947515,0.0004081639,0.0026841327,0.0010156275,0.0048943036,0.0005763343,0.00070046814,0.025049182],"category_scores_gemma":[0.0014017131,0.00045794933,0.0006695135,0.0040781805,0.0013133042,0.0060539627,0.0019510261,0.0010410998,0.0055857785],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001658352,0.00003542329,0.001169803,0.00048088256,0.000034848505,0.000799724,0.010158899,0.0057331556,0.013065879,0.7033795,0.030743746,0.23423229],"study_design_scores_gemma":[0.000023431174,0.00005054848,0.0013531547,0.00022668985,0.00007948176,0.001100196,0.0055376864,0.033877987,0.009105298,0.27405313,0.6745274,0.00006509634],"about_ca_topic_score_codex":0.0041615227,"about_ca_topic_score_gemma":0.00790648,"teacher_disagreement_score":0.025049182,"about_ca_system_score_codex":0.0010395299,"about_ca_system_score_gemma":0.00090071914,"threshold_uncertainty_score":0.08379787},"labels":[],"label_agreement":null},{"id":"W2972727504","doi":"10.1007/s10831-020-09203-x","title":"The syntax of Korean VP anaphora: an experimental investigation","year":2020,"lang":"en","type":"article","venue":"Journal of East Asian Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; Simon Fraser University","funders":"","keywords":"Pronoun; Ellipsis (linguistics); Anaphora (linguistics); Syntax; Linguistics; Interpretation (philosophy); Population; Subject (documents); Computational linguistics; Contrast (vision); Natural language processing; Computer science; Mathematics; Artificial intelligence; Philosophy; Sociology","score_opus":0.02374035018849262,"score_gpt":0.2806321082751971,"score_spread":0.25689175808670445,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2972727504","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.990642,0.00010795722,0.002142119,0.000069528935,0.000015447567,0.0001816659,0.0002822652,0.00005122803,0.0065077324],"genre_scores_gemma":[0.9962065,0.00010177485,0.002322449,0.00005453026,0.0000101303895,0.00013732055,0.00031121934,0.00008385155,0.0007721637],"study_design_codex":"bench_or_experimental","study_design_gemma":"observational","domain_scores_codex":[0.9976681,0.001060999,0.00042507992,0.00046163297,0.00027308989,0.00011112311],"domain_scores_gemma":[0.9786766,0.01580034,0.0009791474,0.0025407497,0.0016478227,0.0003553036],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004021731,0.00062568823,0.00076961046,0.0005437961,0.0015736918,0.00240582,0.0012568713,0.0010540471,0.010026863],"category_scores_gemma":[0.01992747,0.0008374654,0.000334364,0.0008093667,0.001818331,0.0047842,0.0019947526,0.0014836384,0.00087480224],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.013961013,0.010278572,0.058069766,0.004099743,0.0002502589,0.0028101483,0.109994985,0.0018934679,0.6424185,0.08739082,0.0021560881,0.06667658],"study_design_scores_gemma":[0.008077481,0.01970957,0.25382513,0.0008386682,0.002322624,0.016444094,0.10153335,0.049147565,0.41014257,0.1005275,0.036541358,0.00089010695],"about_ca_topic_score_codex":0.0025562565,"about_ca_topic_score_gemma":0.0015677402,"teacher_disagreement_score":0.010026863,"about_ca_system_score_codex":0.00066426303,"about_ca_system_score_gemma":0.001028641,"threshold_uncertainty_score":0.03354323},"labels":[],"label_agreement":null},{"id":"W2972791632","doi":"10.18653/v1/w19-4202","title":"Cognate Projection for Low-Resource Inflection Generation","year":2019,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Cognate; Inflection; Context (archaeology); Projection (relational algebra); Task (project management); Resource (disambiguation); Artificial intelligence; Natural language processing; Algorithm; Linguistics; Geography; Computer network; Engineering","score_opus":0.01353267116394323,"score_gpt":0.26994282713076023,"score_spread":0.256410155966817,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2972791632","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02300474,0.00019233409,0.96284944,0.00019955152,0.00016543607,0.00012440581,0.0001942399,0.0074163037,0.0058535733],"genre_scores_gemma":[0.45582348,0.00024791047,0.5325129,0.00030939112,0.00019284303,0.00043071673,0.001374873,0.0016246208,0.0074832225],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9983077,0.000646977,0.00009581915,0.0005623441,0.00025618312,0.00013088572],"domain_scores_gemma":[0.9979982,0.0007082736,0.00007423426,0.0007554497,0.000359964,0.00010382902],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022302172,0.0013843596,0.0009143253,0.0012793716,0.0010340845,0.002326164,0.0019296932,0.0010747301,0.012588505],"category_scores_gemma":[0.00435412,0.00059076754,0.0010581253,0.001138513,0.0014033508,0.0039673448,0.0047013187,0.002310084,0.0055034277],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005579859,0.00032348817,0.0013081614,0.00024701067,0.00011252965,0.0002898851,0.00053812907,0.011091844,0.048094474,0.031121155,0.0069172704,0.899398],"study_design_scores_gemma":[0.00015678024,0.00057611585,0.00312641,0.000065526925,0.00013037403,0.0011134887,0.0004426116,0.54188794,0.16906452,0.25618917,0.027045265,0.00020191184],"about_ca_topic_score_codex":0.0009458524,"about_ca_topic_score_gemma":0.0013379877,"teacher_disagreement_score":0.012588505,"about_ca_system_score_codex":0.00043605934,"about_ca_system_score_gemma":0.0013268093,"threshold_uncertainty_score":0.042112708},"labels":[],"label_agreement":null},{"id":"W2972830656","doi":"","title":"Finnish Telegraphese Corpus","year":2019,"lang":"en","type":"article","venue":"UEF eRepo (University of Eastern Finland)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Linguistics; Computer science; Natural language processing; Philosophy","score_opus":0.0073325448705516705,"score_gpt":0.1896597013197922,"score_spread":0.18232715644924052,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2972830656","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014727448,0.0024948257,0.0025758666,0.00089420157,0.0004586296,0.0006150595,0.9176154,0.0011466783,0.05947188],"genre_scores_gemma":[0.020700457,0.00093521073,0.0064736954,0.00021818114,0.00014421773,0.0014087076,0.95057225,0.00043015325,0.019117111],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9986457,0.00017891866,0.00023060052,0.00044035882,0.00037796976,0.0001264117],"domain_scores_gemma":[0.99731725,0.00080342754,0.00015798707,0.0004006077,0.001164762,0.00015592296],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014694234,0.0012314656,0.0010875309,0.007320207,0.0030358168,0.0023605905,0.0018971729,0.0016931285,0.11030838],"category_scores_gemma":[0.004650106,0.00047453708,0.000575237,0.008731106,0.0007394246,0.002065478,0.0019960022,0.0010367784,0.04821881],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037371466,0.00009963562,0.0022521904,0.0037646813,0.00008461612,0.0019994054,0.0027040993,0.00045684824,0.006216961,0.007883698,0.89907086,0.07509321],"study_design_scores_gemma":[0.000087308785,0.000023276403,0.011020584,0.00020997449,0.00006354448,0.000793185,0.0008272587,0.0002609633,0.001570627,0.0009416966,0.9841627,0.000038963062],"about_ca_topic_score_codex":0.035695214,"about_ca_topic_score_gemma":0.043565866,"teacher_disagreement_score":0.11030838,"about_ca_system_score_codex":0.0021574856,"about_ca_system_score_gemma":0.005378667,"threshold_uncertainty_score":0.36901832},"labels":[],"label_agreement":null},{"id":"W2974191861","doi":"10.18653/v1/k19-1008","title":"Say Anything: Automatic Semantic Infelicity Detection in L2 English Indefinite Pronouns","year":2019,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Focus (optics); Linguistics; Task (project management); Natural language processing; Second language; Artificial intelligence; Psychology; Philosophy","score_opus":0.012587784930287954,"score_gpt":0.25817337148105923,"score_spread":0.24558558655077128,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2974191861","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.76231825,0.00043976321,0.2172022,0.001106687,0.00018411655,0.00007317828,0.0008669966,0.008662887,0.009145847],"genre_scores_gemma":[0.9282603,0.000085203385,0.06635925,0.00010164459,0.000044115928,0.000017355922,0.00074777036,0.0004093657,0.0039749136],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989748,0.00027477214,0.000066492765,0.00037135155,0.00022001674,0.00009257164],"domain_scores_gemma":[0.99632215,0.0020504382,0.00027338965,0.0005408327,0.000641973,0.00017122211],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013898675,0.00053521065,0.00051609613,0.0009772507,0.0007368343,0.0022178877,0.000861375,0.0009910341,0.0038734183],"category_scores_gemma":[0.0058880285,0.00031553645,0.00023244668,0.0007077533,0.00073124166,0.0020903442,0.001211487,0.0009962489,0.0025363644],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012809932,0.00036794713,0.046234984,0.00042486415,0.00008857484,0.0016946329,0.0046274946,0.0059767812,0.18871655,0.01929877,0.018150674,0.7131376],"study_design_scores_gemma":[0.00011352543,0.00031841215,0.04733774,0.00009343755,0.000099504075,0.002912401,0.004080361,0.56991386,0.28476354,0.070250876,0.019943707,0.00017261172],"about_ca_topic_score_codex":0.0017605578,"about_ca_topic_score_gemma":0.0020301272,"teacher_disagreement_score":0.0038734183,"about_ca_system_score_codex":0.0003076885,"about_ca_system_score_gemma":0.00046219915,"threshold_uncertainty_score":0.012957811},"labels":[],"label_agreement":null},{"id":"W2975636453","doi":"","title":"Automatic annotation of content units in TANGRAM","year":2006,"lang":"en","type":"article","venue":"The Web Conference","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Annotation; Content (measure theory); Artificial intelligence; Mathematics","score_opus":0.03985435792517278,"score_gpt":0.2592653735791736,"score_spread":0.21941101565400084,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2975636453","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.43117207,0.0033991677,0.4396548,0.00097272586,0.0011330369,0.0008172182,0.023654195,0.05790419,0.041292623],"genre_scores_gemma":[0.6525556,0.0008263613,0.29887027,0.00020372754,0.00026666405,0.00040826006,0.021780638,0.004212885,0.020875622],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99954957,0.00008852933,0.00005364354,0.00013162344,0.00012020827,0.000056435394],"domain_scores_gemma":[0.99776864,0.00087618554,0.00022706148,0.0003170702,0.0006567667,0.00015427351],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004146516,0.0007550483,0.00058162306,0.0060424996,0.0010438657,0.0012791815,0.0007033032,0.00073429814,0.011302878],"category_scores_gemma":[0.0029978014,0.0003426423,0.0004092135,0.003133796,0.00046416942,0.0022815024,0.0012003,0.0008582289,0.0040108436],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015347998,0.0002590691,0.013590643,0.0020947736,0.00010761494,0.003139158,0.0035011806,0.003048023,0.19074817,0.027303493,0.059059076,0.69561404],"study_design_scores_gemma":[0.00020081003,0.0005404348,0.048824206,0.0010608911,0.0005611422,0.0033608628,0.0038525078,0.30728942,0.2721555,0.04387664,0.31800961,0.00026789308],"about_ca_topic_score_codex":0.003712872,"about_ca_topic_score_gemma":0.0062310845,"teacher_disagreement_score":0.011302878,"about_ca_system_score_codex":0.0005952042,"about_ca_system_score_gemma":0.0008645048,"threshold_uncertainty_score":0.037811935},"labels":[],"label_agreement":null},{"id":"W2976223060","doi":"10.7557/5.4876","title":"Data citation in linguistics publications","year":2019,"lang":"en","type":"article","venue":"Septentrio Conference Series","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Carleton University","funders":"","keywords":"Citation; Presentation (obstetrics); Scholarly communication; Publishing; Linguistics; Applied linguistics; Computer science; Field (mathematics); Sociology; Library science; Political science","score_opus":0.05117528215963513,"score_gpt":0.30785708418584307,"score_spread":0.25668180202620794,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2976223060","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"reproducibility","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"reproducibility","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018597228,0.18545052,0.065362886,0.16320693,0.06365361,0.0011836533,0.03895077,0.0025038293,0.46109056],"genre_scores_gemma":[0.4418613,0.17975141,0.119393036,0.025749288,0.057479344,0.005200964,0.057098974,0.004655791,0.10880989],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.83382964,0.080925934,0.027685523,0.013903992,0.040254183,0.003400694],"domain_scores_gemma":[0.4879638,0.36368722,0.046762347,0.03791664,0.055349562,0.008320475],"candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.06779395,0.0012544899,0.0021048777,0.07764613,0.008052224,0.03627788,0.0037117924,0.007674774,0.04277026],"category_scores_gemma":[0.38180852,0.0011255157,0.0014959724,0.14416756,0.010390658,0.027658798,0.016067374,0.006295711,0.0137593085],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010648515,0.00005149752,0.007839209,0.0066933865,0.00015421282,0.0002632566,0.014353679,0.00055998506,0.00023663638,0.4956618,0.2312388,0.24284104],"study_design_scores_gemma":[0.00002302286,0.000034225533,0.0050553754,0.0063552307,0.000059322316,0.00039531145,0.0045829415,0.00055881887,0.00031116258,0.112409435,0.87012094,0.000094159135],"about_ca_topic_score_codex":0.004779589,"about_ca_topic_score_gemma":0.003858346,"teacher_disagreement_score":0.93220603,"about_ca_system_score_codex":0.013981088,"about_ca_system_score_gemma":0.01680159,"threshold_uncertainty_score":0.35853297},"labels":[],"label_agreement":null},{"id":"W2976940412","doi":"10.5539/ijel.v9n5p430","title":"A Corpus-Based Study of Hypotactic and Paratactic Thematic Relations in English and Urdu Clause Complexes","year":2019,"lang":"en","type":"article","venue":"International Journal of English Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Urdu; Linguistics; Realization (probability); Computer science; Sentence; Meaning (existential); Psychology; Natural language processing; Mathematics; Philosophy; Statistics","score_opus":0.01597520679602705,"score_gpt":0.2942876903531285,"score_spread":0.2783124835571015,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2976940412","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.97279197,0.001821836,0.0026567322,0.0002034247,0.000041854815,0.00018377819,0.0039015308,0.000049462476,0.018349444],"genre_scores_gemma":[0.97996134,0.001035495,0.007755939,0.000076570526,0.00002904106,0.00036891617,0.007310935,0.00006492287,0.0033968233],"study_design_codex":"qualitative","study_design_gemma":"observational","domain_scores_codex":[0.9991478,0.00033251126,0.00010409534,0.00020629306,0.0001695007,0.000039788396],"domain_scores_gemma":[0.995378,0.0028601505,0.00051894045,0.0004213208,0.00068875815,0.00013278445],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00092456216,0.00023890247,0.00030579555,0.004191757,0.0016947235,0.0011950207,0.0005054101,0.00043434268,0.0038424523],"category_scores_gemma":[0.005522656,0.0001831871,0.000120924684,0.0065625613,0.0012352985,0.0014854466,0.0010152285,0.00055117847,0.0006606099],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007047302,0.0007506938,0.13263096,0.005772932,0.00009885038,0.008408691,0.4121075,0.0010465567,0.043724343,0.02517522,0.024342675,0.34523687],"study_design_scores_gemma":[0.00008099771,0.00019921853,0.5973725,0.0008433311,0.00012714347,0.0055195456,0.13397735,0.0036570572,0.014823068,0.0023746337,0.24090047,0.00012471461],"about_ca_topic_score_codex":0.009103029,"about_ca_topic_score_gemma":0.022765828,"teacher_disagreement_score":0.009103029,"about_ca_system_score_codex":0.0010254063,"about_ca_system_score_gemma":0.0007116531,"threshold_uncertainty_score":0.018100083},"labels":[],"label_agreement":null},{"id":"W2977000626","doi":"10.1007/978-3-030-32236-6_76","title":"Overview of the NLPCC 2019 Shared Task: Open Domain Conversation Evaluation","year":2019,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Blackberry (Canada)","funders":"","keywords":"Computer science; Conversation; Task (project management); Domain (mathematical analysis); Open domain; Human–computer interaction; Artificial intelligence; Programming language; Systems engineering; Linguistics; Question answering","score_opus":0.036854881994546826,"score_gpt":0.3154158096821468,"score_spread":0.27856092768759994,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2977000626","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.055029467,0.023455689,0.44084865,0.00781,0.0041212365,0.015277604,0.16401887,0.14338943,0.14604913],"genre_scores_gemma":[0.11995817,0.002889913,0.36474493,0.0020467057,0.0009644299,0.012754993,0.43101677,0.011992553,0.053631537],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9789256,0.010675551,0.001185029,0.0030765529,0.0047947685,0.0013425668],"domain_scores_gemma":[0.98716134,0.0040187235,0.0002493007,0.002453623,0.0044854875,0.0016316416],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.015910087,0.0040709903,0.0032674617,0.005317968,0.004506465,0.0064127045,0.006164842,0.0041733487,0.035274576],"category_scores_gemma":[0.024428172,0.0015256657,0.0017894334,0.0042436863,0.001467778,0.0076192017,0.012245681,0.0058536557,0.042114053],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015275907,0.0013998556,0.00170042,0.0023446567,0.00032913729,0.0002029049,0.0010679085,0.0064876773,0.01472502,0.0054756515,0.47276142,0.49197778],"study_design_scores_gemma":[0.0011882359,0.0016566771,0.009428459,0.0009754109,0.00036798546,0.0009873856,0.0024104728,0.16538753,0.049823042,0.02905546,0.7380636,0.00065570266],"about_ca_topic_score_codex":0.02888783,"about_ca_topic_score_gemma":0.032098375,"teacher_disagreement_score":0.035274576,"about_ca_system_score_codex":0.003968615,"about_ca_system_score_gemma":0.007813534,"threshold_uncertainty_score":0.11800516},"labels":[],"label_agreement":null},{"id":"W29771567","doi":"10.1037/pspa0000113","title":"Lexicography in an Interlingual Ontology: An Introduction to EuroWordNet","year":2004,"lang":"en","type":"article","venue":"Journal of Personality and Social Psychology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Social Sciences and Humanities Research Council of Canada; National Natural Science Foundation of China","keywords":"WordNet; Lexicon; Computer science; Lexical database; Ontology; Natural language processing; Artificial intelligence; Lexical item; Lexical semantics; Point (geometry); Linguistics; Information retrieval; Mathematics","score_opus":0.03452074635652499,"score_gpt":0.38404194820190507,"score_spread":0.3495212018453801,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W29771567","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0042613805,0.0131195495,0.9383125,0.0034453068,0.0016327348,0.00034999612,0.0018632588,0.0029575329,0.034057822],"genre_scores_gemma":[0.051592506,0.016143046,0.90236956,0.0019458354,0.0014391968,0.0011967758,0.005545717,0.002051533,0.017715765],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99786466,0.0008046865,0.00051974144,0.00035102275,0.00036219633,0.00009776368],"domain_scores_gemma":[0.99820185,0.0009699157,0.00017710285,0.00030199243,0.00022772835,0.00012136573],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029564647,0.0012216148,0.0014777087,0.007209348,0.0015165774,0.006345367,0.0023544692,0.00237748,0.019265478],"category_scores_gemma":[0.0053644436,0.0011067494,0.002473869,0.011288803,0.0031328218,0.014850562,0.0039743646,0.003568072,0.006595303],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000081026534,0.0000904156,0.00070332753,0.0011078849,0.000067622255,0.00056182285,0.0010825114,0.004404345,0.0016691457,0.6761556,0.02093135,0.2931449],"study_design_scores_gemma":[0.000025067588,0.00003704488,0.0008613529,0.0007833984,0.000022437742,0.00083604304,0.00048802176,0.016216854,0.000917789,0.44016272,0.5395816,0.0000676557],"about_ca_topic_score_codex":0.004557312,"about_ca_topic_score_gemma":0.0047333734,"teacher_disagreement_score":0.019265478,"about_ca_system_score_codex":0.0021323063,"about_ca_system_score_gemma":0.001736087,"threshold_uncertainty_score":0.06444943},"labels":[],"label_agreement":null},{"id":"W2977506914","doi":"10.5220/0008124300800087","title":"A New Data Structure for Processing Natural Language Database Queries","year":2019,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Windsor","funders":"","keywords":"Computer science; Transitive relation; Natural language processing; Denotation (semiotics); Query language; Semantics (computer science); RDF query language; Artificial intelligence; Web query classification; Linguistics; Database; Web search query; Information retrieval; Programming language; Mathematics","score_opus":0.014897143188045869,"score_gpt":0.3051602631309952,"score_spread":0.29026311994294934,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2977506914","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0022155198,0.0004331198,0.9764366,0.0009172321,0.00022834084,0.0003412365,0.0020583887,0.014359956,0.003009623],"genre_scores_gemma":[0.048006132,0.00061833026,0.92823845,0.0019105419,0.00047838423,0.0010653011,0.008313457,0.0025509372,0.008818482],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9933674,0.0010691595,0.0015876384,0.0014428074,0.0022161575,0.000316788],"domain_scores_gemma":[0.99251395,0.0021261675,0.0004781362,0.002982185,0.0015865513,0.0003130811],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0061592027,0.0011813832,0.0023078015,0.005147866,0.0019799175,0.0071343533,0.003964435,0.0020118917,0.012180045],"category_scores_gemma":[0.014116885,0.0016942137,0.0034270787,0.005343938,0.0029632244,0.020000784,0.008263059,0.004293721,0.0063888505],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006998003,0.00024895294,0.002441132,0.0009282127,0.00016268587,0.00041346755,0.0021370791,0.0050444882,0.015303321,0.64499664,0.05595858,0.27166572],"study_design_scores_gemma":[0.0002634437,0.00036948192,0.00086653174,0.00029759886,0.0001783032,0.00082816277,0.00052542373,0.09143579,0.022564126,0.42375723,0.45866385,0.00024995342],"about_ca_topic_score_codex":0.004825493,"about_ca_topic_score_gemma":0.0061112377,"teacher_disagreement_score":0.012180045,"about_ca_system_score_codex":0.0027394642,"about_ca_system_score_gemma":0.002783477,"threshold_uncertainty_score":0.04074633},"labels":[],"label_agreement":null},{"id":"W2977842567","doi":"","title":"Chinese fK Automatic Recognition Based on HNC Theory","year":2009,"lang":"en","type":"article","venue":"Microcomputer applications","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Parsing; Natural language processing; Artificial intelligence; Natural language; Semantic computing; Semantic analysis (machine learning); Natural language understanding; Semantic role labeling; Semantics (computer science); Programming language; Semantic Web","score_opus":0.00672590814852891,"score_gpt":0.2672453461620855,"score_spread":0.2605194380135566,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2977842567","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11779895,0.00076070696,0.83576494,0.0003454469,0.0002632567,0.00022595798,0.0018405457,0.006890038,0.036110163],"genre_scores_gemma":[0.65240896,0.0006017748,0.31835717,0.00020134068,0.0001472603,0.00021006857,0.0038509453,0.0003837006,0.023838779],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99950147,0.000051022467,0.00003769178,0.00016705421,0.00017668368,0.00006611305],"domain_scores_gemma":[0.99954563,0.00008668901,0.00002811298,0.00008809431,0.00023769298,0.000013765637],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00029229687,0.00039869169,0.00046604415,0.0018394977,0.00074885745,0.00052091683,0.0006076064,0.00031493208,0.0055752387],"category_scores_gemma":[0.00086138723,0.00018294028,0.0005733081,0.0010340662,0.00060165726,0.0009979251,0.00034375506,0.00034625444,0.0015759615],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020533736,0.000079559075,0.007268787,0.0003215768,0.00006183258,0.0006577046,0.00052923145,0.0325203,0.058382154,0.09848299,0.025005044,0.77648544],"study_design_scores_gemma":[0.00005206195,0.00007278047,0.012080047,0.000038049544,0.00012764751,0.0007172148,0.00017971285,0.8427296,0.056211244,0.052438807,0.035239037,0.00011382898],"about_ca_topic_score_codex":0.023000622,"about_ca_topic_score_gemma":0.0151874935,"teacher_disagreement_score":0.023000622,"about_ca_system_score_codex":0.0008155622,"about_ca_system_score_gemma":0.0012301883,"threshold_uncertainty_score":0.04573351},"labels":[],"label_agreement":null},{"id":"W2978957403","doi":"10.5220/0008319301550162","title":"Vocab Learn: A Text Mining System to Assist Vocabulary Learning","year":2019,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Word (group theory); Ranking (information retrieval); Vocabulary; Natural language processing; Knowledge base; Artificial intelligence; Set (abstract data type); Base (topology); Word lists by frequency; Information retrieval; tf–idf; Levenshtein distance; Zipf's law; Sentence; Linguistics; Mathematics; Term (time)","score_opus":0.00754809350283542,"score_gpt":0.24257142086962608,"score_spread":0.23502332736679066,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2978957403","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.043459173,0.0028349252,0.50717866,0.0012470459,0.0005655432,0.0019385496,0.108306006,0.31836557,0.016104626],"genre_scores_gemma":[0.085897274,0.001232076,0.7021836,0.00096646824,0.00018914635,0.0016488218,0.18973967,0.006152155,0.011990801],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989374,0.00020580066,0.00015440902,0.00038293214,0.00026384633,0.000055508313],"domain_scores_gemma":[0.9968208,0.0017161613,0.0002617813,0.00046017455,0.00062314305,0.00011786463],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016468437,0.0016651002,0.0010119337,0.0050799213,0.0005489923,0.0014749251,0.0022228283,0.00097237894,0.012272298],"category_scores_gemma":[0.009367153,0.00067871937,0.0010205124,0.0024012043,0.000326692,0.0049637333,0.0023664695,0.0010361653,0.009660745],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006268065,0.0005405722,0.008515911,0.0025759477,0.00026372887,0.00046630512,0.00068087893,0.0062197917,0.024371935,0.0059756823,0.24526134,0.70450115],"study_design_scores_gemma":[0.0006759116,0.0010049151,0.007258417,0.0006808115,0.0004134146,0.0012795089,0.000842583,0.37075594,0.08027621,0.03403411,0.5025257,0.0002525449],"about_ca_topic_score_codex":0.0026419368,"about_ca_topic_score_gemma":0.0067063943,"teacher_disagreement_score":0.012272298,"about_ca_system_score_codex":0.00073964096,"about_ca_system_score_gemma":0.0014608073,"threshold_uncertainty_score":0.041054904},"labels":[],"label_agreement":null},{"id":"W2980183677","doi":"10.1007/978-3-030-32520-6_6","title":"Machine Translation from Natural Language to Code Using Long-Short Term Memory","year":2019,"lang":"en","type":"book-chapter","venue":"Advances in intelligent systems and computing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University; University of Calgary","funders":"","keywords":"Programmer; Language primitive; First-generation programming language; Programming language implementation; Very high-level programming language; Natural language programming; Machine translation; High-level programming language; Fifth-generation programming language; Programming paradigm","score_opus":0.020691207709983367,"score_gpt":0.30439818892663945,"score_spread":0.28370698121665605,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2980183677","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.030175295,0.0034802537,0.8724237,0.0016387712,0.0018087279,0.00027105288,0.0047513256,0.025574842,0.05987602],"genre_scores_gemma":[0.18784675,0.0036196338,0.72778845,0.00060629915,0.00037561945,0.00034496616,0.015297909,0.004527585,0.05959285],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99969816,0.000059520058,0.00003520637,0.000076462886,0.00010556055,0.000025154211],"domain_scores_gemma":[0.99910885,0.00036228204,0.00005499551,0.00016625304,0.00029103115,0.000016518727],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00024965729,0.0007620832,0.0005227921,0.00090233184,0.0005106131,0.0014527612,0.00076195586,0.0006363959,0.012485507],"category_scores_gemma":[0.0017433207,0.00031977074,0.00059295207,0.001427731,0.00043054702,0.0016235034,0.00079436396,0.0009895539,0.009446685],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012639751,0.000108541215,0.00037521883,0.0010391511,0.000054615237,0.00046837944,0.0003870727,0.010277119,0.03608529,0.046470314,0.070303336,0.8343046],"study_design_scores_gemma":[0.000089350484,0.0002850033,0.0010358572,0.00043329812,0.00016972811,0.0014091663,0.000578913,0.25911012,0.16979389,0.13628806,0.4307107,0.00009589849],"about_ca_topic_score_codex":0.0017699281,"about_ca_topic_score_gemma":0.0023632678,"teacher_disagreement_score":0.012485507,"about_ca_system_score_codex":0.0004882295,"about_ca_system_score_gemma":0.000995643,"threshold_uncertainty_score":0.041768193},"labels":[],"label_agreement":null},{"id":"W2980478932","doi":"10.1146/annurev-linguistics-011619-030452","title":"Techniques in Complex Semantic Fieldwork","year":2019,"lang":"en","type":"article","venue":"Annual Review of Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":79,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Storyboard; Ambiguity; Computer science; Context (archaeology); Natural language; Natural language processing; Linguistics; Artificial intelligence; History; Philosophy; Multimedia","score_opus":0.014680810755669634,"score_gpt":0.3240125467029751,"score_spread":0.30933173594730545,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2980478932","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0030272955,0.0008214649,0.98709726,0.00066861836,0.000076507815,0.00013281577,0.00009494975,0.0004743084,0.0076067774],"genre_scores_gemma":[0.0805503,0.0009496904,0.91184604,0.00027826615,0.0002152028,0.0006821283,0.00036540645,0.0004452948,0.0046675624],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.98809475,0.0075257747,0.00070739014,0.0020780133,0.0012383375,0.00035575713],"domain_scores_gemma":[0.9747297,0.017502042,0.0010019516,0.0057087988,0.0007663172,0.0002910921],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012865908,0.0017976244,0.0014361604,0.0044209696,0.0029570083,0.0073262383,0.00365152,0.002773548,0.019884096],"category_scores_gemma":[0.028948335,0.0015700654,0.002717061,0.0042381124,0.0124320015,0.017394563,0.010126027,0.0042383075,0.0037197485],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008785143,0.000060417537,0.00039057538,0.0005250181,0.000044567074,0.00031976134,0.007439356,0.0028596125,0.0026949444,0.849231,0.0031859358,0.1331609],"study_design_scores_gemma":[0.000022880855,0.000016094718,0.0001485375,0.00011058979,0.000008269457,0.0002074411,0.0010261841,0.0072688246,0.0010712609,0.95374936,0.036350068,0.000020368187],"about_ca_topic_score_codex":0.000986517,"about_ca_topic_score_gemma":0.00113882,"teacher_disagreement_score":0.019884096,"about_ca_system_score_codex":0.0022934952,"about_ca_system_score_gemma":0.0015774777,"threshold_uncertainty_score":0.06804228},"labels":[],"label_agreement":null},{"id":"W2980662877","doi":"10.1002/pra2.108","title":"Machine translation literacy: Academic libraries' role","year":2019,"lang":"en","type":"article","venue":"Proceedings of the Association for Information Science and Technology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Literacy; Machine translation; Information literacy; Computer science; Academic library; Quality (philosophy); Translation (biology); Intervention (counseling); Library science; Mathematics education; Sociology; Psychology; Pedagogy; Artificial intelligence; Chemistry","score_opus":0.005233977429844764,"score_gpt":0.24255409204646414,"score_spread":0.23732011461661937,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2980662877","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09662881,0.016221594,0.0057817632,0.5173224,0.001716425,0.00017247153,0.00030808157,0.0023675212,0.35948086],"genre_scores_gemma":[0.91477484,0.006032152,0.0033213387,0.022092247,0.0016669019,0.000117719,0.00015862605,0.0003537973,0.051482376],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.968999,0.02083686,0.0012210541,0.0012589715,0.003935369,0.0037487235],"domain_scores_gemma":[0.87413013,0.045708008,0.00939354,0.005555837,0.026458945,0.038753577],"candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.033459898,0.0005215881,0.0007958069,0.004247268,0.0133590065,0.046114355,0.002942505,0.0038296082,0.041419424],"category_scores_gemma":[0.057789627,0.000498144,0.0003038412,0.0058733267,0.009690288,0.017217472,0.018453108,0.005028347,0.008689222],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004462976,0.0011012744,0.03522103,0.0015505656,0.000043218526,0.00097541645,0.0703642,0.00028468977,0.0009207242,0.13945043,0.26236898,0.48727316],"study_design_scores_gemma":[0.00012509584,0.00049624447,0.034661625,0.0031435017,0.00007684607,0.0014734709,0.14945567,0.0018904194,0.0041490667,0.058254775,0.74605685,0.00021638158],"about_ca_topic_score_codex":0.009888255,"about_ca_topic_score_gemma":0.0081529245,"teacher_disagreement_score":0.9538857,"about_ca_system_score_codex":0.010106732,"about_ca_system_score_gemma":0.040390685,"threshold_uncertainty_score":0.17695498},"labels":[],"label_agreement":null},{"id":"W2980941473","doi":"10.18653/v1/w19-5351","title":"Linguistic Evaluation of German-English Machine Translation Using a Test Suite","year":2019,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Banting and Best Diabetes Centre, University of Toronto; Bundesministerium für Bildung und Forschung","keywords":"German; Valency; Computer science; Punctuation; Linguistics; Natural language processing; Test suite; Verb; Suite; Artificial intelligence; Test (biology); Machine translation; Test case; History","score_opus":0.056887177927362184,"score_gpt":0.3638439371814072,"score_spread":0.306956759254045,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2980941473","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.97223943,0.00097341253,0.013845342,0.00027713456,0.00013487268,0.00026282523,0.003981008,0.0042208172,0.0040652235],"genre_scores_gemma":[0.9264615,0.00045137547,0.026134703,0.00016662727,0.00011764068,0.00043226092,0.040615708,0.00128164,0.00433851],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9836451,0.011061537,0.0015718578,0.0011768843,0.001968185,0.00057646906],"domain_scores_gemma":[0.9659515,0.02061754,0.0012284963,0.0031731173,0.00784123,0.0011880209],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007834214,0.0023909905,0.0013027218,0.003758762,0.0011136004,0.0015968059,0.0013925962,0.0014568106,0.0027803364],"category_scores_gemma":[0.026274424,0.00052047736,0.0010786827,0.0032107227,0.00075975707,0.0010483409,0.0019964962,0.000916367,0.0022784306],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.008546036,0.008467501,0.06841766,0.004475452,0.00319482,0.006294451,0.007796585,0.13724053,0.14922266,0.0034338294,0.052043337,0.55086714],"study_design_scores_gemma":[0.002812764,0.015285338,0.21617554,0.0004044883,0.0019342327,0.0054003336,0.004048725,0.4141931,0.27421466,0.0025478664,0.062489256,0.0004937606],"about_ca_topic_score_codex":0.0047540017,"about_ca_topic_score_gemma":0.0046060486,"teacher_disagreement_score":0.007834214,"about_ca_system_score_codex":0.0009987653,"about_ca_system_score_gemma":0.0008827948,"threshold_uncertainty_score":0.041431785},"labels":[],"label_agreement":null},{"id":"W2981225511","doi":"10.5539/ijel.v9n6p125","title":"Near-Synonyms Within the Same Qur’anic Verse: A Contrastive English-Arabic Lexical Analysis","year":2019,"lang":"en","type":"article","venue":"International Journal of English Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Linguistics; Arabic; Variety (cybernetics); Meaning (existential); Face (sociological concept); Repetition (rhetorical device); Contrastive analysis; Psychology; Computer science; Philosophy; Artificial intelligence","score_opus":0.008966563169421566,"score_gpt":0.27639107885627184,"score_spread":0.2674245156868503,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2981225511","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.92570275,0.000898151,0.010858304,0.00063909945,0.00007485766,0.00019782985,0.0003355159,0.000029120692,0.06126444],"genre_scores_gemma":[0.9930942,0.00017758652,0.0044037905,0.00006300827,0.000028906581,0.000096442694,0.00020932332,0.000030215904,0.0018964995],"study_design_codex":"qualitative","study_design_gemma":"not_applicable","domain_scores_codex":[0.99850535,0.0007724424,0.00013250973,0.00016173802,0.00032337068,0.0001045751],"domain_scores_gemma":[0.9934684,0.0048973793,0.0004015956,0.00016295479,0.0009639549,0.000105714804],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014379238,0.00039882035,0.00028546294,0.0037174076,0.0028395744,0.003556649,0.0005018136,0.00044472152,0.0063344445],"category_scores_gemma":[0.0068291794,0.00022030367,0.0003032812,0.0030580056,0.002752517,0.0033366738,0.0018544331,0.0013137921,0.0007938683],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017381756,0.00082367554,0.08011234,0.0016800393,0.00015122618,0.01009561,0.4360811,0.00053905905,0.062947564,0.26444548,0.006325035,0.13506073],"study_design_scores_gemma":[0.0001960496,0.0006719882,0.23059554,0.0010971574,0.00037090597,0.007860614,0.55500346,0.014315363,0.022804331,0.045318972,0.12159765,0.0001679496],"about_ca_topic_score_codex":0.003964243,"about_ca_topic_score_gemma":0.0053147534,"teacher_disagreement_score":0.0063344445,"about_ca_system_score_codex":0.0016969305,"about_ca_system_score_gemma":0.00080734934,"threshold_uncertainty_score":0.021190882},"labels":[],"label_agreement":null},{"id":"W2982290345","doi":"10.5539/ijel.v9n6p226","title":"The Analysis of Verbless Sentences","year":2019,"lang":"en","type":"article","venue":"International Journal of English Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"Deanship of Scientific Research, Prince Sattam bin Abdulaziz University; Prince Sattam bin Abdulaziz University","keywords":"Adjective; Predicate (mathematical logic); Complement (music); Computer science; Linguistics; Natural language processing; Sentence; Artificial intelligence; Copula (linguistics); Grammar; Parsing; Noun; Philosophy; Programming language","score_opus":0.008544295026062155,"score_gpt":0.2868153322851585,"score_spread":0.27827103725909635,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2982290345","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.26981604,0.001383154,0.69329864,0.00127603,0.00018992856,0.00017663624,0.00082154566,0.0011775345,0.03186051],"genre_scores_gemma":[0.915434,0.0003448182,0.07916483,0.0002504186,0.00012972877,0.00007955323,0.0006340961,0.00042752066,0.0035349878],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99890697,0.0004657607,0.0000902895,0.0002159724,0.00023329613,0.00008774324],"domain_scores_gemma":[0.9982058,0.00079365604,0.00025820392,0.00023479084,0.0004562265,0.000051305313],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009511626,0.0006179167,0.00032351204,0.0026132888,0.0010831559,0.0019803387,0.000633142,0.0006158723,0.0046173437],"category_scores_gemma":[0.0036858858,0.00023421552,0.0007441534,0.0012302996,0.0022586188,0.0035328886,0.0010113067,0.0008560471,0.0006806153],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015360046,0.00005229253,0.005477459,0.00067204295,0.0000851878,0.002506409,0.009781818,0.0035999545,0.076632634,0.78635496,0.003929139,0.11075447],"study_design_scores_gemma":[0.000025617439,0.0001287058,0.015173084,0.00021677233,0.0001363004,0.0033302167,0.005321618,0.05441313,0.039120946,0.7838795,0.09814604,0.00010811728],"about_ca_topic_score_codex":0.00080082845,"about_ca_topic_score_gemma":0.00037157084,"teacher_disagreement_score":0.0046173437,"about_ca_system_score_codex":0.00080476317,"about_ca_system_score_gemma":0.0004887499,"threshold_uncertainty_score":0.015446544},"labels":[],"label_agreement":null},{"id":"W2982500662","doi":"10.4000/discours.10032","title":"Multiple Signals of Coherence Relations","year":2019,"lang":"en","type":"article","venue":"Discours","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Coherence (philosophical gambling strategy); Relation (database); Discourse marker; SIGNAL (programming language); Linguistics; Computer science; Mathematics; Data mining; Statistics","score_opus":0.011201616507368348,"score_gpt":0.2749923360209535,"score_spread":0.2637907195135852,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2982500662","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.62575895,0.0035107045,0.30619875,0.0029951741,0.00045042948,0.00030436745,0.0015239374,0.00116822,0.058089484],"genre_scores_gemma":[0.9565931,0.00037815113,0.038101442,0.00018852868,0.00013365013,0.00013102849,0.0005004346,0.00023126545,0.0037423654],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99296004,0.0022482031,0.0005274699,0.0016436861,0.0021789146,0.0004416601],"domain_scores_gemma":[0.96022826,0.026546538,0.005455321,0.0037136604,0.0032694458,0.00078671623],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036120212,0.0006719204,0.00051597314,0.003947682,0.0017805531,0.004075869,0.0009195085,0.0016345326,0.0067201494],"category_scores_gemma":[0.037271567,0.0006276018,0.00044132106,0.0034457778,0.0029404394,0.007893672,0.004118389,0.0021570257,0.00095458754],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0022877431,0.00018936182,0.05056062,0.0024362446,0.00024847538,0.0035955983,0.09559778,0.0025171177,0.18079713,0.3677403,0.0046857754,0.28934386],"study_design_scores_gemma":[0.00030010304,0.0014255646,0.16481705,0.0020127443,0.0007606141,0.008582139,0.061105866,0.045042444,0.1266235,0.3473104,0.24120578,0.00081380334],"about_ca_topic_score_codex":0.0015563035,"about_ca_topic_score_gemma":0.0017885026,"teacher_disagreement_score":0.0067201494,"about_ca_system_score_codex":0.0011039454,"about_ca_system_score_gemma":0.0008763161,"threshold_uncertainty_score":0.022481143},"labels":[],"label_agreement":null},{"id":"W2983791227","doi":"10.33011/computel.v1i.421","title":"A Preliminary Plains Cree Speech Synthesizer","year":2019,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Sentence; Natural language processing; Task (project management); Identification (biology); Nonsense; Variation (astronomy); Linguistics; Indigenous; Speech synthesis; Artificial intelligence; Speech recognition; Engineering","score_opus":0.008585136888335001,"score_gpt":0.2442791576832241,"score_spread":0.2356940207948891,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2983791227","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7397489,0.00047422072,0.20481819,0.00039504748,0.00019206002,0.0040284586,0.0023021053,0.008739144,0.039301954],"genre_scores_gemma":[0.6826569,0.0003123933,0.23890187,0.00018488172,0.000066584405,0.0014259893,0.0048042447,0.0012872817,0.070359804],"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99872893,0.00023820854,0.00009520824,0.0002882601,0.00054352963,0.00010585935],"domain_scores_gemma":[0.9983565,0.0004841353,0.00003882566,0.00015153617,0.00081433327,0.00015470812],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001911199,0.00055383367,0.00050458766,0.00043604104,0.0006602949,0.00084253075,0.0006539209,0.00065253937,0.012866975],"category_scores_gemma":[0.0036176543,0.0002919401,0.00026863263,0.00018524128,0.00046342894,0.0008718279,0.00084255286,0.0006757356,0.00340023],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012277424,0.00044379293,0.00276397,0.0005047513,0.00005449609,0.0011314511,0.00468293,0.0040538963,0.76569366,0.0018430781,0.00284435,0.21475597],"study_design_scores_gemma":[0.00059893937,0.008473629,0.031735733,0.00013097787,0.00020750613,0.0023910555,0.002399916,0.04764534,0.77213824,0.0007149557,0.13335057,0.00021324217],"about_ca_topic_score_codex":0.0049805464,"about_ca_topic_score_gemma":0.005870615,"teacher_disagreement_score":0.012866975,"about_ca_system_score_codex":0.00048994634,"about_ca_system_score_gemma":0.00078547804,"threshold_uncertainty_score":0.04304433},"labels":[],"label_agreement":null},{"id":"W2984256198","doi":"10.18653/v1/d19-6115","title":"Unlearn Dataset Bias in Natural Language Inference by Fitting the Residual","year":2019,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":165,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"York University","keywords":"Debiasing; Residual; Computer science; Artificial intelligence; Inference; Machine learning; Benchmark (surveying); Natural language processing; Baseline (sea); Algorithm; Psychology","score_opus":0.017414129089580985,"score_gpt":0.3055379553608717,"score_spread":0.2881238262712907,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2984256198","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.092936546,0.0013465623,0.890214,0.0015917644,0.00013866296,0.00021776414,0.00095579855,0.010656665,0.0019422189],"genre_scores_gemma":[0.6014112,0.00034261766,0.38738018,0.0019636487,0.00015372032,0.00041858308,0.0048624906,0.000994792,0.0024727525],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9953675,0.0019394243,0.00029955246,0.001504584,0.00062413065,0.00026473403],"domain_scores_gemma":[0.9790446,0.012188671,0.0013382537,0.0055256807,0.0015833309,0.0003195026],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012887371,0.0016758012,0.001366721,0.0015748987,0.0007991167,0.0019265779,0.0034803452,0.0024144,0.0018906126],"category_scores_gemma":[0.044488914,0.0009059005,0.001561196,0.0013517599,0.0024495907,0.0053831814,0.0038103233,0.005803228,0.0015639128],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009166778,0.00043613094,0.03897644,0.00059125933,0.0005151083,0.0003453958,0.0010465345,0.45914572,0.015626444,0.025343154,0.018599879,0.4384573],"study_design_scores_gemma":[0.000056376033,0.00013075433,0.0014896216,0.000058873185,0.000038539787,0.00013147207,0.000083627885,0.96332264,0.0055507533,0.026685718,0.00241607,0.000035535442],"about_ca_topic_score_codex":0.00516036,"about_ca_topic_score_gemma":0.0087188445,"teacher_disagreement_score":0.012887371,"about_ca_system_score_codex":0.0018365872,"about_ca_system_score_gemma":0.0021009052,"threshold_uncertainty_score":0.068155766},"labels":[],"label_agreement":null},{"id":"W2984285789","doi":"10.18653/v1/d19-6308","title":"The Concordia NLG Surface Realizer at SRST 2019","year":2019,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Pointer (user interface); Transformer; Encoder; Word order; Artificial intelligence; Natural language processing; Task (project management); Realization (probability); Speech recognition; Engineering; Mathematics","score_opus":0.006283303890126029,"score_gpt":0.25021094742379163,"score_spread":0.2439276435336656,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2984285789","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1916265,0.00039855952,0.6011064,0.0025297967,0.00059430295,0.0009356057,0.014920292,0.13623244,0.05165608],"genre_scores_gemma":[0.6769137,0.0001365637,0.2562897,0.0006104255,0.00010102771,0.0006012157,0.024413457,0.004575723,0.036358215],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9990583,0.0001925026,0.000044988235,0.00031204626,0.00026397317,0.00012819745],"domain_scores_gemma":[0.99953353,0.0001521234,0.00002270129,0.00012172668,0.00013234178,0.000037555634],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001270588,0.0009503605,0.00084202987,0.0004785967,0.0006970785,0.0011766545,0.0018268245,0.0017724313,0.01278958],"category_scores_gemma":[0.002247318,0.0005881363,0.000901479,0.00040850666,0.0007784769,0.0024752966,0.0017118615,0.0017767857,0.0096711805],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0030535932,0.0005503297,0.0066849245,0.00085934834,0.0001965966,0.0034285684,0.0020766493,0.14954837,0.09972969,0.10285009,0.17440492,0.45661703],"study_design_scores_gemma":[0.00023951668,0.00035154686,0.001047213,0.00002504113,0.000040878076,0.00037272536,0.00019163563,0.8846852,0.045773614,0.015006364,0.052192226,0.00007413818],"about_ca_topic_score_codex":0.019938849,"about_ca_topic_score_gemma":0.019640429,"teacher_disagreement_score":0.019938849,"about_ca_system_score_codex":0.0019808128,"about_ca_system_score_gemma":0.002227371,"threshold_uncertainty_score":0.042785406},"labels":[],"label_agreement":null},{"id":"W2984511782","doi":"10.18653/v1/k19-1082","title":"Slang Detection and Identification","year":2019,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Slang; Computer science; Natural language processing; Artificial intelligence; Identification (biology); Sentence; Natural language; Noun; Feature (linguistics); Linguistics; Speech recognition","score_opus":0.005560888302506523,"score_gpt":0.23947519466414402,"score_spread":0.2339143063616375,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2984511782","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.48328826,0.0004567324,0.48745131,0.00043439757,0.00028843206,0.00050136173,0.0029380857,0.014870514,0.00977092],"genre_scores_gemma":[0.89902675,0.00008656992,0.094556294,0.00013988136,0.000037393176,0.00011910463,0.0024174117,0.00030620364,0.003310477],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9979925,0.00040945318,0.00017850501,0.00048048876,0.0005998156,0.0003392308],"domain_scores_gemma":[0.99514455,0.0012067291,0.0008934205,0.0007878797,0.0017624584,0.00020494731],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014951078,0.0007351419,0.000697565,0.0020901153,0.00050012395,0.0009646519,0.00077307655,0.0009495921,0.0033514523],"category_scores_gemma":[0.005748821,0.00024101186,0.00063109695,0.00073483295,0.0005633273,0.0016816108,0.0013312376,0.0009286175,0.0030970017],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000847427,0.00041290958,0.08388782,0.0010282756,0.00013768097,0.0017643106,0.0017178728,0.015705962,0.28228822,0.005870308,0.020921558,0.5854177],"study_design_scores_gemma":[0.000048142876,0.0004981652,0.06655066,0.00015726015,0.0000754205,0.0016050925,0.0014677205,0.7558267,0.14672332,0.0093842065,0.017540522,0.00012280777],"about_ca_topic_score_codex":0.0022880787,"about_ca_topic_score_gemma":0.0035549372,"teacher_disagreement_score":0.0033514523,"about_ca_system_score_codex":0.00037444796,"about_ca_system_score_gemma":0.0009791156,"threshold_uncertainty_score":0.011211693},"labels":[],"label_agreement":null},{"id":"W2985695201","doi":"10.2200/s00935ed1v02y201907hlt043","title":"Linguistic Fundamentals for Natural Language Processing II: 100 Essentials from Semantics and Pragmatics","year":2019,"lang":"en","type":"article","venue":"Synthesis lectures on human language technologies","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Pragmatics; Computer science; Linguistics; Semantics (computer science); Selection (genetic algorithm); Meaning (existential); Natural language processing; Natural language; Natural language understanding; Task (project management); Artificial intelligence; Deep linguistic processing; Computational linguistics; Natural (archaeology); Psychology; Philosophy","score_opus":0.014404604552958743,"score_gpt":0.3076386150326261,"score_spread":0.29323401047966735,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2985695201","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005984187,0.07414568,0.6208547,0.05167623,0.010197217,0.00012570873,0.0007273326,0.0010664798,0.23522243],"genre_scores_gemma":[0.34073058,0.07630874,0.34728193,0.012227605,0.030173918,0.0010229434,0.0018476377,0.0031342453,0.18727243],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9988356,0.0003911054,0.000101535654,0.00018845481,0.00041023587,0.000073098316],"domain_scores_gemma":[0.99826103,0.0011339065,0.000060797425,0.00020684996,0.00026665413,0.000070686365],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021673434,0.0010253794,0.0010860297,0.0018743889,0.0018540386,0.0064977836,0.0012952277,0.0020195723,0.011448483],"category_scores_gemma":[0.004963352,0.00096978893,0.0008089919,0.001755943,0.006668236,0.015371922,0.0022340934,0.0060796253,0.005620298],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00001174868,0.000022466855,0.00006271041,0.00015190261,0.0000060812117,0.000032509757,0.0005609867,0.00041729823,0.0005384009,0.9444659,0.02184134,0.031888515],"study_design_scores_gemma":[0.0000034748148,0.0000058726064,0.00009205552,0.000089873065,0.0000043415807,0.000039935396,0.00011101389,0.0009881698,0.000187158,0.9153562,0.083112076,0.000009712126],"about_ca_topic_score_codex":0.0011390651,"about_ca_topic_score_gemma":0.00079613307,"teacher_disagreement_score":0.011448483,"about_ca_system_score_codex":0.0023000818,"about_ca_system_score_gemma":0.0014050986,"threshold_uncertainty_score":0.038298965},"labels":[],"label_agreement":null},{"id":"W2985754245","doi":"10.26615/978-954-452-056-4_117","title":"Analysing the Impact of Supervised Machine Learning on Automatic Term Extraction: HAMLET vs TermoStat","year":2019,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"Vlaamse regering; Fonds Wetenschappelijk Onderzoek","keywords":"Computer science; Artificial intelligence; Term (time); Machine learning; Recall; Point (geometry); Precision and recall; Natural language processing; Pattern recognition (psychology); Mathematics","score_opus":0.011433824788806241,"score_gpt":0.3019232286953674,"score_spread":0.2904894039065612,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2985754245","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.83689684,0.0065354505,0.13353617,0.0005174169,0.0003603564,0.00049931346,0.0022641304,0.005539579,0.013850731],"genre_scores_gemma":[0.8894026,0.0009504564,0.09805082,0.00016059447,0.0001979232,0.00024500347,0.0060081864,0.0006230125,0.0043613175],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.989186,0.0053344485,0.00093040103,0.0012569444,0.0029019425,0.0003902589],"domain_scores_gemma":[0.9300077,0.05745006,0.0020810866,0.003964482,0.0059874197,0.000509256],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0103102075,0.0011023249,0.0013588344,0.004087136,0.0006810468,0.002097052,0.00084611087,0.0010856481,0.0017833675],"category_scores_gemma":[0.04018761,0.000328871,0.00080103445,0.0028555344,0.00088141207,0.0037282542,0.0018387344,0.0010189211,0.0019204846],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004328932,0.00058148964,0.039628964,0.0027252766,0.0006087659,0.0006108914,0.0013858627,0.032111354,0.08606791,0.0022208488,0.005647416,0.8240824],"study_design_scores_gemma":[0.00028746968,0.0043656984,0.13614257,0.00034198884,0.0010629387,0.002121777,0.0019233893,0.607662,0.21679148,0.0072003943,0.021713004,0.00038726642],"about_ca_topic_score_codex":0.0028552306,"about_ca_topic_score_gemma":0.004310224,"teacher_disagreement_score":0.0103102075,"about_ca_system_score_codex":0.0009319675,"about_ca_system_score_gemma":0.0009215096,"threshold_uncertainty_score":0.05452621},"labels":[],"label_agreement":null},{"id":"W2986265153","doi":"10.1162/coli_a_00367","title":"On the Linguistic Representational Power of Neural Machine Translation Models","year":2020,"lang":"en","type":"preprint","venue":"Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Computer science; Natural language processing; Machine translation; Artificial intelligence; Interpretability; Semantics (computer science); Rule-based machine translation; Word (group theory); Contrast (vision); Linguistics","score_opus":0.05564198337220849,"score_gpt":0.3233539129337652,"score_spread":0.2677119295615567,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2986265153","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14155184,0.0039680544,0.83159286,0.008338322,0.0001519323,0.000050806324,0.0006413536,0.0019731687,0.011731599],"genre_scores_gemma":[0.910987,0.0018447106,0.082507536,0.0006262787,0.00018196789,0.00008072705,0.0007805653,0.00022319617,0.0027679913],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9983962,0.00092781556,0.00008252268,0.00025503643,0.0002538452,0.00008459042],"domain_scores_gemma":[0.9872896,0.009721479,0.0005635863,0.0014478988,0.0008556562,0.000121727615],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0042717378,0.000984534,0.00072946405,0.0009573865,0.0005879973,0.0025446222,0.001313966,0.0015737354,0.0021771633],"category_scores_gemma":[0.028945982,0.0006340668,0.0007393414,0.0011066487,0.001836443,0.0050288453,0.0016316754,0.0029753058,0.0007315818],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025068253,0.00007495966,0.002945811,0.0002172839,0.0001474907,0.00014618531,0.00040061362,0.7520987,0.005330381,0.1041379,0.0022800483,0.13196997],"study_design_scores_gemma":[0.0000079610745,0.00002995369,0.00035614966,0.000036733363,0.000018075752,0.000027780681,0.000028822338,0.9441421,0.0010956395,0.053554654,0.00069115986,0.0000109517105],"about_ca_topic_score_codex":0.0050890273,"about_ca_topic_score_gemma":0.0048038354,"teacher_disagreement_score":0.0050890273,"about_ca_system_score_codex":0.0012967853,"about_ca_system_score_gemma":0.0007839593,"threshold_uncertainty_score":0.022591352},"labels":[],"label_agreement":null},{"id":"W2986280923","doi":"10.26615/978-954-452-056-4_006","title":"Multilingual Sentence-Level Bias Detection inWikipedia","year":2019,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; Computer Research Institute of Montréal","funders":"","keywords":"Bulgarian; Computer science; Point (geometry); Exploit; Sentence; Natural language processing; Code (set theory); Artificial intelligence; Information retrieval; Linguistics; Programming language; Mathematics","score_opus":0.0386665625107681,"score_gpt":0.2841131196151448,"score_spread":0.24544655710437668,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2986280923","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6659199,0.007093772,0.23236509,0.00092157675,0.00062733766,0.00074604084,0.052155044,0.030348597,0.009822613],"genre_scores_gemma":[0.67612714,0.0011964556,0.22524847,0.00038388508,0.00027813783,0.0008669947,0.08906858,0.0021799842,0.0046503735],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99671245,0.0006801825,0.00040118422,0.0011173687,0.0007805231,0.00030836876],"domain_scores_gemma":[0.9910561,0.0030493748,0.0013171765,0.001451754,0.0028284674,0.00029709697],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023947274,0.0013785457,0.00093356275,0.008316557,0.0011781697,0.0018018924,0.0010165684,0.00093020656,0.0013446036],"category_scores_gemma":[0.014670208,0.0005850041,0.00072920724,0.004716755,0.00067789,0.0028845433,0.0025526711,0.0010117461,0.0020444463],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001080926,0.00033508576,0.14380269,0.004039582,0.0008181238,0.0037481594,0.0075607644,0.008698512,0.111333065,0.005336583,0.08881938,0.62442714],"study_design_scores_gemma":[0.00019810592,0.00045603988,0.31778312,0.00071560644,0.00085822825,0.0065252814,0.005719626,0.21984269,0.19880608,0.021241374,0.22726376,0.00059010147],"about_ca_topic_score_codex":0.011451272,"about_ca_topic_score_gemma":0.021411143,"teacher_disagreement_score":0.011451272,"about_ca_system_score_codex":0.0011867745,"about_ca_system_score_gemma":0.0018141858,"threshold_uncertainty_score":0.022769213},"labels":[],"label_agreement":null},{"id":"W2986463783","doi":"10.16995/dm.83","title":"Data-Driven Syllabification for Middle Dutch","year":2019,"lang":"en","type":"article","venue":"Digital Medievalist","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Syllabification; Computer science; Speech recognition","score_opus":0.04931733786985358,"score_gpt":0.30194990793541493,"score_spread":0.25263257006556133,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2986463783","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.23314372,0.0011796729,0.733214,0.0005643496,0.00036784474,0.00021449721,0.0031539076,0.01892035,0.0092417365],"genre_scores_gemma":[0.6298722,0.0002913071,0.3573755,0.0001251147,0.000048886544,0.00015318501,0.00505035,0.00074462965,0.0063388543],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994653,0.000059648693,0.000060701845,0.00023181483,0.0001277286,0.00005485053],"domain_scores_gemma":[0.99935406,0.00022387314,0.0000563263,0.00010741623,0.0002166716,0.00004159127],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000548776,0.00054616586,0.00062604353,0.0010338692,0.0005448685,0.0011346175,0.00098257,0.00062256504,0.0040046633],"category_scores_gemma":[0.0024030346,0.00039737785,0.0004760453,0.000701418,0.0003535756,0.0013485253,0.0011205955,0.0010532351,0.002288659],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003301221,0.00008457861,0.0025370608,0.00034189018,0.00007087961,0.0004836414,0.0005200721,0.014269138,0.17729995,0.0038280466,0.006426509,0.79380804],"study_design_scores_gemma":[0.000046763194,0.000078710604,0.004729025,0.000042754687,0.00005122139,0.0003735781,0.00025372452,0.8228986,0.14428495,0.0095030945,0.017663367,0.00007414075],"about_ca_topic_score_codex":0.017417569,"about_ca_topic_score_gemma":0.024951357,"teacher_disagreement_score":0.017417569,"about_ca_system_score_codex":0.00081518537,"about_ca_system_score_gemma":0.0011135181,"threshold_uncertainty_score":0.034632385},"labels":[],"label_agreement":null},{"id":"W2988314392","doi":"10.22148/16.049","title":"A Shared Task for the Digital Humanities Chapter 2: Evaluating Annotation Guidelines","year":2019,"lang":"en","type":"article","venue":"Journal of Cultural Analytics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Guideline; Annotation; Task (project management); Section (typography); Computer science; Data science; Artificial intelligence; Political science; Engineering; Systems engineering","score_opus":0.10184203879222058,"score_gpt":0.37150933491015053,"score_spread":0.26966729611792994,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2988314392","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.33678702,0.0044904035,0.45525396,0.015219102,0.0015578519,0.02218469,0.016419388,0.016011091,0.13207641],"genre_scores_gemma":[0.36572716,0.0008203166,0.58838856,0.0016069547,0.00019733925,0.0066732056,0.021418124,0.0017437465,0.013424638],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9087276,0.060268015,0.0068121324,0.005144856,0.017433802,0.0016136637],"domain_scores_gemma":[0.85231113,0.08454446,0.0053472547,0.019036774,0.034563694,0.004196706],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.046922028,0.001505454,0.0017673471,0.006084652,0.0044777486,0.007984232,0.0033743987,0.0042480323,0.011629173],"category_scores_gemma":[0.18854834,0.0007979593,0.0011069212,0.004650766,0.002310216,0.011458816,0.009599663,0.0036291715,0.0046548313],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016787106,0.0029275,0.021988919,0.005334939,0.0004857744,0.0006118375,0.025123045,0.014112217,0.017618652,0.045829356,0.17771167,0.6865774],"study_design_scores_gemma":[0.0012190704,0.002410392,0.032782506,0.006029101,0.0006928158,0.000903485,0.044076595,0.17820436,0.058671243,0.1627481,0.5114981,0.0007642117],"about_ca_topic_score_codex":0.011446832,"about_ca_topic_score_gemma":0.017809628,"teacher_disagreement_score":0.953078,"about_ca_system_score_codex":0.0045891297,"about_ca_system_score_gemma":0.008193662,"threshold_uncertainty_score":0.24815035},"labels":[],"label_agreement":null},{"id":"W2989359117","doi":"","title":"A Context-Based Approach for Linguistic Matching.","year":2007,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Context (archaeology); Linguistics; Computer science; Natural language processing; Matching (statistics); Artificial intelligence; History; Mathematics; Philosophy; Statistics","score_opus":0.01799752504731392,"score_gpt":0.29166541241775085,"score_spread":0.2736678873704369,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2989359117","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00905258,0.0017543371,0.9709622,0.00051090063,0.00034105417,0.00032274413,0.0008312087,0.0031934779,0.013031439],"genre_scores_gemma":[0.14830302,0.000852727,0.8401904,0.00042776487,0.00019052204,0.00029409304,0.0022443866,0.00046618446,0.0070310337],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99614966,0.001411596,0.00030808238,0.0008775187,0.0010258292,0.0002272851],"domain_scores_gemma":[0.9964669,0.001410257,0.00014693182,0.0009508855,0.0008837813,0.00014119396],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021127064,0.00054838695,0.00083264423,0.0039882567,0.0021702088,0.0029337772,0.0025330875,0.0020119825,0.009463123],"category_scores_gemma":[0.012873136,0.0005895461,0.0011316285,0.0036231328,0.0010260894,0.005960984,0.0035474487,0.0019130656,0.0039885845],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004905566,0.0003492843,0.0024766172,0.00054271193,0.00025065397,0.0006803647,0.0011152204,0.009029009,0.018483603,0.2032846,0.02764965,0.73564774],"study_design_scores_gemma":[0.00009831576,0.00018680232,0.0020719287,0.00036394323,0.0003715624,0.001499812,0.0013967011,0.27225226,0.028961912,0.58491987,0.10773481,0.0001420296],"about_ca_topic_score_codex":0.0061895247,"about_ca_topic_score_gemma":0.010263849,"teacher_disagreement_score":0.009463123,"about_ca_system_score_codex":0.0007305751,"about_ca_system_score_gemma":0.0018066913,"threshold_uncertainty_score":0.03165734},"labels":[],"label_agreement":null},{"id":"W2989723222","doi":"10.22148/16.054","title":"Annotating Narrative Levels: Review of Guideline No. 2","year":2019,"lang":"en","type":"article","venue":"Journal of Cultural Analytics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Soundness; Annotation; Guideline; Narrative; Computer science; Information retrieval; Psychology; Linguistics; Political science; Artificial intelligence; Philosophy; Programming language; Law","score_opus":0.02791739552624839,"score_gpt":0.34099030708607025,"score_spread":0.31307291155982186,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2989723222","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004195926,0.69038093,0.08846933,0.13800023,0.028873239,0.005152633,0.004089053,0.0013957153,0.039442923],"genre_scores_gemma":[0.027788147,0.6036975,0.22517538,0.1002998,0.0039125895,0.012873014,0.010527499,0.0012213523,0.014504829],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.95723265,0.015146953,0.013853248,0.002505904,0.010306964,0.00095429434],"domain_scores_gemma":[0.81811035,0.08522803,0.010425122,0.008455715,0.076049164,0.0017316446],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0646411,0.0009968338,0.0027915013,0.014697445,0.0026563737,0.005196046,0.0076987715,0.0054720454,0.0033753032],"category_scores_gemma":[0.18736707,0.0014463176,0.002708316,0.010378481,0.005457302,0.005788701,0.0048080594,0.0057008234,0.0032744047],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001157537,0.000072895375,0.00075618865,0.084054224,0.00022452637,0.0004008477,0.0056683687,0.00046667564,0.0018840976,0.03648852,0.3047623,0.5651057],"study_design_scores_gemma":[0.000029803328,0.000040973286,0.0008909254,0.10409073,0.00027056405,0.00027970748,0.0010867947,0.00016116997,0.001092723,0.0063133794,0.885686,0.00005742616],"about_ca_topic_score_codex":0.025691485,"about_ca_topic_score_gemma":0.05118683,"teacher_disagreement_score":0.0646411,"about_ca_system_score_codex":0.0072525805,"about_ca_system_score_gemma":0.041700132,"threshold_uncertainty_score":0.34185892},"labels":[],"label_agreement":null},{"id":"W2990438615","doi":"10.26615/978-954-452-056-4_072","title":"Resolving Pronouns for a Resource-Poor Language, Malayalam Using Resource-Rich Language, Tamil","year":2019,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"BC Research (Canada)","funders":"","keywords":"Tamil; Malayalam; Computer science; Natural language processing; Artificial intelligence; Leverage (statistics); Resource (disambiguation); Linguistics","score_opus":0.011881809525071424,"score_gpt":0.2806598751922698,"score_spread":0.2687780656671984,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2990438615","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.60333234,0.0012085019,0.34877127,0.002210096,0.00032212722,0.0003380405,0.008307624,0.011235187,0.024274798],"genre_scores_gemma":[0.7951958,0.00037491467,0.18811058,0.00027286628,0.000046796784,0.000067760884,0.00561404,0.00060444116,0.009712726],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9994128,0.00018253717,0.000064866515,0.00013069875,0.0001463296,0.000062921776],"domain_scores_gemma":[0.9982992,0.00077476184,0.00023089211,0.0002864784,0.00035293843,0.000055706645],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009807985,0.0005166799,0.00040132375,0.0016042984,0.0011820102,0.0010782733,0.00064597174,0.00040904636,0.004529126],"category_scores_gemma":[0.004026754,0.00027611552,0.00051545975,0.0012963723,0.0004353609,0.0021080321,0.0011067162,0.00080041186,0.0019164783],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007577272,0.0002826171,0.07924353,0.0026171487,0.00032468766,0.0072204783,0.013281802,0.02022444,0.14033134,0.02751812,0.047240112,0.66095793],"study_design_scores_gemma":[0.00012253727,0.00037325756,0.094307564,0.0004905475,0.0003829091,0.008317715,0.021059684,0.2718059,0.24382013,0.029089788,0.3297405,0.00048952544],"about_ca_topic_score_codex":0.0071829082,"about_ca_topic_score_gemma":0.020979777,"teacher_disagreement_score":0.0071829082,"about_ca_system_score_codex":0.0007212692,"about_ca_system_score_gemma":0.0010619356,"threshold_uncertainty_score":0.0151513815},"labels":[],"label_agreement":null},{"id":"W2991308980","doi":"10.1007/978-3-030-35288-2_10","title":"Finding ALL Answers to OBDA Queries Using Referring Expressions","year":2019,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Focus (optics); Conjunctive query; Database query; Constant (computer programming); Theoretical computer science; Information retrieval; Query language; French horn; Natural language processing; Programming language; Relational database","score_opus":0.03814284817311122,"score_gpt":0.30710724483534674,"score_spread":0.26896439666223554,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2991308980","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13574682,0.005175268,0.7677322,0.004448393,0.0005843687,0.0010656236,0.018162752,0.031303845,0.035780743],"genre_scores_gemma":[0.33442146,0.0020495232,0.6115369,0.00089353474,0.00034435297,0.00042799627,0.026172504,0.0038781392,0.020275531],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9954531,0.0008946133,0.00059123256,0.0009008606,0.001608025,0.0005521322],"domain_scores_gemma":[0.9929109,0.0037425507,0.0003569027,0.0012659141,0.0015375005,0.00018616047],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022950864,0.0021922805,0.0033013585,0.004607466,0.0021247528,0.004735205,0.0022833636,0.00245766,0.010781953],"category_scores_gemma":[0.01486645,0.0008598435,0.0028149101,0.0036717155,0.0012074147,0.008145751,0.0038334902,0.0018929556,0.0055162096],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018256396,0.0005263383,0.012625778,0.0037623616,0.000526384,0.0020341256,0.0053785206,0.014995474,0.05156017,0.11275544,0.11308741,0.6809224],"study_design_scores_gemma":[0.000324421,0.0005395769,0.0056694862,0.00112629,0.0012790147,0.0026553527,0.011992457,0.27575275,0.1085672,0.3517709,0.23986243,0.0004600463],"about_ca_topic_score_codex":0.0071554943,"about_ca_topic_score_gemma":0.008559294,"teacher_disagreement_score":0.010781953,"about_ca_system_score_codex":0.0013992272,"about_ca_system_score_gemma":0.0023326678,"threshold_uncertainty_score":0.036069214},"labels":[],"label_agreement":null},{"id":"W2991550749","doi":"","title":"SemEval-2010 Task 8: Multi-Way Classification of Semantic Relations Between Pairs of Nominals","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"SemEval; Task (project management); Computer science; Testbed; Natural language processing; Artificial intelligence; Information retrieval; World Wide Web","score_opus":0.09263992999547042,"score_gpt":0.23168870888232876,"score_spread":0.13904877888685835,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2991550749","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4036624,0.008332317,0.12171229,0.005348884,0.0074011893,0.008474827,0.2360008,0.1394914,0.0695759],"genre_scores_gemma":[0.27295548,0.000742573,0.22709699,0.002137124,0.00067388505,0.0039913105,0.45687237,0.0045146733,0.031015553],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9889642,0.0039779516,0.00075146597,0.0038155024,0.0016109959,0.00087983056],"domain_scores_gemma":[0.9839866,0.007935614,0.0008251827,0.0037874063,0.0021089057,0.0013562609],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010345993,0.0060047363,0.0037957733,0.0044787433,0.0030588242,0.004620779,0.0061070276,0.007886427,0.02330754],"category_scores_gemma":[0.022307688,0.0011734242,0.0035364795,0.0029850875,0.0020216904,0.007457794,0.008370015,0.005820986,0.023475172],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004390146,0.00353908,0.011004219,0.0036610973,0.0008327474,0.0011363095,0.001253988,0.0052563515,0.016965272,0.0041102786,0.6164522,0.33139828],"study_design_scores_gemma":[0.005263818,0.0055141696,0.0632395,0.0012600477,0.0008439278,0.007977816,0.006918337,0.22551697,0.10524241,0.03963151,0.53771347,0.0008779311],"about_ca_topic_score_codex":0.007068848,"about_ca_topic_score_gemma":0.011287632,"teacher_disagreement_score":0.02330754,"about_ca_system_score_codex":0.0024293256,"about_ca_system_score_gemma":0.0038332993,"threshold_uncertainty_score":0.07797146},"labels":[],"label_agreement":null},{"id":"W2991553420","doi":"10.14705/rpnet.2019.38.1006","title":"MOOCs as environments for learning spoken academic vocabulary","year":2019,"lang":"en","type":"book-chapter","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Dialogic; Vocabulary; Context (archaeology); Computer science; Contrast (vision); Subject (documents); Academic year; Academic library; Linguistics; Mathematics education; Artificial intelligence; Psychology; World Wide Web; History; Pedagogy; Library science","score_opus":0.014892902491929382,"score_gpt":0.2623797754881905,"score_spread":0.24748687299626115,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2991553420","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20218647,0.0120483795,0.42100957,0.002810051,0.0015554669,0.00105169,0.007241379,0.01544484,0.33665216],"genre_scores_gemma":[0.50009304,0.0048927763,0.31850648,0.0008971942,0.0006095639,0.001352803,0.018611122,0.0032348344,0.15180214],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99921703,0.00032096962,0.000036227008,0.00013408848,0.00023935153,0.000052373714],"domain_scores_gemma":[0.9983388,0.0010406395,0.0000781323,0.00022249078,0.00012345403,0.0001964555],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00071872567,0.00043028608,0.00024713084,0.0011540035,0.0008137647,0.0032326428,0.0010338268,0.00054073165,0.0171009],"category_scores_gemma":[0.00408972,0.00024354777,0.00024121645,0.0013119089,0.00073271105,0.0034749235,0.0030667128,0.00089258584,0.005397569],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014259927,0.00023339833,0.0035493106,0.0010539202,0.000029537152,0.0003638147,0.009147741,0.0033667441,0.015545168,0.1269993,0.08229861,0.7572698],"study_design_scores_gemma":[0.000041454678,0.0001537607,0.008636294,0.00040361925,0.00001707597,0.00055711117,0.0046420093,0.007833832,0.007912014,0.05134535,0.9183977,0.00005965239],"about_ca_topic_score_codex":0.0011889548,"about_ca_topic_score_gemma":0.003943983,"teacher_disagreement_score":0.0171009,"about_ca_system_score_codex":0.00058035104,"about_ca_system_score_gemma":0.0008106171,"threshold_uncertainty_score":0.05720824},"labels":[],"label_agreement":null},{"id":"W2991733328","doi":"","title":"Using two-dimensional box plots to visualize the vowel space: A study of rounded vowel allophones in Tigrinya","year":2011,"lang":"en","type":"article","venue":"Canadian acoustics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Vowel; Formant; Rounding; Range (aeronautics); Mid vowel; Mathematics; Word (group theory); Computer science; Speech recognition; Engineering; Geometry","score_opus":0.04993480815219229,"score_gpt":0.30682905720204395,"score_spread":0.25689424904985164,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2991733328","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99642307,0.000098779026,0.002179172,0.000027943543,0.000009074277,0.000050160466,0.00004877611,0.000013817434,0.0011493809],"genre_scores_gemma":[0.9926805,0.00010051482,0.0061087543,0.000023437435,0.000005611728,0.00006351772,0.00004044307,0.000011954356,0.0009653045],"study_design_codex":"qualitative","study_design_gemma":"observational","domain_scores_codex":[0.9992848,0.00037180723,0.000044303673,0.00011849993,0.00009866155,0.00008191515],"domain_scores_gemma":[0.9945528,0.004109847,0.00031312136,0.00025517863,0.0006062056,0.00016292681],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019322352,0.0003592171,0.00027593342,0.0010533279,0.00090265216,0.00092531566,0.0004890563,0.00050396036,0.0019691011],"category_scores_gemma":[0.005969755,0.00025946877,0.00024075648,0.0008951814,0.0014068579,0.0006798001,0.00064896204,0.00057302567,0.0003657624],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014792825,0.00079081015,0.15305915,0.0007843425,0.000049974497,0.0039731627,0.60783046,0.00072886556,0.13963404,0.0019668075,0.00058341556,0.08911964],"study_design_scores_gemma":[0.000054202803,0.0036460073,0.6728969,0.00014957869,0.000056832116,0.0036514392,0.28324315,0.0018693333,0.024142187,0.0009219275,0.0091951685,0.00017322216],"about_ca_topic_score_codex":0.0046492675,"about_ca_topic_score_gemma":0.012328416,"teacher_disagreement_score":0.0046492675,"about_ca_system_score_codex":0.0004715421,"about_ca_system_score_gemma":0.00041236807,"threshold_uncertainty_score":0.0102187395},"labels":[],"label_agreement":null},{"id":"W2992420505","doi":"10.22148/16.056","title":"Annotation Guideline No. 5: Annotation Guidelines for Narrative Levels and Narrative Acts","year":2019,"lang":"en","type":"article","venue":"Journal of Cultural Analytics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Narrative; Annotation; Plot (graphics); Conjunction (astronomy); Guideline; Task (project management); Computer science; Narrative criticism; Natural language processing; Information retrieval; Narrative inquiry; Linguistics; Artificial intelligence; Political science; Engineering; Mathematics; Philosophy","score_opus":0.06077488116399827,"score_gpt":0.38185205362106406,"score_spread":0.3210771724570658,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2992420505","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013482621,0.0014818166,0.6344794,0.017289914,0.00784718,0.0144089395,0.064677365,0.022628894,0.22370385],"genre_scores_gemma":[0.047899768,0.001508582,0.7207044,0.008364479,0.0011449837,0.027446412,0.07611427,0.01223065,0.10458647],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9787394,0.0073290234,0.0061993375,0.0021340232,0.0044572377,0.0011410969],"domain_scores_gemma":[0.8771055,0.03772457,0.0042168624,0.014811948,0.06449036,0.0016507314],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.022780461,0.0013897325,0.0016027648,0.008335207,0.0052260202,0.006457365,0.00360652,0.006072878,0.038008563],"category_scores_gemma":[0.09387158,0.0017954782,0.0012698064,0.0047068666,0.0034718784,0.006433053,0.0051075835,0.0052242675,0.03906385],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029963272,0.00018000069,0.002006218,0.0039424254,0.000033825774,0.000977776,0.019400928,0.0007459729,0.018351104,0.078090824,0.75929666,0.11667461],"study_design_scores_gemma":[0.00003368699,0.00002313249,0.0019974832,0.001866095,0.000033595108,0.00032341346,0.002064404,0.001281099,0.006628926,0.016959276,0.96868825,0.00010057922],"about_ca_topic_score_codex":0.018220438,"about_ca_topic_score_gemma":0.027868442,"teacher_disagreement_score":0.038008563,"about_ca_system_score_codex":0.0033284998,"about_ca_system_score_gemma":0.009403546,"threshold_uncertainty_score":0.12715131},"labels":[],"label_agreement":null},{"id":"W2992775586","doi":"10.22148/16.055","title":"Annotation Guideline No. 4: Annotating Narrative Levels in Literature","year":2019,"lang":"en","type":"article","venue":"Journal of Cultural Analytics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Narratology; Narrative; Interpretation (philosophy); Task (project management); Field (mathematics); Annotation; Intersubjectivity; Focus (optics); Computer science; Narrative structure; Epistemology; Linguistics; Sociology; Literature; Art; Artificial intelligence; Philosophy","score_opus":0.016682114255803713,"score_gpt":0.3160439169189949,"score_spread":0.2993618026631912,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2992775586","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02786278,0.0019284878,0.7148044,0.02016324,0.006494274,0.02415097,0.054993812,0.018428884,0.13117307],"genre_scores_gemma":[0.06458747,0.0017606773,0.75326806,0.0077508087,0.0008854939,0.035640027,0.05885159,0.008044055,0.06921181],"study_design_codex":"not_applicable","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.97137105,0.010225293,0.008023725,0.0033518479,0.0054596383,0.0015683516],"domain_scores_gemma":[0.89312315,0.030259637,0.0044979146,0.021158183,0.048302375,0.0026588095],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.031401686,0.0012653375,0.0017578405,0.010075185,0.0070810304,0.006767443,0.004460602,0.0062532867,0.022075942],"category_scores_gemma":[0.07943671,0.0015204751,0.0015545061,0.0056141317,0.004682498,0.006179095,0.009004139,0.0046803094,0.021473773],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006483687,0.00033155028,0.0060006524,0.012087378,0.00012601109,0.0032933585,0.069622785,0.001254049,0.093529694,0.06678721,0.5437967,0.20252228],"study_design_scores_gemma":[0.000044922206,0.000050068913,0.0034100835,0.0025761449,0.00007318157,0.000683037,0.005196147,0.0015476311,0.015000676,0.017153936,0.95414746,0.00011670138],"about_ca_topic_score_codex":0.012006604,"about_ca_topic_score_gemma":0.017657304,"teacher_disagreement_score":0.031401686,"about_ca_system_score_codex":0.0034161182,"about_ca_system_score_gemma":0.013737998,"threshold_uncertainty_score":0.16606992},"labels":[],"label_agreement":null},{"id":"W2994260601","doi":"","title":"The Formalization of English Structures with Preposition \"in\" and Their Chinese Translations","year":2011,"lang":"en","type":"article","venue":"Studies in literature and language","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Linguistics; Element (criminal law); Perspective (graphical); Natural language processing; Machine translation; Artificial intelligence; Key (lock); Quality (philosophy); Translation (biology); Philosophy; Epistemology","score_opus":0.008101004144323638,"score_gpt":0.26152559857828606,"score_spread":0.2534245944339624,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2994260601","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.059546407,0.0040114718,0.8411932,0.00489276,0.0007670231,0.00042948616,0.001108805,0.0007548619,0.087296054],"genre_scores_gemma":[0.5566459,0.0032609948,0.41751057,0.0010301725,0.00053566543,0.00067440816,0.0015532386,0.00022877344,0.018560382],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99857056,0.00047085868,0.00022929421,0.00033132298,0.00026720148,0.00013071432],"domain_scores_gemma":[0.9987445,0.00044764625,0.0002165415,0.00020253881,0.00035185178,0.000036905392],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019409677,0.00092268066,0.0004580814,0.0022154439,0.0014538968,0.0025714259,0.00092128303,0.0007950402,0.0058657555],"category_scores_gemma":[0.0030685267,0.00041043657,0.0012684587,0.0022803412,0.005427494,0.006076387,0.0013100156,0.0017017988,0.0009920667],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000015064001,0.000012845293,0.00014606414,0.00011488598,0.000006478365,0.00015679843,0.0010100662,0.0008008251,0.0011762709,0.9847595,0.0011068485,0.010694233],"study_design_scores_gemma":[0.00003844858,0.000089706664,0.00090615207,0.00028009483,0.00007511122,0.00089930097,0.001265068,0.019653745,0.008334819,0.84589195,0.12249095,0.00007463872],"about_ca_topic_score_codex":0.0050293254,"about_ca_topic_score_gemma":0.0034968427,"teacher_disagreement_score":0.0058657555,"about_ca_system_score_codex":0.0016528803,"about_ca_system_score_gemma":0.002766142,"threshold_uncertainty_score":0.019622922},"labels":[],"label_agreement":null},{"id":"W2994792908","doi":"10.1109/visual.2019.8933537","title":"H-Matrix: Hierarchical Matrix for Visual Analysis of Cross-Linguistic Features in Large Learner Corpora","year":2019,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ontario Tech University","funders":"","keywords":"Computer science; Matrix (chemical analysis); Context (archaeology); Visualization; Natural language processing; Artificial intelligence; Tree (set theory); Row; Reading (process); Aggregate (composite); Linguistics; Programming language; Mathematics","score_opus":0.00960229218598067,"score_gpt":0.36140672589996914,"score_spread":0.35180443371398845,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2994792908","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0047064475,0.00033808738,0.9292747,0.0004482199,0.00022817278,0.00034474913,0.007369093,0.054687794,0.0026027462],"genre_scores_gemma":[0.034342583,0.0002707852,0.9502774,0.0001302497,0.000091902046,0.0007076726,0.005439219,0.0065007135,0.002239554],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9981042,0.00057094096,0.00020926255,0.00037642592,0.0006247658,0.00011429707],"domain_scores_gemma":[0.99120367,0.0048350315,0.000688127,0.00117388,0.0017839,0.00031542918],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004260407,0.001992884,0.00076074706,0.005805124,0.0012577856,0.0037483305,0.0018214345,0.0011454849,0.038041428],"category_scores_gemma":[0.01677921,0.00087706157,0.0011707938,0.004129018,0.00076430297,0.0044848747,0.0035836708,0.0022851839,0.007852207],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010594389,0.00023764878,0.004015471,0.0023419063,0.0003403405,0.0009064736,0.0067041963,0.00568101,0.031608988,0.033646673,0.19963922,0.71381867],"study_design_scores_gemma":[0.00043602483,0.00038028546,0.0120363105,0.00076877937,0.0002626789,0.0016568934,0.004403256,0.28521156,0.06236637,0.16136526,0.47059017,0.00052241085],"about_ca_topic_score_codex":0.0036065183,"about_ca_topic_score_gemma":0.005123192,"teacher_disagreement_score":0.038041428,"about_ca_system_score_codex":0.00060924515,"about_ca_system_score_gemma":0.0014553907,"threshold_uncertainty_score":0.12726128},"labels":[],"label_agreement":null},{"id":"W2997019142","doi":"10.4000/traduire.1848","title":"La plus-value de la biotraduction face à la machine","year":2019,"lang":"fr","type":"article","venue":"Traduire","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Historical Studies in Education","funders":"","keywords":"Physics","score_opus":0.012058207084081713,"score_gpt":0.28912308301624784,"score_spread":0.2770648759321661,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2997019142","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.133695,0.0023548056,0.78540105,0.010052388,0.0013880811,0.00008318294,0.0014324468,0.0030450295,0.06254803],"genre_scores_gemma":[0.77128136,0.00087907567,0.1726827,0.0011453734,0.00097360316,0.0001773063,0.0010341136,0.00071069074,0.051115893],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9950282,0.00080773595,0.00031991105,0.0012166464,0.0022153156,0.00041221138],"domain_scores_gemma":[0.9914151,0.0045363754,0.0004669722,0.0015164352,0.0017385401,0.00032663925],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025320565,0.0010427157,0.0009892795,0.0020112847,0.001173104,0.004053254,0.001678538,0.0019715475,0.015125628],"category_scores_gemma":[0.0146552995,0.00054096157,0.0012641154,0.0012314739,0.0031248233,0.0068387273,0.0027958187,0.0038072818,0.0036829712],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00088240294,0.000109832996,0.0045075226,0.0004921641,0.00016596296,0.00039417818,0.0003282008,0.031331573,0.027826918,0.52675045,0.012393231,0.39481753],"study_design_scores_gemma":[0.00009305278,0.00039584463,0.005822169,0.00016308905,0.000164235,0.0017230919,0.00026397218,0.19622576,0.082803935,0.6632829,0.048922066,0.0001400087],"about_ca_topic_score_codex":0.0019150914,"about_ca_topic_score_gemma":0.0021707944,"teacher_disagreement_score":0.015125628,"about_ca_system_score_codex":0.0021841896,"about_ca_system_score_gemma":0.0011453977,"threshold_uncertainty_score":0.05060029},"labels":[],"label_agreement":null},{"id":"W2997248598","doi":"10.1609/aaai.v34i05.6296","title":"One Homonym per Translation","year":2020,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Innovates","keywords":"Homonym (biology); Polysemy; Context (archaeology); Semantics (computer science); Computer science; Linguistics; Natural language processing; Lexical semantics; Syntax; Resource (disambiguation); Translation (biology); Artificial intelligence; Lexical item; History; Programming language; Philosophy; Biology","score_opus":0.03522183704819519,"score_gpt":0.26679430593317344,"score_spread":0.23157246888497823,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2997248598","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4997355,0.0028179206,0.33266363,0.0033071802,0.0011177525,0.00036999048,0.013578725,0.0036475663,0.14276166],"genre_scores_gemma":[0.86585253,0.00052039826,0.11679547,0.0003183378,0.00015171175,0.00012223444,0.0046328884,0.0005011463,0.01110533],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99798644,0.00056206156,0.0002297189,0.00060731714,0.0005189051,0.000095554555],"domain_scores_gemma":[0.9958111,0.0017715474,0.00023631986,0.0015208905,0.0005099026,0.00015025186],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012313443,0.00032657958,0.0005908239,0.0020229348,0.0016316655,0.0019037911,0.00048583985,0.00076304324,0.016827106],"category_scores_gemma":[0.005437119,0.00031278186,0.00037571305,0.0023167138,0.0017279825,0.006302408,0.0023691163,0.0008873478,0.0029525387],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00083985634,0.00018461287,0.01795412,0.0015534395,0.00012470357,0.0020057042,0.011681912,0.0023719857,0.0971712,0.41360784,0.020088932,0.43241572],"study_design_scores_gemma":[0.00007354223,0.00023431281,0.023865482,0.00045037316,0.00016896953,0.0052768597,0.00823809,0.019792672,0.043161906,0.5225565,0.3759434,0.00023781961],"about_ca_topic_score_codex":0.000534267,"about_ca_topic_score_gemma":0.00089555676,"teacher_disagreement_score":0.016827106,"about_ca_system_score_codex":0.00049942243,"about_ca_system_score_gemma":0.00088706636,"threshold_uncertainty_score":0.056292295},"labels":[],"label_agreement":null},{"id":"W2998081612","doi":"10.1609/aaai.v34i05.6513","title":"Reinforced Curriculum Learning on Pre-Trained Neural Machine Translation Models","year":2020,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Machine learning; Heuristics; Artificial intelligence; Artificial neural network; Reinforcement learning; Machine translation; Curriculum; Task (project management); Set (abstract data type); Sample (material); Baseline (sea)","score_opus":0.07127140276444606,"score_gpt":0.2974927484928364,"score_spread":0.22622134572839037,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2998081612","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09941277,0.00089248514,0.89244145,0.000657719,0.00011627058,0.00015499469,0.00015204931,0.0018368278,0.0043353955],"genre_scores_gemma":[0.8632382,0.0003597666,0.1294917,0.00036560636,0.00010216704,0.0004025834,0.00049007684,0.0001787392,0.005371109],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99929786,0.00028310015,0.00004073178,0.00020588886,0.00009049668,0.000081879516],"domain_scores_gemma":[0.9958611,0.0027893037,0.0003148709,0.0002681189,0.00060763495,0.00015910596],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021659082,0.0013606993,0.0014880693,0.0006674385,0.0005176994,0.0008916087,0.0018060928,0.0014531458,0.0020011798],"category_scores_gemma":[0.010802254,0.00074710115,0.00062636164,0.00064020563,0.0011459499,0.0016818173,0.0013967601,0.002402052,0.00064856734],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000084968946,0.000081822305,0.00085195975,0.000077014556,0.000035173452,0.000065109314,0.00007601879,0.9274735,0.0013017965,0.004420809,0.0010516472,0.06448019],"study_design_scores_gemma":[0.000009768115,0.00002382514,0.000046369434,0.000006174077,0.0000043004325,0.000005467509,0.000002911856,0.9973775,0.0003333942,0.0020329517,0.00015451203,0.0000028046422],"about_ca_topic_score_codex":0.0053061265,"about_ca_topic_score_gemma":0.0066595655,"teacher_disagreement_score":0.0053061265,"about_ca_system_score_codex":0.001608183,"about_ca_system_score_gemma":0.0013785312,"threshold_uncertainty_score":0.011668205},"labels":[],"label_agreement":null},{"id":"W2998767185","doi":"10.3765/salt.v29i0.4637","title":"Singular which, mention-some, and variable scope uniqueness","year":2020,"lang":"en","type":"article","venue":"Proceedings from Semantics and Linguistic Theory","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Uniqueness; Presupposition; Scope (computer science); Operator (biology); Variable (mathematics); Mathematics; Epistemology; Computer science; Philosophy; Mathematical analysis","score_opus":0.008662183449825401,"score_gpt":0.22522212321429919,"score_spread":0.2165599397644738,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2998767185","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5188526,0.00212008,0.21697417,0.0062426846,0.00056283106,0.00013770588,0.00092977146,0.00095745665,0.25322264],"genre_scores_gemma":[0.9869985,0.000107088825,0.0073494148,0.0004184624,0.000120087156,0.00003542181,0.00021801413,0.00014466059,0.0046083727],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99726605,0.0008328055,0.00026235747,0.00090646656,0.00042578767,0.00030651494],"domain_scores_gemma":[0.9912685,0.004518923,0.0005777599,0.0023224866,0.0009089179,0.00040341818],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003288086,0.0004966465,0.00063803926,0.0010120908,0.0024555998,0.0030584824,0.0012051458,0.0024307116,0.007668291],"category_scores_gemma":[0.013532455,0.00073403365,0.0011571925,0.000679702,0.007049462,0.011296614,0.0053886552,0.003011547,0.00060092943],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020809734,0.000029951429,0.0060854754,0.0001601335,0.00003491769,0.0007922905,0.007348217,0.00040989384,0.00538673,0.9639018,0.0014438848,0.014198594],"study_design_scores_gemma":[0.000051496198,0.000107576096,0.016343895,0.00014438444,0.0001320599,0.003817363,0.006379134,0.0048453263,0.01117797,0.9179526,0.038923435,0.00012474487],"about_ca_topic_score_codex":0.0017111127,"about_ca_topic_score_gemma":0.0016345015,"teacher_disagreement_score":0.007668291,"about_ca_system_score_codex":0.0011340664,"about_ca_system_score_gemma":0.000641783,"threshold_uncertainty_score":0.025652945},"labels":[],"label_agreement":null},{"id":"W2998938469","doi":"10.22148/001c.11772","title":"Annotation Guideline No. 7: Guidelines for annotation of narrative structure","year":2020,"lang":"en","type":"article","venue":"Journal of Cultural Analytics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Narrative; Annotation; Plot (graphics); Character (mathematics); Perspective (graphical); Narrative structure; Guideline; Sequence (biology); Linguistics; Literature; History; Computer science; Artificial intelligence; Philosophy; Art; Mathematics; Political science","score_opus":0.06051591390825047,"score_gpt":0.37230084988058393,"score_spread":0.3117849359723335,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2998938469","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004548248,0.0018934872,0.6724639,0.016570996,0.0053044436,0.021260675,0.058478475,0.030139124,0.18934058],"genre_scores_gemma":[0.013414931,0.0019352416,0.8008269,0.0066259876,0.0007838338,0.031702437,0.07424064,0.011822548,0.058647435],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.94053614,0.02497347,0.01695015,0.003983426,0.011236947,0.0023198584],"domain_scores_gemma":[0.7489852,0.09128536,0.007507145,0.032662634,0.116587915,0.0029716562],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.05574661,0.0018154089,0.002130493,0.013931277,0.006939171,0.011138231,0.0070152655,0.0101792,0.049732823],"category_scores_gemma":[0.2001026,0.003086317,0.0022197657,0.009409411,0.0046565356,0.009031833,0.008599568,0.0070831897,0.07693699],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024140105,0.00018284819,0.0009773909,0.0055485247,0.000057543202,0.001065021,0.02203367,0.0007124419,0.01709933,0.05477626,0.7769753,0.120330386],"study_design_scores_gemma":[0.000036139572,0.000022279619,0.00077473576,0.0031405224,0.000033878245,0.00033724372,0.0018278526,0.00093519484,0.0056242654,0.014084626,0.9730867,0.00009656198],"about_ca_topic_score_codex":0.021962466,"about_ca_topic_score_gemma":0.029444046,"teacher_disagreement_score":0.05574661,"about_ca_system_score_codex":0.004935144,"about_ca_system_score_gemma":0.017642388,"threshold_uncertainty_score":0.29481983},"labels":[],"label_agreement":null},{"id":"W3001536763","doi":"10.22148/001c.11773","title":"Annotation Guideline No. 8: Annotation Guidelines for Narrative Levels","year":2020,"lang":"en","type":"article","venue":"Journal of Cultural Analytics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Narrative; Annotation; Phenomenon; Set (abstract data type); Class (philosophy); Value (mathematics); Perspective (graphical); Element (criminal law); Linguistics; Mathematics education; Computer science; Psychology; Epistemology; Artificial intelligence; Philosophy","score_opus":0.1214185467744917,"score_gpt":0.3977763826092004,"score_spread":0.2763578358347087,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3001536763","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008133921,0.0014562825,0.688272,0.019579755,0.00711319,0.018385127,0.049630564,0.03324727,0.17418194],"genre_scores_gemma":[0.03127528,0.0012010655,0.7868016,0.008944815,0.00086586433,0.02984237,0.04177484,0.018003527,0.08129069],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.96879643,0.0119164325,0.009039721,0.003502262,0.005351058,0.001394002],"domain_scores_gemma":[0.8574897,0.04736572,0.0042462293,0.01799665,0.07074359,0.0021580972],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.035543565,0.001579479,0.0016111357,0.008357537,0.0058499468,0.007501065,0.004133283,0.0064766514,0.051655628],"category_scores_gemma":[0.12119696,0.0022850013,0.0017009063,0.0045748954,0.0035843756,0.007636209,0.006141188,0.006185036,0.05100483],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037816993,0.00015933556,0.0016386663,0.0042731618,0.00005223154,0.0012712701,0.02911854,0.0007353545,0.019558825,0.0640476,0.7524916,0.1262753],"study_design_scores_gemma":[0.00003399469,0.000016955066,0.0009062489,0.0017835257,0.000027275466,0.00023566541,0.0020268592,0.00082273275,0.0051069907,0.012970781,0.9759956,0.00007328826],"about_ca_topic_score_codex":0.017418059,"about_ca_topic_score_gemma":0.026560659,"teacher_disagreement_score":0.051655628,"about_ca_system_score_codex":0.0041628825,"about_ca_system_score_gemma":0.011116356,"threshold_uncertainty_score":0.18797457},"labels":[],"label_agreement":null},{"id":"W3004368580","doi":"","title":"Improving the neural network-based machine transliteration for low-resourced language pair.","year":2018,"lang":"en","type":"article","venue":"Institutional Repositories DataBase (IRDB)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Transliteration; Computer science; Artificial intelligence; Artificial neural network; Machine translation; Natural language processing","score_opus":0.010421851881281625,"score_gpt":0.25877528616847784,"score_spread":0.24835343428719622,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3004368580","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.18748824,0.0037219848,0.7388645,0.0010421898,0.0014372813,0.00050785206,0.0052871224,0.042749,0.018901758],"genre_scores_gemma":[0.475739,0.0009533576,0.47792768,0.0006189138,0.0002960068,0.00040664058,0.017201168,0.0024524676,0.024404733],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99876475,0.00034879445,0.00011394942,0.0004108896,0.0002146985,0.00014697503],"domain_scores_gemma":[0.9973814,0.0011452712,0.00013889563,0.00043256968,0.00081717223,0.00008468669],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009927138,0.001211929,0.0009775232,0.0012891754,0.0008239277,0.001239447,0.0016453399,0.0010444671,0.013973333],"category_scores_gemma":[0.0058063404,0.0003526752,0.0006968161,0.0012730059,0.00030731148,0.0031254352,0.0018582782,0.0017511935,0.01242013],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007234291,0.00038330155,0.001822793,0.0006861972,0.00016778246,0.0008589671,0.00037212204,0.016036099,0.06584761,0.0036473963,0.03977768,0.8696766],"study_design_scores_gemma":[0.00016371414,0.0005408051,0.0032027992,0.00011971278,0.00030370156,0.0012382441,0.001045204,0.7818157,0.15717895,0.010657584,0.04362366,0.00010993574],"about_ca_topic_score_codex":0.005092599,"about_ca_topic_score_gemma":0.009160936,"teacher_disagreement_score":0.013973333,"about_ca_system_score_codex":0.0005462161,"about_ca_system_score_gemma":0.0018787,"threshold_uncertainty_score":0.04674542},"labels":[],"label_agreement":null},{"id":"W3004469778","doi":"","title":"Utilisation d'un score de qualité de traduction pour le résumé multi-document cross-lingue","year":2011,"lang":"fr","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Mathematics","score_opus":0.044491740347850625,"score_gpt":0.29626941797558826,"score_spread":0.2517776776277376,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3004469778","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.743076,0.005470661,0.16452716,0.0022139035,0.0013987998,0.0004245602,0.026015883,0.015733093,0.04113995],"genre_scores_gemma":[0.91404206,0.00059333484,0.05216284,0.00017279838,0.00024093282,0.00023447296,0.019639147,0.0011273319,0.01178708],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9815429,0.0056267027,0.0027292457,0.0034432758,0.005808154,0.00084972946],"domain_scores_gemma":[0.9059869,0.062582865,0.0050438168,0.0062670857,0.01852502,0.0015942691],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012317221,0.0017664363,0.001370024,0.013738947,0.0012427464,0.007883704,0.0012567814,0.003455976,0.0097973095],"category_scores_gemma":[0.058103547,0.0004940467,0.0019033833,0.0069050416,0.0015241215,0.0071259444,0.0026029984,0.0018224613,0.0071688714],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0047521633,0.00064107473,0.16159166,0.0032189968,0.002867783,0.0007013634,0.0043832497,0.021929158,0.039447412,0.0034656406,0.02332878,0.73367274],"study_design_scores_gemma":[0.00048747848,0.0041036285,0.43073508,0.0011150133,0.0027946786,0.002901362,0.010484989,0.3341702,0.12627588,0.012565299,0.073320605,0.0010457461],"about_ca_topic_score_codex":0.008882142,"about_ca_topic_score_gemma":0.006697497,"teacher_disagreement_score":0.013738947,"about_ca_system_score_codex":0.001235475,"about_ca_system_score_gemma":0.0012877617,"threshold_uncertainty_score":0.065140486},"labels":[],"label_agreement":null},{"id":"W3004672535","doi":"10.5539/ijel.v10n2p184","title":"Teaching Arabic Machine Translation to EFL Student Translators: A Case Study of Omani Translation Undergraduates","year":2020,"lang":"en","type":"article","venue":"International Journal of English Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Arabic; Machine translation; Computer science; Mathematics education; Translation (biology); Focus (optics); Natural language processing; Translation studies; English language; Artificial intelligence; Linguistics; Psychology","score_opus":0.03164431339074281,"score_gpt":0.33595123009756594,"score_spread":0.30430691670682314,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3004672535","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99146336,0.0002281071,0.0015613971,0.0018714523,0.00007099554,0.00013422877,0.000017794408,0.000066495166,0.004586189],"genre_scores_gemma":[0.9810829,0.0004997415,0.004353945,0.0017697145,0.000078763216,0.0001537715,0.0000583337,0.00009929901,0.011903598],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.994381,0.003684283,0.00018965075,0.00038996257,0.00055619056,0.00079887104],"domain_scores_gemma":[0.9895167,0.0045381025,0.0007580446,0.0005477274,0.0014687715,0.0031707648],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005285787,0.0008278145,0.0007209665,0.0010349615,0.011344903,0.0043331184,0.0021658218,0.0038967398,0.005704988],"category_scores_gemma":[0.015147173,0.0006151742,0.00050690817,0.0015301468,0.0029557562,0.00213749,0.0039179563,0.0031544454,0.0029777256],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00032815948,0.0052304035,0.013888365,0.0004148752,0.000024389592,0.021857016,0.85506076,0.0004731747,0.004998369,0.0019816102,0.0045326645,0.09121023],"study_design_scores_gemma":[0.00018518565,0.003686866,0.016946472,0.0002795284,0.00005181438,0.010481555,0.87026364,0.0025698347,0.0074900836,0.0014569481,0.086489156,0.00009887245],"about_ca_topic_score_codex":0.0027588082,"about_ca_topic_score_gemma":0.0101687405,"teacher_disagreement_score":0.011344903,"about_ca_system_score_codex":0.0026071155,"about_ca_system_score_gemma":0.0027311423,"threshold_uncertainty_score":0.02795422},"labels":[],"label_agreement":null},{"id":"W3005088344","doi":"10.25365/cts-2019-1-2-5","title":"Chaos out of Order","year":2019,"lang":"en","type":"article","venue":"DOAJ (DOAJ: Directory of Open Access Journals)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Romanian; Perspective (graphical); Poetry; Context (archaeology); Order (exchange); Sociology; Interpersonal communication; Epistemology; History; Linguistics; Social science; Literature; Computer science; Artificial intelligence; Art; Philosophy; Economics","score_opus":0.16309743268926807,"score_gpt":0.5365800938479841,"score_spread":0.37348266115871603,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3005088344","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17755851,0.0065159546,0.124843985,0.034307323,0.0013556489,0.00019479435,0.0009401264,0.00073615584,0.6535475],"genre_scores_gemma":[0.96196973,0.0011922837,0.009437774,0.00095161994,0.00031512932,0.00010084142,0.0002810208,0.00019259262,0.025559075],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9979304,0.0007136202,0.0000937213,0.0005283649,0.0004986973,0.00023514348],"domain_scores_gemma":[0.9962351,0.0015359788,0.00041002757,0.0011049184,0.00044247176,0.00027141362],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002082383,0.00033354707,0.0005404696,0.0017617331,0.004871387,0.00880043,0.00066674873,0.000952635,0.007494979],"category_scores_gemma":[0.007973762,0.00032248016,0.0005484544,0.001544233,0.023331149,0.011860768,0.0043931417,0.0019758078,0.0013050964],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000018498376,0.000004442295,0.00072666875,0.000038385497,0.000007950242,0.00009292635,0.007788386,0.00063411525,0.00017764836,0.9819119,0.0029392543,0.0056598554],"study_design_scores_gemma":[0.000008992475,0.000010869062,0.0008352966,0.000048168127,0.0000063306784,0.00010475982,0.003425036,0.002023281,0.00018736236,0.9000246,0.09330971,0.000015683514],"about_ca_topic_score_codex":0.006667342,"about_ca_topic_score_gemma":0.0068848203,"teacher_disagreement_score":0.00880043,"about_ca_system_score_codex":0.003703373,"about_ca_system_score_gemma":0.0028115618,"threshold_uncertainty_score":0.026869953},"labels":[],"label_agreement":null},{"id":"W3005998127","doi":"10.22148/16.034","title":"A BLAST-based, Language-agnostic Text Reuse Algorithm with a MARKUS Implementation and Sequence Alignment Optimized for Large Chinese Corpora","year":2019,"lang":"en","type":"article","venue":"Journal of Cultural Analytics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Reuse; Natural language processing; Phrase; Sequence (biology); Artificial intelligence; Engineering","score_opus":0.011587360958005563,"score_gpt":0.3090124364441694,"score_spread":0.29742507548616387,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3005998127","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07497234,0.0012284219,0.7125255,0.0005807053,0.00069078733,0.0014980449,0.01887269,0.17939287,0.010238719],"genre_scores_gemma":[0.060246382,0.0003611268,0.89324963,0.0002877522,0.000069429276,0.0012600374,0.02999623,0.008391843,0.006137557],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.998014,0.00034958922,0.00026696213,0.00083234906,0.00037583732,0.00016131022],"domain_scores_gemma":[0.9983144,0.0005388206,0.00016939065,0.00040224887,0.0004556921,0.00011954309],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002832057,0.0028661569,0.002179322,0.0029747356,0.0031082593,0.0022764737,0.002799159,0.0018079928,0.012937353],"category_scores_gemma":[0.007849076,0.0016979584,0.0019108097,0.0059376885,0.0009527617,0.0024325114,0.002236833,0.003262169,0.018239282],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0041649416,0.0008601502,0.0111502,0.004060322,0.0011761625,0.0019538086,0.0028779337,0.010109292,0.30039865,0.013505395,0.14732836,0.5024148],"study_design_scores_gemma":[0.0014257352,0.0017829287,0.021882512,0.00046865852,0.0013823707,0.00553958,0.0020957699,0.36279774,0.30897424,0.029602483,0.26337075,0.0006772167],"about_ca_topic_score_codex":0.004480249,"about_ca_topic_score_gemma":0.013475975,"teacher_disagreement_score":0.012937353,"about_ca_system_score_codex":0.0011652336,"about_ca_system_score_gemma":0.0035929864,"threshold_uncertainty_score":0.043279827},"labels":[],"label_agreement":null},{"id":"W3006288522","doi":"10.1016/j.jcin.2019.11.030","title":"When SVGs “Had Enough”","year":2020,"lang":"en","type":"letter","venue":"JACC: Cardiovascular Interventions","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; McGill University Health Centre","funders":"","keywords":"Computer science","score_opus":0.03340529118534909,"score_gpt":0.27476195107149587,"score_spread":0.2413566598861468,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3006288522","genre_codex":"commentary","genre_gemma":"commentary","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"commentary","genre_consensus":"commentary","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.002439149,0.001310361,0.002120337,0.9289611,0.037674144,0.00003002925,0.00023511935,0.00031895275,0.02691078],"genre_scores_gemma":[0.030322023,0.00077893963,0.0025758138,0.9065556,0.025485916,0.000046434405,0.00020611913,0.00021892798,0.03381032],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99706167,0.0008716424,0.00033688123,0.00034095094,0.0007750662,0.000613765],"domain_scores_gemma":[0.99179345,0.0044306866,0.0004116031,0.0004040252,0.0014959809,0.0014642889],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003272914,0.0004164139,0.0006142158,0.0006721475,0.0021967252,0.0032944917,0.00054009934,0.012971083,0.016615508],"category_scores_gemma":[0.02784237,0.00026572842,0.0009028724,0.00040082083,0.0020173648,0.0023818938,0.001144824,0.015668163,0.010461526],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001695487,0.000034000055,0.0014045677,0.0000647481,0.000017753264,0.0019398909,0.00025377457,0.000065045606,0.0003960749,0.012762784,0.94207585,0.040815823],"study_design_scores_gemma":[0.00009942303,0.00013610358,0.0019618955,0.0006454296,0.000046030185,0.0053416914,0.0014504181,0.0009021263,0.0011624378,0.05379623,0.93438596,0.000072310126],"about_ca_topic_score_codex":0.0039363066,"about_ca_topic_score_gemma":0.010446673,"teacher_disagreement_score":0.016615508,"about_ca_system_score_codex":0.0019485228,"about_ca_system_score_gemma":0.0024308644,"threshold_uncertainty_score":0.05558443},"labels":[],"label_agreement":null},{"id":"W3007223417","doi":"10.5539/ijel.v10n2p311","title":"A Corpus-Based Study on Mood Combination Preference in Two-Clause Composite Sentences in Modern Chinese","year":2020,"lang":"en","type":"article","venue":"International Journal of English Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Mood; Sentence; Non-finite clause; Dependent clause; Psychology; Preference; Linguistics; Meaning (existential); Realization (probability); Adverb; Interrogative; Cognitive psychology; Interpretation (philosophy); Computer science; Natural language processing; Mathematics; Social psychology; Statistics; Noun; Philosophy; Psychotherapist","score_opus":0.025737020542446525,"score_gpt":0.31991480309387665,"score_spread":0.2941777825514301,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3007223417","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9963792,0.00026694048,0.00042026647,0.000042522584,0.000009505842,0.000051189163,0.0014652242,0.000023777378,0.0013413769],"genre_scores_gemma":[0.9933295,0.0002185013,0.0010371825,0.000032999083,0.000021120803,0.00007360398,0.0045391642,0.00001850117,0.0007294367],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9994442,0.00012239815,0.000091490925,0.00018787962,0.000105864965,0.000048121794],"domain_scores_gemma":[0.99715436,0.001614652,0.0003517883,0.0002093609,0.0005158523,0.00015399573],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000850451,0.00033409125,0.00040705787,0.003237074,0.0013938985,0.0006501659,0.00038554906,0.0003466934,0.002808291],"category_scores_gemma":[0.0034941735,0.00022036061,0.00033605567,0.00469872,0.00076341047,0.0009942576,0.00062487065,0.00036302584,0.00037105542],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013751625,0.0008613487,0.6827233,0.0031460994,0.00043272355,0.0067425077,0.051297158,0.001458584,0.058522575,0.0033972948,0.011539874,0.17850342],"study_design_scores_gemma":[0.000046824625,0.00015900972,0.9748209,0.0000438589,0.00023945725,0.0014509439,0.007283899,0.0042600497,0.0044379793,0.0002445483,0.0069514522,0.000060971837],"about_ca_topic_score_codex":0.03638429,"about_ca_topic_score_gemma":0.055875126,"teacher_disagreement_score":0.03638429,"about_ca_system_score_codex":0.0010830263,"about_ca_system_score_gemma":0.0010572703,"threshold_uncertainty_score":0.07234502},"labels":[],"label_agreement":null},{"id":"W3007937028","doi":"10.5821/dissertation-2117-177244","title":"Document-level machine translation : ensuring translational consistency of non-local phenomena","year":2019,"lang":"en","type":"dissertation","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Alberta-Pacific Forest Industries","keywords":"Computer science; Machine translation; Natural language processing; Consistency (knowledge bases); Artificial intelligence; Coreference; Context (archaeology); Sentence; Focus (optics); Word (group theory); Source text; Process (computing); Translation (biology); Linguistics; Resolution (logic)","score_opus":0.01734654602421447,"score_gpt":0.28145068257467204,"score_spread":0.26410413655045756,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3007937028","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017015424,0.00040716835,0.9802798,0.00015038475,0.000056684526,0.000049598053,0.0000681643,0.00064042583,0.0013324296],"genre_scores_gemma":[0.31940112,0.00095364527,0.6722983,0.00027276104,0.00029269326,0.00025007807,0.0008395822,0.0013070181,0.0043848357],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9965491,0.0013162046,0.00028761697,0.0010388588,0.00062860094,0.00017960854],"domain_scores_gemma":[0.99031466,0.0040012244,0.001147175,0.0030996848,0.0012985938,0.00013867434],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037833522,0.00096642453,0.0012573405,0.0007883771,0.00088542176,0.0022879806,0.0015004099,0.0012798858,0.002185777],"category_scores_gemma":[0.015963847,0.00056570175,0.0008881713,0.0017525002,0.0014069531,0.0037094394,0.0028737097,0.0017245577,0.0028488894],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00059243705,0.00021380448,0.0035258753,0.0010912814,0.0002153008,0.00083130825,0.0018785676,0.15643533,0.14041911,0.0794685,0.0047924975,0.61053586],"study_design_scores_gemma":[0.000081707374,0.00046442528,0.0018272684,0.00010094249,0.00015084946,0.00085078616,0.00041082618,0.7930038,0.106274575,0.07818715,0.0185562,0.00009156606],"about_ca_topic_score_codex":0.00089492725,"about_ca_topic_score_gemma":0.0007697993,"teacher_disagreement_score":0.0037833522,"about_ca_system_score_codex":0.0005686743,"about_ca_system_score_gemma":0.001337145,"threshold_uncertainty_score":0.020008504},"labels":[],"label_agreement":null},{"id":"W3011800382","doi":"10.1017/9781108674553","title":"An Advanced Introduction to Semantics: A Meaning-Text Approach","year":2020,"lang":"en","type":"book","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University; Université de Montréal","funders":"","keywords":"Computer science; Semantics (computer science); Natural language processing; Computational semantics; Meaning (existential); Representation (politics); Linguistics; Artificial intelligence; Natural language; Sentence; Set (abstract data type); Natural language understanding; Knowledge representation and reasoning; Programming language; Operational semantics; Psychology","score_opus":0.010083839003709762,"score_gpt":0.25507643130761176,"score_spread":0.244992592303902,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3011800382","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0010177525,0.04780287,0.1502078,0.013778512,0.0058549405,0.00008856771,0.00067695364,0.0010693723,0.7795031],"genre_scores_gemma":[0.035657827,0.06563775,0.1534725,0.007056489,0.007057728,0.00031782282,0.0016074292,0.0017244137,0.72746813],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9995882,0.00009597261,0.00002733115,0.000064150925,0.00019589582,0.000028554678],"domain_scores_gemma":[0.99957436,0.00026466575,0.000014198719,0.000034722252,0.000077610224,0.000034527693],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005275619,0.0009691086,0.0006238849,0.0019005819,0.0011781433,0.0041889157,0.0010009906,0.0013542417,0.031321734],"category_scores_gemma":[0.0011277727,0.00049619505,0.0008191141,0.0026455694,0.003120756,0.008297914,0.0015585982,0.0037615304,0.01774652],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000008187867,0.000020016161,0.000041564057,0.00019106577,0.0000061862347,0.00008698456,0.00038533166,0.0004912168,0.00044589787,0.7828093,0.14833799,0.06717633],"study_design_scores_gemma":[0.000001486839,0.0000062278323,0.000048690126,0.00008885953,0.0000018950644,0.0001683568,0.000076840086,0.0004738025,0.00011300352,0.29002246,0.7089929,0.000005593688],"about_ca_topic_score_codex":0.0010773599,"about_ca_topic_score_gemma":0.0017452096,"teacher_disagreement_score":0.031321734,"about_ca_system_score_codex":0.0016695227,"about_ca_system_score_gemma":0.0011404788,"threshold_uncertainty_score":0.10478163},"labels":[],"label_agreement":null},{"id":"W3012481964","doi":"10.5220/0009168200760083","title":"Model Transformation by Example with Statistical Machine Translation","year":2020,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Machine translation; Translation (biology); Transformation (genetics); Artificial intelligence; Model transformation; Natural language processing","score_opus":0.0313576534232818,"score_gpt":0.2606412706402783,"score_spread":0.22928361721699653,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3012481964","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007020916,0.000102892394,0.9771433,0.0004623279,0.00012714398,0.00006843347,0.0005853517,0.0062431893,0.008246442],"genre_scores_gemma":[0.27858844,0.00027377938,0.70835936,0.0002537091,0.00007410837,0.0001973353,0.0024410465,0.0017294581,0.008082784],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99890065,0.0004872927,0.00006885597,0.0001991963,0.00028691272,0.000057134308],"domain_scores_gemma":[0.9985367,0.0006899258,0.00004779785,0.00048897136,0.0002178869,0.00001872637],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007185702,0.00072211,0.000574716,0.000781024,0.0005565409,0.0012068293,0.0009226227,0.0009521628,0.013836777],"category_scores_gemma":[0.0044419267,0.00044206152,0.0012596132,0.00092370017,0.00049138145,0.0016347201,0.0015969656,0.0013688919,0.0055810474],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004885392,0.00025939711,0.0015223351,0.0007662001,0.00020475745,0.0019549378,0.0006059937,0.12610541,0.019078646,0.21137357,0.041247968,0.5963922],"study_design_scores_gemma":[0.00006898159,0.00009116302,0.00031543142,0.00006199933,0.000077769844,0.0006867071,0.00012709721,0.7300426,0.029175706,0.19632304,0.0429936,0.00003602245],"about_ca_topic_score_codex":0.0013814081,"about_ca_topic_score_gemma":0.0026057158,"teacher_disagreement_score":0.013836777,"about_ca_system_score_codex":0.00032557052,"about_ca_system_score_gemma":0.0008185913,"threshold_uncertainty_score":0.04628861},"labels":[],"label_agreement":null},{"id":"W3013357622","doi":"","title":"Rhetorical Figure Annotation with XML.","year":2016,"lang":"en","type":"article","venue":"International Joint Conference on Artificial Intelligence","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Annotation; XML; Rhetorical question; Streaming XML; Document Structure Description; Information retrieval; World Wide Web; Programming language; Artificial intelligence; Linguistics","score_opus":0.08496063314218631,"score_gpt":0.3292529870922604,"score_spread":0.2442923539500741,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3013357622","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005568987,0.0009264815,0.7845275,0.0024458985,0.0012022778,0.0003978953,0.032315563,0.055610735,0.11700466],"genre_scores_gemma":[0.10364737,0.0017567262,0.7084883,0.0012286595,0.0005263925,0.00058017566,0.08611539,0.01451819,0.083138876],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9979824,0.000732423,0.00025119673,0.000325909,0.00060763385,0.00010047414],"domain_scores_gemma":[0.99549,0.0021736396,0.0002003504,0.0010078148,0.0010178909,0.000110344634],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002425462,0.00095476676,0.000673824,0.004340659,0.0014644216,0.004006345,0.0019863672,0.0018644761,0.058255114],"category_scores_gemma":[0.009934767,0.00088458194,0.00093744363,0.0031436607,0.0008993333,0.006431383,0.0037124483,0.0021885773,0.026597133],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004137303,0.00013923332,0.0015505109,0.0022863974,0.0000872793,0.0010771469,0.0035540406,0.0032769614,0.012177415,0.32204023,0.33804533,0.31535167],"study_design_scores_gemma":[0.000032477907,0.000022813783,0.00045171918,0.0005188955,0.000043783064,0.00029870743,0.00045632423,0.013993874,0.0127130505,0.06223851,0.9091906,0.000039224655],"about_ca_topic_score_codex":0.004503469,"about_ca_topic_score_gemma":0.006193157,"teacher_disagreement_score":0.058255114,"about_ca_system_score_codex":0.0010864539,"about_ca_system_score_gemma":0.0020986532,"threshold_uncertainty_score":0.19488275},"labels":[],"label_agreement":null},{"id":"W3013461566","doi":"10.7202/1068022ar","title":"Repérage des décalages informationnels de traduction au moyen du criblage automatique des segments hétéromorphes d’un corpus parallèle","year":2020,"lang":"fr","type":"article","venue":"TTR traduction terminologie rédaction","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"Université du Québec à Trois-Rivières","funders":"","keywords":"Humanities; Art; Philosophy","score_opus":0.07088031131719662,"score_gpt":0.2892182493132572,"score_spread":0.21833793799606055,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3013461566","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.33083904,0.00562821,0.60520476,0.0031479306,0.0010482001,0.0012640214,0.008290973,0.008169722,0.03640724],"genre_scores_gemma":[0.36674798,0.0010397948,0.5988439,0.00033890395,0.00016444921,0.0011247039,0.008255234,0.0023237404,0.02116136],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9924665,0.0021368538,0.00075192255,0.002496463,0.0018292562,0.0003190731],"domain_scores_gemma":[0.97122395,0.015268956,0.0015236161,0.0038944043,0.007760526,0.0003284801],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0059389262,0.0013713758,0.0010949908,0.0042978083,0.0021189023,0.004989241,0.0014572588,0.0014722134,0.013559942],"category_scores_gemma":[0.034902662,0.0010311936,0.00103165,0.0049198414,0.0023587572,0.0048224484,0.002982863,0.0024347524,0.0054980153],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019301417,0.00013686191,0.013659281,0.0032351147,0.00030625126,0.0011425284,0.02662381,0.0050638486,0.11762949,0.021775603,0.019097678,0.7893995],"study_design_scores_gemma":[0.0003681083,0.0008301542,0.07599648,0.0016258453,0.00084516994,0.0033617897,0.028319804,0.11132209,0.25573388,0.044901088,0.47610807,0.00058749143],"about_ca_topic_score_codex":0.013249647,"about_ca_topic_score_gemma":0.019489784,"teacher_disagreement_score":0.013559942,"about_ca_system_score_codex":0.0020612678,"about_ca_system_score_gemma":0.00351622,"threshold_uncertainty_score":0.045362532},"labels":[],"label_agreement":null},{"id":"W3013646476","doi":"","title":"WaterlooClarke at the TREC 2019 Conversational Assistant Track.","year":2019,"lang":"en","type":"article","venue":"Text REtrieval Conference","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Track (disk drive); Artificial intelligence; Natural language processing; World Wide Web; Multimedia; Operating system","score_opus":0.015280036519228327,"score_gpt":0.25267818070601816,"score_spread":0.23739814418678984,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3013646476","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011779571,0.014904829,0.056634862,0.0600019,0.045535445,0.0031936592,0.21026652,0.028680753,0.5690025],"genre_scores_gemma":[0.011703557,0.0024916222,0.015876776,0.003845547,0.0026078639,0.00053721195,0.08272737,0.0016751642,0.8785349],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99818295,0.00043208702,0.00006721081,0.00036034427,0.0006767044,0.00028067172],"domain_scores_gemma":[0.99604875,0.00053420727,0.00007979164,0.00026031633,0.0018217997,0.0012551951],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.004510692,0.0015325506,0.0019389053,0.0019700432,0.003663611,0.005401331,0.0017093371,0.001901205,0.31464654],"category_scores_gemma":[0.0053135455,0.0006013637,0.0005520565,0.0020709895,0.0006956467,0.0049832505,0.0024227088,0.0024812273,0.15327705],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007305862,0.00005271658,0.00006307087,0.00006269767,0.000007070028,0.000020246129,0.000032727377,0.000040118433,0.00087110786,0.0005153955,0.9858078,0.012454148],"study_design_scores_gemma":[0.00009765957,0.00007961549,0.0012433855,0.00007472182,0.000021362803,0.000042721553,0.00024853577,0.0013824549,0.0019676406,0.0020024115,0.99279845,0.000041037456],"about_ca_topic_score_codex":0.124098636,"about_ca_topic_score_gemma":0.39462975,"teacher_disagreement_score":0.31464654,"about_ca_system_score_codex":0.004086189,"about_ca_system_score_gemma":0.00661627,"threshold_uncertainty_score":0.9775735},"labels":[],"label_agreement":null},{"id":"W3015694352","doi":"10.3989/loquens.2019.062","title":"Rule Interaction Conversion Operations","year":2019,"lang":"en","type":"article","venue":"Loquens","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Pairwise comparison; Set (abstract data type); Computer science; Theoretical computer science; Mathematics; Artificial intelligence; Programming language","score_opus":0.00879885792406336,"score_gpt":0.26657843811869925,"score_spread":0.25777958019463587,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3015694352","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.033636417,0.00060883665,0.83984625,0.00059041515,0.0007589086,0.0008278323,0.0016348982,0.012672388,0.10942398],"genre_scores_gemma":[0.36258233,0.00061248563,0.5948994,0.00083406863,0.0003389453,0.0013176453,0.0025768809,0.0025545328,0.03428378],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99619293,0.00082229025,0.00051528076,0.00087351055,0.0012401738,0.0003557623],"domain_scores_gemma":[0.9945496,0.002299818,0.00025659395,0.0019877446,0.00077700726,0.000129241],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002480288,0.0011776446,0.0006805362,0.002086435,0.0012914594,0.0037871625,0.0022441489,0.0012411659,0.020690491],"category_scores_gemma":[0.009693844,0.00078971323,0.0016901426,0.001357863,0.002972735,0.0049104593,0.0033548037,0.0031740041,0.0056160702],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002640059,0.0002131378,0.0015726065,0.00039165773,0.00007634115,0.0012283371,0.0015851877,0.0034779243,0.0219468,0.7551166,0.011773338,0.20235403],"study_design_scores_gemma":[0.00010011087,0.00016572399,0.0016750161,0.00020803738,0.00016281445,0.0027614804,0.00080792035,0.04094279,0.052355304,0.64196837,0.2586294,0.00022293966],"about_ca_topic_score_codex":0.0010317788,"about_ca_topic_score_gemma":0.0011555782,"teacher_disagreement_score":0.020690491,"about_ca_system_score_codex":0.0010067276,"about_ca_system_score_gemma":0.0010802051,"threshold_uncertainty_score":0.06921661},"labels":[],"label_agreement":null},{"id":"W3015757654","doi":"10.1109/icassp40776.2020.9053236","title":"From Unsupervised Machine Translation to Adversarial Text Generation","year":2020,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada)","funders":"","keywords":"Computer science; Machine translation; Encoder; Generator (circuit theory); Artificial intelligence; Representation (politics); Text generation; Natural language processing; Translation (biology); Adversarial system; Speech recognition; Domain (mathematical analysis); Power (physics); Mathematics","score_opus":0.03474685314239162,"score_gpt":0.2652846707236395,"score_spread":0.23053781758124786,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3015757654","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013539872,0.00041379465,0.97712666,0.00048484677,0.00011216686,0.00008267457,0.00023243575,0.0023106646,0.0056968657],"genre_scores_gemma":[0.5655666,0.0005883813,0.41678748,0.00079011684,0.00022725771,0.0004411844,0.0015511751,0.00075718336,0.01329057],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99938357,0.00029953022,0.000021207841,0.00011811943,0.00013258289,0.0000449015],"domain_scores_gemma":[0.99868983,0.0008685194,0.000071233044,0.00022466296,0.00011013195,0.00003568471],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011007977,0.0009176255,0.00053008646,0.0003214237,0.00027389737,0.0005291006,0.0009182964,0.00085657666,0.003738081],"category_scores_gemma":[0.0032518266,0.00030790624,0.00046128125,0.0003376593,0.00078646484,0.0009253461,0.0013216389,0.0013865505,0.0015219896],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002473939,0.00012016128,0.0005057573,0.00019213698,0.000075772776,0.0003210746,0.00011843381,0.752998,0.01461072,0.05164236,0.0123214945,0.16684666],"study_design_scores_gemma":[0.000017639553,0.00004717966,0.00006492334,0.000010555403,0.000005885195,0.00006978961,0.0000073953297,0.97134596,0.004940841,0.020703387,0.0027776626,0.000008831219],"about_ca_topic_score_codex":0.0007452644,"about_ca_topic_score_gemma":0.0012027193,"teacher_disagreement_score":0.003738081,"about_ca_system_score_codex":0.00046675085,"about_ca_system_score_gemma":0.0005834867,"threshold_uncertainty_score":0.012505114},"labels":[],"label_agreement":null},{"id":"W3016027609","doi":"10.5281/zenodo.3629884","title":"TEI Lex-0 In Action: Improving the Encoding of the Dictionary of the Academia das Ciências de Lisboa","year":2019,"lang":"en","type":"article","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Linguistic Association","funders":"Fundação para a Ciência e a Tecnologia; Horizon 2020 Framework Programme; Universidade Nova de Lisboa; European Commission","keywords":"Encoding (memory); Computer science; Context (archaeology); Natural language processing; Interoperability; Artificial intelligence; Information retrieval; World Wide Web; History","score_opus":0.025758744792893595,"score_gpt":0.2688100019633009,"score_spread":0.24305125717040732,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3016027609","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1621661,0.0010654271,0.7147448,0.002543909,0.0011312566,0.0007314059,0.017133398,0.045691185,0.05479256],"genre_scores_gemma":[0.36447325,0.00070687704,0.57859784,0.000553406,0.00006647411,0.0003650422,0.024723545,0.011718092,0.01879541],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9985305,0.00042159724,0.00023420046,0.00028327285,0.00039482658,0.00013552647],"domain_scores_gemma":[0.9937302,0.0020872296,0.00038957206,0.0018751633,0.0016736395,0.0002441846],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015225505,0.00062584097,0.0006643609,0.0016305192,0.0007513464,0.003444282,0.0010088856,0.00078486785,0.009835473],"category_scores_gemma":[0.009345046,0.00049259135,0.00059805886,0.0026720108,0.0009951565,0.004358848,0.0027853996,0.0017324144,0.0035878513],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002835431,0.00044566233,0.0074461,0.0023896287,0.00008459165,0.0012113306,0.009278175,0.014121212,0.057577368,0.17363158,0.053887323,0.6770916],"study_design_scores_gemma":[0.00026126477,0.0005066428,0.003594286,0.00058212527,0.00011281016,0.0015904133,0.0031110642,0.08432673,0.16280843,0.03595754,0.7068966,0.00025200087],"about_ca_topic_score_codex":0.006555666,"about_ca_topic_score_gemma":0.007675377,"teacher_disagreement_score":0.009835473,"about_ca_system_score_codex":0.0023892042,"about_ca_system_score_gemma":0.0030890205,"threshold_uncertainty_score":0.032902956},"labels":[],"label_agreement":null},{"id":"W3016980879","doi":"10.5539/ijel.v10n3p241","title":"On the Explanatory Power of the Hypothesis of Semantic Determination to Causative Alternation","year":2020,"lang":"en","type":"article","venue":"International Journal of English Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"Linyi University","keywords":"Alternation (linguistics); Verb; Meaning (existential); Causative; Linguistics; Explanatory power; Semantic property; Computer science; Class (philosophy); Component (thermodynamics); Psychology; Natural language processing; Artificial intelligence; Epistemology; Philosophy","score_opus":0.020589216625596925,"score_gpt":0.27642705328440903,"score_spread":0.2558378366588121,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3016980879","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6327831,0.0014888841,0.27252018,0.01501156,0.00076842343,0.00031389968,0.0009060754,0.0007584729,0.07544944],"genre_scores_gemma":[0.98307276,0.00037439785,0.013675712,0.00056802324,0.00031633262,0.00011931603,0.00040193542,0.000070885224,0.001400694],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9959472,0.0017603617,0.00020557045,0.0010455173,0.00071631564,0.0003251082],"domain_scores_gemma":[0.91867006,0.071247175,0.0033238954,0.0033831901,0.002392947,0.0009827145],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010533528,0.0009021174,0.0009082209,0.0039040793,0.0014987235,0.0029578228,0.0023839888,0.0026605304,0.012961811],"category_scores_gemma":[0.04943358,0.00052022486,0.0015628916,0.0018833607,0.010191084,0.00856895,0.0029428953,0.0021951871,0.0011834302],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014587637,0.0003941565,0.06653304,0.0004543979,0.00021600425,0.0021832928,0.0045525786,0.008990556,0.001790459,0.8588043,0.0034593209,0.051163077],"study_design_scores_gemma":[0.00027871353,0.00040044173,0.029396664,0.00018197567,0.00018899173,0.0012755976,0.0023700243,0.07678226,0.0018006145,0.8811191,0.0060999547,0.00010577118],"about_ca_topic_score_codex":0.0018829708,"about_ca_topic_score_gemma":0.0007468908,"teacher_disagreement_score":0.012961811,"about_ca_system_score_codex":0.00075253594,"about_ca_system_score_gemma":0.001301186,"threshold_uncertainty_score":0.055707276},"labels":[],"label_agreement":null},{"id":"W3017045910","doi":"10.1075/ivitra.24.08lho","title":"Collecting collocations from general and specialised corpora","year":2020,"lang":"en","type":"book-chapter","venue":"IVITRA research in linguistics and literature","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Natural language processing; Linguistics; Computer science; Artificial intelligence; History; Philosophy","score_opus":0.06525551906718656,"score_gpt":0.34953108434128943,"score_spread":0.28427556527410286,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3017045910","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8396538,0.0027075016,0.11082172,0.00045864264,0.0001784223,0.0010093236,0.02259496,0.0017177657,0.020857777],"genre_scores_gemma":[0.6378834,0.0016084409,0.29502684,0.0001662683,0.00012212642,0.0010674648,0.056186594,0.001150258,0.0067885625],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99553937,0.0015451913,0.00070047565,0.0010937029,0.0009357493,0.00018546797],"domain_scores_gemma":[0.97440016,0.013499593,0.0015190692,0.005853482,0.0042134705,0.00051421043],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036577033,0.00042191803,0.00076435535,0.013487449,0.0019800758,0.002149255,0.00090699515,0.00074158434,0.005715468],"category_scores_gemma":[0.016021727,0.0005671757,0.0004500596,0.014767838,0.001598283,0.002604331,0.0032226543,0.00077855395,0.002374277],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007051208,0.00030569473,0.11636915,0.0056143594,0.0003163007,0.0041444777,0.047944963,0.003975247,0.16486365,0.0138468165,0.036281478,0.60563284],"study_design_scores_gemma":[0.0001395972,0.00036668123,0.4760817,0.0012336664,0.0003710453,0.007824872,0.02505036,0.01791014,0.079466045,0.013209009,0.37804136,0.00030558486],"about_ca_topic_score_codex":0.0024561826,"about_ca_topic_score_gemma":0.0061339857,"teacher_disagreement_score":0.013487449,"about_ca_system_score_codex":0.0007975431,"about_ca_system_score_gemma":0.0010996222,"threshold_uncertainty_score":0.019343972},"labels":[],"label_agreement":null},{"id":"W301965306","doi":"10.63317/52wv2es2kv7z","title":"Finding Semantic Associations on Express Lane","year":2004,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Scheme (mathematics); Computation; Semantic computing; Semantic search; Operator (biology); Semantic grid; Semantics (computer science); Information retrieval; Semantic technology; Theoretical computer science; Semantic Web; Artificial intelligence; Algorithm; Mathematics; Programming language","score_opus":0.019184670948980832,"score_gpt":0.2883790241723465,"score_spread":0.26919435322336566,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W301965306","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.036439918,0.00014850238,0.95173603,0.00030849123,0.00009792688,0.00008562514,0.00042317345,0.0007476957,0.010012557],"genre_scores_gemma":[0.35781,0.00028883418,0.6282129,0.00017106194,0.00010446261,0.00026185697,0.0014932096,0.00032000482,0.011337586],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986021,0.00020742882,0.00012946861,0.0004123116,0.00045018992,0.00019838578],"domain_scores_gemma":[0.9985612,0.00033998166,0.00018482385,0.0004196728,0.00038001337,0.000114348964],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012718508,0.00059027283,0.00057161716,0.0034846708,0.0020774712,0.0042441413,0.0017590524,0.0012694775,0.008117683],"category_scores_gemma":[0.0062355977,0.0005403395,0.0010665979,0.0029186585,0.0022860025,0.010346436,0.0049154176,0.0015788354,0.002049247],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007130023,0.00003817591,0.0013693443,0.00008566085,0.000019994746,0.00016690367,0.00048905937,0.006720271,0.004488709,0.91394824,0.0017125744,0.070889875],"study_design_scores_gemma":[0.000010042193,0.000034363646,0.0006604345,0.000047957947,0.000032483003,0.0001512834,0.0002965165,0.07874398,0.007762219,0.88813573,0.024089454,0.00003554021],"about_ca_topic_score_codex":0.0028036453,"about_ca_topic_score_gemma":0.0031018637,"teacher_disagreement_score":0.008117683,"about_ca_system_score_codex":0.0012811109,"about_ca_system_score_gemma":0.00097498205,"threshold_uncertainty_score":0.027156353},"labels":[],"label_agreement":null},{"id":"W3020039577","doi":"10.48550/arxiv.2004.12527","title":"Neural Machine Translation with Monte-Carlo Tree Search","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Leverage (statistics); Machine translation; Monte Carlo tree search; Artificial intelligence; Machine learning; Artificial neural network; Value network; Value (mathematics); Translation (biology); Tree (set theory); Word (group theory); Monte Carlo method; Mathematics; Statistics","score_opus":0.0870489873693031,"score_gpt":0.2088420559375958,"score_spread":0.12179306856829271,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3020039577","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013287902,0.00084199925,0.97665215,0.0005879137,0.00015835461,0.00008463978,0.00010700655,0.0019266867,0.0063533657],"genre_scores_gemma":[0.553299,0.00050660264,0.43226817,0.00060033286,0.00024743914,0.000521882,0.0006833643,0.00065037154,0.011222837],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.998992,0.0005009151,0.000053358784,0.00019901154,0.00018061858,0.00007403853],"domain_scores_gemma":[0.9962555,0.0029557056,0.00016767513,0.00021840572,0.00032129962,0.000081476705],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016892496,0.0011426928,0.0017338515,0.0010345387,0.00079943775,0.0013391025,0.0018046313,0.0023372197,0.0058277426],"category_scores_gemma":[0.008345236,0.00090151164,0.0008680881,0.001817834,0.0011435203,0.0016295251,0.0011225197,0.002257783,0.0020113047],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000119818804,0.000047459518,0.0003291404,0.00008311812,0.00005705046,0.00009141756,0.000059880986,0.89924693,0.00059088634,0.024297917,0.0037175792,0.071358845],"study_design_scores_gemma":[0.000011054403,0.000006321284,0.000014454433,0.00000451181,0.0000031170625,0.0000070131105,0.000002466164,0.99019974,0.0001341499,0.009269556,0.0003449958,0.0000025577638],"about_ca_topic_score_codex":0.006968823,"about_ca_topic_score_gemma":0.011643426,"teacher_disagreement_score":0.006968823,"about_ca_system_score_codex":0.0016333751,"about_ca_system_score_gemma":0.001954702,"threshold_uncertainty_score":0.019495785},"labels":[],"label_agreement":null},{"id":"W3020103006","doi":"10.24506/jsda.4.2_165","title":"[B23] KuroNet: End-to-end Kuzushiji Transcription System for Understanding Historical Documents","year":2020,"lang":"zh","type":"article","venue":"デジタルアーカイブ学会誌","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; Mila - Quebec Artificial Intelligence Institute","funders":"","keywords":"Transcription (linguistics); End-to-end principle; Computer science; Computer security; Linguistics; Philosophy","score_opus":0.06758246242188858,"score_gpt":0.2841860408501607,"score_spread":0.2166035784282721,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3020103006","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015760398,0.0007920792,0.77765495,0.0005541345,0.00071881304,0.00040666226,0.041054763,0.14684647,0.016211761],"genre_scores_gemma":[0.06289489,0.0006893414,0.7888618,0.0003326289,0.0001865871,0.0004868135,0.100590296,0.008717167,0.037240516],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9995851,0.00004749552,0.00006739191,0.00013190368,0.00011848751,0.000049638635],"domain_scores_gemma":[0.999121,0.00014190351,0.0000612956,0.00016877144,0.00045385357,0.00005305911],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00043260324,0.0017368955,0.0007007999,0.002250122,0.0010614631,0.0010546132,0.0014177901,0.0015310504,0.03295087],"category_scores_gemma":[0.0024296758,0.0008415964,0.00069998024,0.002153534,0.00046487214,0.0022953989,0.0014137062,0.0011768221,0.033378568],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007880523,0.00009627654,0.001834846,0.0011993322,0.00009983117,0.0011193305,0.001507237,0.0014354017,0.16797127,0.0040671383,0.24039471,0.57948655],"study_design_scores_gemma":[0.00025114117,0.00024864532,0.010917001,0.00026946934,0.0003606995,0.0027981924,0.0012738066,0.051108725,0.23682973,0.0061525665,0.68936247,0.00042744234],"about_ca_topic_score_codex":0.010640216,"about_ca_topic_score_gemma":0.015148431,"teacher_disagreement_score":0.03295087,"about_ca_system_score_codex":0.0003494899,"about_ca_system_score_gemma":0.001053636,"threshold_uncertainty_score":0.1102317},"labels":[],"label_agreement":null},{"id":"W3020909159","doi":"10.48550/arxiv.2004.13886","title":"Synonymy = Translational Equivalence","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Equivalence (formal languages); Computer science; Semantic equivalence; Principal (computer security); Semantics (computer science); Meaning (existential); Field (mathematics); Natural language processing; Artificial intelligence; Mathematics; Programming language; Pure mathematics; Psychology","score_opus":0.0872343428309226,"score_gpt":0.21342336546069302,"score_spread":0.12618902262977041,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3020909159","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05432393,0.00512271,0.6168365,0.0146186985,0.006266709,0.0005594476,0.0038090611,0.002696872,0.29576612],"genre_scores_gemma":[0.7564356,0.0036532045,0.19422953,0.0067833727,0.0029544034,0.0011943849,0.0042857663,0.0014612831,0.029002557],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.98668426,0.0039118906,0.0020792603,0.0038932464,0.0026214158,0.0008099606],"domain_scores_gemma":[0.9857531,0.004537701,0.0013044499,0.00513158,0.0027492486,0.0005238873],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004952068,0.0009437577,0.0014087699,0.002502606,0.002821577,0.007452015,0.0013264593,0.0025657364,0.017905207],"category_scores_gemma":[0.021959897,0.0006262539,0.0013799617,0.0028047585,0.012569048,0.025033794,0.009054113,0.0046178517,0.006226156],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000057179583,0.00002643071,0.000820854,0.00024177131,0.00004102411,0.000098322416,0.0010874149,0.00026576282,0.0013700569,0.9555501,0.0074980943,0.0329431],"study_design_scores_gemma":[0.000024735875,0.000038355833,0.00049551314,0.000070180984,0.000037594276,0.00048128061,0.00061704725,0.00081871793,0.001087663,0.9391766,0.057123587,0.000028606628],"about_ca_topic_score_codex":0.0011886072,"about_ca_topic_score_gemma":0.0010540803,"teacher_disagreement_score":0.017905207,"about_ca_system_score_codex":0.0015573842,"about_ca_system_score_gemma":0.0024417718,"threshold_uncertainty_score":0.059898853},"labels":[],"label_agreement":null},{"id":"W3021520429","doi":"10.1007/978-3-031-02138-1","title":"Cross-Language Information Retrieval","year":2010,"lang":"en","type":"book","venue":"Synthesis lectures on human language technologies","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":109,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Cross-language information retrieval; Artificial intelligence; Natural language processing; Information retrieval; Linguistics; Machine translation; Philosophy","score_opus":0.011538167100995443,"score_gpt":0.2955057917965311,"score_spread":0.28396762469553566,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3021520429","genre_codex":"other","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0070682587,0.08329187,0.42142975,0.0023835432,0.006876368,0.000224032,0.0027490093,0.011337247,0.4646399],"genre_scores_gemma":[0.030228872,0.021645214,0.116847225,0.0012432262,0.0020449995,0.0001411146,0.008510413,0.0022374787,0.8171014],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99937075,0.000056100627,0.000051075185,0.00015455602,0.00031417588,0.00005340009],"domain_scores_gemma":[0.99937063,0.00015378822,0.0000286966,0.00014441843,0.00025520037,0.000047283604],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00073109847,0.0013992632,0.0012569443,0.0035663128,0.00063212967,0.0034142907,0.0012023682,0.00078823284,0.091499604],"category_scores_gemma":[0.001462885,0.00053006725,0.00089793594,0.0045160414,0.0005350576,0.0060467967,0.0019657565,0.0010695211,0.07154461],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005163432,0.00006241315,0.00009736546,0.00049302075,0.000051822568,0.00007847514,0.00006464534,0.00047135536,0.0079295635,0.02050977,0.22349125,0.7466986],"study_design_scores_gemma":[0.00001982119,0.000084818654,0.0008272881,0.0002258925,0.000091222,0.0013695245,0.00007922467,0.0059409016,0.017047537,0.038521226,0.93574286,0.00004970052],"about_ca_topic_score_codex":0.00081525836,"about_ca_topic_score_gemma":0.0012667503,"teacher_disagreement_score":0.091499604,"about_ca_system_score_codex":0.000725483,"about_ca_system_score_gemma":0.00083924277,"threshold_uncertainty_score":0.30609667},"labels":[],"label_agreement":null},{"id":"W3022586029","doi":"10.1007/978-3-030-47358-7_28","title":"From Explicit to Implicit Entity Linking: A Learn to Rank Framework","year":2020,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Computer science; Entity linking; Task (project management); Phrase; Natural language processing; Feature (linguistics); Rank (graph theory); Ranking (information retrieval); Information retrieval; Artificial intelligence; Linguistics; Knowledge base; Mathematics","score_opus":0.016466575616550433,"score_gpt":0.2771304388561439,"score_spread":0.26066386323959345,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3022586029","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012772185,0.0013126375,0.97632056,0.0009915065,0.00014982591,0.000081751255,0.0011045233,0.0039532944,0.0033136252],"genre_scores_gemma":[0.34847495,0.0017481704,0.6187403,0.0008749029,0.00082070235,0.0002667433,0.0066359993,0.0010594289,0.021378716],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9970174,0.00097560807,0.00017263982,0.0008608619,0.00065604004,0.0003174112],"domain_scores_gemma":[0.9916544,0.004710903,0.00050948345,0.0017594112,0.0010797941,0.00028600267],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0042273803,0.0019628326,0.0027159024,0.0032737197,0.0013273516,0.0043645906,0.0053591616,0.004104899,0.009724383],"category_scores_gemma":[0.013131278,0.0011407025,0.0016441246,0.003776567,0.0017306685,0.011460329,0.004691269,0.005523442,0.0047066165],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00060117577,0.0006235217,0.0021806706,0.00038512127,0.0002641513,0.00023552027,0.0002408382,0.10586541,0.0035101005,0.072471194,0.035048388,0.77857393],"study_design_scores_gemma":[0.000044081346,0.00011600692,0.00032960557,0.00005909975,0.00008427041,0.0001237186,0.00008635203,0.8319838,0.0033921825,0.15903956,0.0046925917,0.000048771828],"about_ca_topic_score_codex":0.006430914,"about_ca_topic_score_gemma":0.011909829,"teacher_disagreement_score":0.009724383,"about_ca_system_score_codex":0.001198678,"about_ca_system_score_gemma":0.002171901,"threshold_uncertainty_score":0.03253126},"labels":[],"label_agreement":null},{"id":"W3022934019","doi":"10.13025/29462","title":"A Multilingual Evaluation Dataset for Monolingual Word Sense Alignment","year":2020,"lang":"en","type":"article","venue":"Arrow@dit (Dublin Institute of Technology)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canarie","funders":"Irish Research Council; Science Foundation Ireland; European Commission","keywords":"Word (group theory); Natural language processing; Computer science; Word-sense disambiguation; Linguistics; Artificial intelligence","score_opus":0.04416201849000012,"score_gpt":0.3289890192432291,"score_spread":0.284827000753229,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3022934019","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.039436,0.0014320675,0.00786995,0.000659438,0.00056861894,0.0007515102,0.9189061,0.009689941,0.020686423],"genre_scores_gemma":[0.0094413,0.000117872776,0.0104806535,0.00012811979,0.000025628477,0.00044889154,0.9769995,0.0002924251,0.0020655706],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9952153,0.0012333947,0.00082232396,0.0011716154,0.001215152,0.00034226757],"domain_scores_gemma":[0.9937511,0.0013991119,0.00047548802,0.0017305443,0.001996697,0.00064696034],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034850284,0.002450375,0.0014323374,0.007158222,0.00220443,0.0019536652,0.0031409096,0.0022368599,0.015946897],"category_scores_gemma":[0.0099214325,0.00055484124,0.0015425134,0.0055361805,0.0010049476,0.0037221576,0.0046149115,0.001837715,0.026434086],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00082545413,0.0007783909,0.0074094883,0.0026000314,0.00030248053,0.00045051848,0.00052203576,0.0018980621,0.0075322525,0.004304336,0.8998101,0.07356684],"study_design_scores_gemma":[0.0010137749,0.0005721188,0.035366286,0.0007050454,0.00021865538,0.0015562741,0.0021454329,0.018595928,0.013551378,0.0071459045,0.91888916,0.00024016887],"about_ca_topic_score_codex":0.013665395,"about_ca_topic_score_gemma":0.03336049,"teacher_disagreement_score":0.015946897,"about_ca_system_score_codex":0.0017859652,"about_ca_system_score_gemma":0.0029137365,"threshold_uncertainty_score":0.053347707},"labels":[],"label_agreement":null},{"id":"W3023215888","doi":"10.18653/v1/2020.acl-main.142","title":"TACRED Revisited: A Thorough Evaluation of the TACRED Relation Extraction Task","year":2020,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Bundesministerium für Wirtschaft und Energie; Banting and Best Diabetes Centre, University of Toronto; Bundesministerium für Bildung und Forschung","keywords":"Computer science; Heuristics; Word error rate; Test set; Ceiling (cloud); Task (project management); Artificial intelligence; Machine learning; Categorization; Set (abstract data type); Relation (database); Relationship extraction; Baseline (sea); Test (biology); Natural language processing; Data mining","score_opus":0.052764233413581406,"score_gpt":0.35168764493614646,"score_spread":0.2989234115225651,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3023215888","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.43952736,0.028291665,0.27045545,0.004628534,0.0041193487,0.0024771474,0.057320643,0.12326263,0.06991725],"genre_scores_gemma":[0.5151126,0.0030282652,0.2614387,0.0031000164,0.00049012346,0.00083370705,0.17016089,0.009430806,0.036404893],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98853636,0.004880492,0.000886531,0.0031279668,0.0019837336,0.0005848597],"domain_scores_gemma":[0.97945625,0.009406908,0.0006098997,0.0070589874,0.0028712237,0.00059666694],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0142024765,0.0048860665,0.0021200348,0.003938848,0.0019785725,0.0032170669,0.0054475823,0.0040316563,0.009481296],"category_scores_gemma":[0.030052112,0.0009468564,0.0022362226,0.002971274,0.0014466343,0.007945906,0.0038206086,0.0043214983,0.008641817],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002678733,0.0020049836,0.009012975,0.0052993945,0.001685611,0.001836288,0.0014256184,0.07022765,0.024595706,0.0042464654,0.23701964,0.63996685],"study_design_scores_gemma":[0.0006738497,0.0023494994,0.020257901,0.0011002414,0.0008620145,0.004723191,0.002741831,0.6517718,0.07523753,0.011944482,0.2278597,0.00047793833],"about_ca_topic_score_codex":0.015265028,"about_ca_topic_score_gemma":0.02510771,"teacher_disagreement_score":0.015265028,"about_ca_system_score_codex":0.0020834128,"about_ca_system_score_gemma":0.0017397426,"threshold_uncertainty_score":0.07511079},"labels":[],"label_agreement":null},{"id":"W3023412526","doi":"10.5539/ijel.v10n4p43","title":"Neural Machine Translation: Fine-Grained Evaluation of Google Translate Output for English-to-Arabic Translation","year":2020,"lang":"en","type":"article","venue":"International Journal of English Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Machine translation; Fluency; Computer science; Evaluation of machine translation; Natural language processing; Sentence; Quality (philosophy); Machine translation software usability; Artificial intelligence; Readability; Translation (biology); Arabic; Point (geometry); Example-based machine translation; Linguistics; Mathematics","score_opus":0.05410833016145049,"score_gpt":0.33995500573784576,"score_spread":0.28584667557639526,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3023412526","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9498222,0.001446963,0.023891399,0.0003057907,0.00036817466,0.0008274266,0.0039307745,0.008365332,0.011041943],"genre_scores_gemma":[0.93922126,0.00039146084,0.0432356,0.00011700293,0.00006326299,0.00039163334,0.0120964665,0.0005681437,0.003914991],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9953935,0.0020836382,0.00063886074,0.0004960311,0.0011983691,0.00018962652],"domain_scores_gemma":[0.9912548,0.00378612,0.0003878486,0.00075364916,0.003557419,0.0002601749],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0042066597,0.0012776819,0.0009122351,0.0022005853,0.0007917873,0.0013010405,0.0009179359,0.0009786674,0.0029479212],"category_scores_gemma":[0.01775753,0.00021249251,0.0005629905,0.0019612461,0.0006783482,0.0015643411,0.0011242764,0.0005327709,0.0016608135],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.011042663,0.0029999972,0.031498905,0.005825513,0.000981708,0.0029009983,0.0044745826,0.09619575,0.077478215,0.0032493044,0.044596806,0.7187556],"study_design_scores_gemma":[0.0012710438,0.00531714,0.08972216,0.00028471835,0.00058405515,0.001840778,0.0029788187,0.7636133,0.10625617,0.002905813,0.024910996,0.00031507862],"about_ca_topic_score_codex":0.0116732335,"about_ca_topic_score_gemma":0.012526332,"teacher_disagreement_score":0.0116732335,"about_ca_system_score_codex":0.0010482606,"about_ca_system_score_gemma":0.00091060903,"threshold_uncertainty_score":0.023210585},"labels":[],"label_agreement":null},{"id":"W3023780644","doi":"10.3389/fpsyg.2020.00860","title":"Corrigendum: Eliciting ERP Components for Morphosyntactic Agreement Mismatches in Perfectly Grammatical Sentences","year":2020,"lang":"en","type":"erratum","venue":"Frontiers in Psychology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; McGill University; Centre for Research on Brain Language and Music","funders":"","keywords":"Psychology; Agreement; Linguistics; Natural language processing; Syntax; Artificial intelligence; Cognitive psychology; Computer science","score_opus":0.03807109998218798,"score_gpt":0.3146890615411368,"score_spread":0.2766179615589488,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3023780644","genre_codex":"editorial","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0009092682,0.0006736624,0.005463602,0.04696096,0.9264658,0.00007476641,0.0047986656,0.0042697317,0.010383466],"genre_scores_gemma":[0.083118975,0.004952394,0.026011746,0.08089675,0.16931227,0.0008589326,0.024444096,0.016124163,0.5942806],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99596345,0.00049851835,0.0007946981,0.00076914014,0.0017009553,0.00027330974],"domain_scores_gemma":[0.95603013,0.010124442,0.0012480562,0.004379137,0.027345845,0.00087241404],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030556847,0.0022179098,0.001947074,0.0025016253,0.0034517713,0.0032025534,0.0032076947,0.005402776,0.1579546],"category_scores_gemma":[0.057447575,0.0011046191,0.0016466469,0.002246797,0.0019934932,0.0023619563,0.0024847584,0.004905268,0.07352575],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000036563168,0.000010645533,0.00006923128,0.00007859633,0.000009158707,0.00023961738,0.000036142435,0.00003640076,0.00016438613,0.0006835399,0.99124825,0.007387471],"study_design_scores_gemma":[0.00009601618,0.00009017116,0.004013656,0.00037340814,0.000103715334,0.0027607179,0.00026984036,0.0012462896,0.004849929,0.005372433,0.98066574,0.00015807808],"about_ca_topic_score_codex":0.018013403,"about_ca_topic_score_gemma":0.01885027,"teacher_disagreement_score":0.1579546,"about_ca_system_score_codex":0.004278552,"about_ca_system_score_gemma":0.0025528336,"threshold_uncertainty_score":0.5284108},"labels":[],"label_agreement":null},{"id":"W3023872708","doi":"10.13025/htat-gw66","title":"Challenges of word sense alignment: Portuguese language resources","year":2020,"lang":"en","type":"article","venue":"Arrow@dit (Dublin Institute of Technology)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Linguistic Association","funders":"Fundação para a Ciência e a Tecnologia; Ministério da Ciência, Tecnologia e Ensino Superior; Universidade Nova de Lisboa; European Commission","keywords":"Portuguese; Political science; European union; Brazilian Portuguese; Word (group theory); Library science; Linguistics; Sociology; Computer science; Business; Philosophy","score_opus":0.022414594597802075,"score_gpt":0.2619537631504967,"score_spread":0.2395391685526946,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3023872708","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06596568,0.010936306,0.8564039,0.02386762,0.0012811793,0.0005830169,0.0036204709,0.0050560865,0.03228577],"genre_scores_gemma":[0.3537089,0.007426936,0.61407864,0.002607346,0.000515388,0.00071237446,0.009043103,0.002846874,0.009060432],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.977783,0.012169058,0.0021348197,0.0029489887,0.0042576995,0.0007063655],"domain_scores_gemma":[0.96697146,0.019531382,0.0024155132,0.005598701,0.004652069,0.00083092443],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.015794462,0.0010005353,0.0015849333,0.0049320003,0.0040934957,0.009288014,0.002572255,0.0018330023,0.0028961364],"category_scores_gemma":[0.043624986,0.0008365151,0.0007450979,0.0075138593,0.0026467824,0.012229381,0.0063953525,0.0023135145,0.0030830018],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035340618,0.0002412631,0.00403263,0.003971424,0.00015372733,0.0022512793,0.019819887,0.007953016,0.021806326,0.14891843,0.037378393,0.7531202],"study_design_scores_gemma":[0.0000685336,0.00014351585,0.004911122,0.0026475075,0.000121913865,0.0030350613,0.04118828,0.044502303,0.03372434,0.30086765,0.5685381,0.00025172497],"about_ca_topic_score_codex":0.0056715026,"about_ca_topic_score_gemma":0.0060576485,"teacher_disagreement_score":0.015794462,"about_ca_system_score_codex":0.0019343707,"about_ca_system_score_gemma":0.0063710944,"threshold_uncertainty_score":0.08353007},"labels":[],"label_agreement":null},{"id":"W3024973587","doi":"10.5539/elt.v13n6p58","title":"An Empirical Study of Chinese EFL Learners’ Understanding and Translation of Expressions of Multiplication Entailing “Times”","year":2020,"lang":"en","type":"article","venue":"English Language Teaching","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"English grammar; Linguistics; Grammar; Philosophy","score_opus":0.03686835495413723,"score_gpt":0.34241457030538025,"score_spread":0.30554621535124304,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3024973587","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99922395,0.000018222156,0.00004507459,0.000016904924,0.0000012195305,0.000014681006,0.000008617721,0.0000010111758,0.0006702517],"genre_scores_gemma":[0.9981177,0.000092274335,0.0002590367,0.00004035552,0.0000024705482,0.000048554954,0.000040337167,0.0000032349215,0.0013960889],"study_design_codex":"qualitative","study_design_gemma":"observational","domain_scores_codex":[0.99818355,0.00066476193,0.00018504525,0.00030097083,0.00042303832,0.00024268085],"domain_scores_gemma":[0.98838973,0.0066599934,0.0017819225,0.00063003553,0.0019612636,0.0005770137],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037333162,0.00080778566,0.0006108612,0.0012394488,0.0019718055,0.0020017389,0.00082715886,0.0009412272,0.0036276921],"category_scores_gemma":[0.014355896,0.0004153841,0.0003439296,0.0016580289,0.0027299125,0.0019739133,0.0015936513,0.001256466,0.000568931],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00026559999,0.0007100623,0.123069175,0.00030191464,0.000024248222,0.0012945255,0.8429068,0.00015312784,0.010022297,0.00056571106,0.00032211043,0.020364542],"study_design_scores_gemma":[0.000058626498,0.0015896551,0.24708752,0.00020121777,0.00006507296,0.0008983509,0.7332096,0.0014633007,0.008944343,0.0005275147,0.0058452515,0.000109500004],"about_ca_topic_score_codex":0.015313169,"about_ca_topic_score_gemma":0.015446516,"teacher_disagreement_score":0.015313169,"about_ca_system_score_codex":0.0015996096,"about_ca_system_score_gemma":0.0024763201,"threshold_uncertainty_score":0.03044808},"labels":[],"label_agreement":null},{"id":"W3025024122","doi":"10.46430/phfr0003","title":"Introduction à la stylométrie en Python","year":2019,"lang":"fr","type":"article","venue":"The Programming Historian en français","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Humanities; Philosophy","score_opus":0.006513480275634305,"score_gpt":0.23525616926980628,"score_spread":0.22874268899417197,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3025024122","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0010188359,0.0002486528,0.9281179,0.00046656202,0.00017603074,0.000093130606,0.0021247044,0.05843524,0.009318982],"genre_scores_gemma":[0.023696644,0.0006919145,0.9263611,0.0005841786,0.00023853449,0.00040994972,0.0033947283,0.021035504,0.0235874],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9971583,0.0005576307,0.00030474475,0.0006470949,0.0011716966,0.00016056055],"domain_scores_gemma":[0.996545,0.001521602,0.00018261845,0.00084180996,0.00075607107,0.0001529303],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019092935,0.0012438229,0.0009465886,0.0017657594,0.0010082038,0.0038975466,0.0021574115,0.0010122085,0.083267935],"category_scores_gemma":[0.008822318,0.0014768274,0.0019841902,0.0017315082,0.001352259,0.003558001,0.0029934521,0.004323752,0.042188194],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00030107633,0.00013878937,0.0023043063,0.0013204475,0.0001483544,0.00037184067,0.0012416979,0.009843157,0.013631935,0.13566434,0.14613897,0.6888951],"study_design_scores_gemma":[0.00007028306,0.00005224468,0.0017546823,0.00033818593,0.0000440612,0.0007255438,0.00017224306,0.053769264,0.02320171,0.11541886,0.8043346,0.00011841739],"about_ca_topic_score_codex":0.0027871418,"about_ca_topic_score_gemma":0.0030632813,"teacher_disagreement_score":0.083267935,"about_ca_system_score_codex":0.0009162063,"about_ca_system_score_gemma":0.0024296679,"threshold_uncertainty_score":0.27855897},"labels":[],"label_agreement":null},{"id":"W3025183185","doi":"","title":"Rapid automatized naming and reading: a review","year":2013,"lang":"en","type":"review","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reading (process); Computer science; Linguistics; Natural language processing; Psychology; Philosophy","score_opus":0.04463285989429434,"score_gpt":0.34701682454262484,"score_spread":0.3023839646483305,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3025183185","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00028924795,0.9948602,0.0025141467,0.00025620477,0.00034270418,0.00003704058,0.00008429075,0.00011397198,0.0015023206],"genre_scores_gemma":[0.0014833316,0.9904593,0.005762366,0.00037407494,0.00037745983,0.000059534843,0.00025343883,0.00003847511,0.0011920887],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9993585,0.0001150421,0.000111100766,0.0001783201,0.00018766009,0.000049422542],"domain_scores_gemma":[0.99737644,0.0015828942,0.0002200715,0.0001473692,0.00060090085,0.00007237565],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017788952,0.001692711,0.0025044135,0.0033629746,0.00039674932,0.0014472998,0.0034437927,0.0017509334,0.007138121],"category_scores_gemma":[0.0054908497,0.0006358179,0.0010763992,0.0028868716,0.0011972323,0.0030381528,0.0011794048,0.0014883674,0.0062547093],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000082341714,0.00009428072,0.00017890467,0.011308601,0.00007502598,0.00008121029,0.00003489112,0.00014013477,0.00089759496,0.0008037442,0.016711483,0.96959186],"study_design_scores_gemma":[0.00011671453,0.00031850414,0.003287656,0.009792222,0.0009210406,0.0032939974,0.00014902171,0.0005438167,0.0060111764,0.0071078115,0.9683288,0.0001292511],"about_ca_topic_score_codex":0.0032894602,"about_ca_topic_score_gemma":0.0046044705,"teacher_disagreement_score":0.007138121,"about_ca_system_score_codex":0.00066796,"about_ca_system_score_gemma":0.0032412352,"threshold_uncertainty_score":0.023879409},"labels":[],"label_agreement":null},{"id":"W3026572343","doi":"10.7557/1.9.1.5277","title":"Restrictions on ordering of adjectives in Spanish","year":2020,"lang":"en","type":"article","venue":"Borealis – An International Journal of Hispanic Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Adjective; Noun; Linguistics; Interpretation (philosophy); Contrast (vision); Set (abstract data type); Part of speech; Noun phrase; Computer science; Natural language processing; Artificial intelligence; Philosophy","score_opus":0.0271508637144564,"score_gpt":0.31107080426115546,"score_spread":0.28391994054669906,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3026572343","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9561345,0.0008811737,0.0062317206,0.0003367044,0.000047033434,0.00007333607,0.008151209,0.00025797563,0.027886264],"genre_scores_gemma":[0.98005414,0.00027519,0.003443044,0.00011612707,0.000019735717,0.00007892758,0.013727893,0.00017981618,0.002105072],"study_design_codex":"observational","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9969374,0.0010851626,0.0003450678,0.0007513878,0.0006666164,0.00021429668],"domain_scores_gemma":[0.98332846,0.007892804,0.0016719634,0.0026002584,0.004166721,0.00033982008],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002467382,0.00036887068,0.0005239964,0.0018305131,0.0013546397,0.0022096552,0.00049045845,0.00048599503,0.0038781494],"category_scores_gemma":[0.015797716,0.00029389272,0.00036077315,0.0029880975,0.0015057499,0.0018781939,0.0013824543,0.0009164857,0.0011887539],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0043761516,0.0005253568,0.5549629,0.0026455536,0.0003121886,0.001612522,0.035887193,0.004217325,0.08274191,0.047419168,0.018194309,0.24710542],"study_design_scores_gemma":[0.0002147568,0.0003054581,0.6737328,0.0004677961,0.00013985997,0.0015946279,0.025970463,0.009683901,0.018659191,0.037506294,0.23151986,0.00020494816],"about_ca_topic_score_codex":0.032922253,"about_ca_topic_score_gemma":0.03917132,"teacher_disagreement_score":0.032922253,"about_ca_system_score_codex":0.0017078586,"about_ca_system_score_gemma":0.0013882159,"threshold_uncertainty_score":0.06546128},"labels":[],"label_agreement":null},{"id":"W3028807070","doi":"","title":"Extraction of Hyponymic Relations in French with Knowledge-Pattern-Based Word Sketches.","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":35,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec","funders":"","keywords":"Computer science; Sketch; Natural language processing; Artificial intelligence; Word (group theory); Grammar; Domain (mathematical analysis); Thesaurus; Process (computing); Information extraction; Information retrieval; Linguistics","score_opus":0.021142584555562678,"score_gpt":0.29778985129183766,"score_spread":0.276647266736275,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3028807070","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5018258,0.0060219257,0.38191748,0.0010832351,0.00033221475,0.0012669562,0.054479968,0.029933337,0.023139047],"genre_scores_gemma":[0.6395427,0.0014269805,0.29002005,0.00013513844,0.00008488107,0.0004024233,0.06036721,0.0009240129,0.0070966356],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99915814,0.00024491342,0.00010480086,0.0002825802,0.00015322151,0.00005640501],"domain_scores_gemma":[0.9971107,0.0017942967,0.00018238123,0.0002181132,0.00060175225,0.000092837625],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00069470255,0.000888418,0.0005514081,0.005005981,0.0006419041,0.0016604806,0.0006493345,0.00077441445,0.008609612],"category_scores_gemma":[0.005581913,0.00027499738,0.00076620036,0.0022498458,0.00038437283,0.0027433606,0.0010277281,0.00064371014,0.003558904],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009723075,0.0002692672,0.016392764,0.003049782,0.00025781724,0.0016923952,0.0031332045,0.005318453,0.0839956,0.014972135,0.025103947,0.84484226],"study_design_scores_gemma":[0.00062926527,0.0012117832,0.102196455,0.0010513104,0.0011961006,0.0066041816,0.012504187,0.31684932,0.1583902,0.045153033,0.35389614,0.0003180351],"about_ca_topic_score_codex":0.016626177,"about_ca_topic_score_gemma":0.018622916,"teacher_disagreement_score":0.016626177,"about_ca_system_score_codex":0.00063694536,"about_ca_system_score_gemma":0.0013696405,"threshold_uncertainty_score":0.033058822},"labels":[],"label_agreement":null},{"id":"W3029096252","doi":"","title":"A Lexicon-Based Approach for Detecting Hedges in Informal Text","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Computer science; Hedge; Lexicon; Natural language processing; Sentence; Interview; Artificial intelligence; Part-of-speech tagging; Linguistics; Part of speech; Sociology","score_opus":0.03137950489801756,"score_gpt":0.300703666162043,"score_spread":0.2693241612640255,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3029096252","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11037575,0.001223971,0.8496416,0.0006671164,0.00016698401,0.00177843,0.0057747,0.0191111,0.011260258],"genre_scores_gemma":[0.41507038,0.0003425179,0.5716984,0.00022316001,0.000085310756,0.00060676195,0.0073906113,0.0006437546,0.0039391285],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9951448,0.001625462,0.00066842936,0.00062347145,0.0016997816,0.00023801648],"domain_scores_gemma":[0.98874485,0.005880194,0.00075614476,0.001021963,0.0031756382,0.00042130108],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029357919,0.001012575,0.001150095,0.010203019,0.0012591209,0.004010528,0.0014451741,0.0016272693,0.0044845175],"category_scores_gemma":[0.013861601,0.0005037769,0.00081378984,0.003971232,0.00088621775,0.0047040638,0.0023530438,0.0011033057,0.0022646992],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00081861916,0.00080502656,0.018933747,0.0010717184,0.00030420133,0.0010508591,0.0018760443,0.0060852016,0.09664157,0.023047073,0.022099229,0.82726675],"study_design_scores_gemma":[0.00036255628,0.0010322506,0.026328977,0.00042107995,0.0007855103,0.0025809149,0.003098581,0.75227845,0.10982754,0.05682352,0.046051793,0.00040883425],"about_ca_topic_score_codex":0.006771499,"about_ca_topic_score_gemma":0.011549169,"teacher_disagreement_score":0.010203019,"about_ca_system_score_codex":0.0011098805,"about_ca_system_score_gemma":0.0023658145,"threshold_uncertainty_score":0.0155261755},"labels":[],"label_agreement":null},{"id":"W3029245064","doi":"","title":"Tense Interpretation in the Context of Narrative.","year":2005,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Narrative; Interpretation (philosophy); Focus (optics); Past tense; Heuristics; Context (archaeology); Computer science; Statement (logic); Present tense; Linguistics; Natural language processing; Artificial intelligence; Epistemology; History; Philosophy; Verb","score_opus":0.009718917500928103,"score_gpt":0.27986369881505857,"score_spread":0.27014478131413044,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3029245064","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009965885,0.0017277885,0.9649491,0.00086791586,0.00023151322,0.00009810068,0.00030101888,0.0012101538,0.020648636],"genre_scores_gemma":[0.3225802,0.00093796314,0.6693104,0.00024998898,0.00015109658,0.00009586311,0.0008147797,0.00038204383,0.0054776217],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99778724,0.0011644966,0.00014696187,0.00052121794,0.00030409737,0.00007590307],"domain_scores_gemma":[0.9966646,0.0022615471,0.00033668763,0.0003683762,0.0002778892,0.000090824025],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021918851,0.00092349906,0.00046726406,0.0014166881,0.0016743323,0.0042583076,0.0012344786,0.0011687424,0.0077505773],"category_scores_gemma":[0.011146084,0.00087680924,0.00089696114,0.00097383786,0.0024277524,0.008667087,0.0021955038,0.0023124074,0.0018391667],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00027269486,0.000039065402,0.0013963993,0.00058690977,0.000066572946,0.0006109762,0.007058975,0.008701737,0.008092302,0.78163505,0.007257833,0.18428153],"study_design_scores_gemma":[0.000036026937,0.00005149922,0.0006116673,0.00023580456,0.000050586044,0.00083702325,0.0018281982,0.067088716,0.00905032,0.8216489,0.098513685,0.000047462672],"about_ca_topic_score_codex":0.001078786,"about_ca_topic_score_gemma":0.0017577836,"teacher_disagreement_score":0.0077505773,"about_ca_system_score_codex":0.0010359116,"about_ca_system_score_gemma":0.0009073352,"threshold_uncertainty_score":0.025928259},"labels":[],"label_agreement":null},{"id":"W3029265130","doi":"","title":"Multilingual Dictionary Based Construction of Core Vocabulary.","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Vocabulary; Natural language processing; Core (optical fiber); Artificial intelligence; Set (abstract data type); Bilingual dictionary; Field (mathematics); Resource (disambiguation); Machine translation; Linguistics; Programming language","score_opus":0.028694671351722077,"score_gpt":0.3027328626012327,"score_spread":0.2740381912495106,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3029265130","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15100893,0.0016002014,0.7510712,0.00055596064,0.0004962291,0.0023552836,0.021600045,0.01650844,0.054803614],"genre_scores_gemma":[0.47190988,0.0006737193,0.46175256,0.00020992028,0.00007120408,0.0012733493,0.047321428,0.0023178635,0.014470148],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.995415,0.0019576054,0.0006137853,0.00081349374,0.0009146866,0.00028546556],"domain_scores_gemma":[0.99194896,0.0024788082,0.00026219522,0.0011297462,0.0038340536,0.00034628587],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026593744,0.0006668834,0.00089886336,0.004443281,0.0011197777,0.0027894143,0.0013507651,0.0006210512,0.014485645],"category_scores_gemma":[0.013567471,0.00042233735,0.00063853327,0.0026290007,0.000666392,0.007674701,0.0046506734,0.0012083593,0.0068703895],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010801076,0.00060021254,0.00880594,0.0020041289,0.00022163775,0.00038884688,0.002459609,0.004986783,0.061078295,0.058575865,0.0412084,0.8185901],"study_design_scores_gemma":[0.0006009321,0.001316769,0.016810883,0.0010609475,0.00084714266,0.0027819118,0.010712795,0.30541906,0.26967573,0.10784843,0.28259328,0.00033204],"about_ca_topic_score_codex":0.007330266,"about_ca_topic_score_gemma":0.012024214,"teacher_disagreement_score":0.014485645,"about_ca_system_score_codex":0.0011711597,"about_ca_system_score_gemma":0.003453089,"threshold_uncertainty_score":0.04845935},"labels":[],"label_agreement":null},{"id":"W3029286378","doi":"10.3929/ethz-b-000462327","title":"UniMorph 3.0: Universal Morphology","year":2020,"lang":"en","type":"article","venue":"Minerva Access (University of Melbourne)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":54,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Unimorph; Computer science; Schema (genetic algorithms); Annotation; Artificial intelligence; Natural language processing; Software engineering; Information retrieval","score_opus":0.030296802502465348,"score_gpt":0.2431710628218441,"score_spread":0.21287426031937873,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3029286378","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008792526,0.0004488189,0.45284474,0.0005522857,0.00020260009,0.0005272536,0.07184018,0.41918272,0.04560899],"genre_scores_gemma":[0.063467376,0.0005771324,0.5557053,0.0007629126,0.00009834618,0.0018049841,0.21276185,0.13815986,0.026662298],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9973597,0.0003247621,0.00033812714,0.00070901733,0.0011003522,0.00016799533],"domain_scores_gemma":[0.9977742,0.000480745,0.00017637483,0.0009212654,0.0005077345,0.00013966109],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032004123,0.0013227862,0.0010766324,0.0038062783,0.0008849372,0.0036297177,0.0042075766,0.0009627486,0.056942254],"category_scores_gemma":[0.0073135193,0.0019437817,0.0015318649,0.0030829168,0.0011523888,0.0068233865,0.0067165187,0.001744002,0.031549525],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00062946713,0.00014115295,0.003696131,0.001895157,0.00014344322,0.00055324606,0.0013437733,0.0027872894,0.019697608,0.06789909,0.5709409,0.33027267],"study_design_scores_gemma":[0.000073762094,0.000062211075,0.0025128883,0.00018441307,0.00003751881,0.001295445,0.00028321834,0.011155379,0.022301495,0.028444745,0.9335526,0.000096299555],"about_ca_topic_score_codex":0.0024131818,"about_ca_topic_score_gemma":0.0032744422,"teacher_disagreement_score":0.056942254,"about_ca_system_score_codex":0.0014380093,"about_ca_system_score_gemma":0.002508282,"threshold_uncertainty_score":0.19049084},"labels":[],"label_agreement":null},{"id":"W3029712483","doi":"","title":"Evaluating the Impact of Sub-word Information and Cross-lingual Word Embeddings on Mi’kmaq Language Modelling","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of New Brunswick","funders":"","keywords":"Word (group theory); Computer science; Language model; Natural language processing; Artificial intelligence; Indigenous language; Linguistics; Indigenous","score_opus":0.04360119139160799,"score_gpt":0.3828779568686983,"score_spread":0.3392767654770903,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3029712483","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.84851855,0.0032778725,0.11771875,0.0014730984,0.0007099759,0.0003396554,0.005752017,0.010426304,0.011783855],"genre_scores_gemma":[0.91871876,0.0005444863,0.06586937,0.00019270017,0.00005477988,0.00013095797,0.011222334,0.0005426561,0.002723952],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99202543,0.0048142183,0.0006949856,0.0012383033,0.0008478443,0.00037917058],"domain_scores_gemma":[0.96874094,0.025128644,0.00043871326,0.0023284839,0.0027479606,0.00061528105],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008722802,0.002061209,0.0010745724,0.00245096,0.00093250175,0.0030636296,0.0019809674,0.0019485293,0.004441273],"category_scores_gemma":[0.035812333,0.000618205,0.0013149782,0.0022337344,0.00083672826,0.0077379285,0.002865872,0.0025266502,0.0024922525],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00650753,0.0022927572,0.03061942,0.0015123787,0.0019766546,0.00039737337,0.0010646958,0.33471388,0.01257837,0.0050132177,0.013896716,0.58942693],"study_design_scores_gemma":[0.00011704185,0.0005223123,0.0040146476,0.00007887805,0.00034482498,0.00011247295,0.00070079503,0.97797555,0.010486633,0.0030231613,0.0025499694,0.000073761636],"about_ca_topic_score_codex":0.03672968,"about_ca_topic_score_gemma":0.03301183,"teacher_disagreement_score":0.03672968,"about_ca_system_score_codex":0.0014181222,"about_ca_system_score_gemma":0.0020245055,"threshold_uncertainty_score":0.07303178},"labels":[],"label_agreement":null},{"id":"W3029889931","doi":"","title":"The Johns Hopkins University Bible Corpus: 1600+ Tongues for Typological Exploration","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":48,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Natural language processing; Variety (cybernetics); Representation (politics); Linguistics; Corpus linguistics; Parallel corpora; Pronoun; Artificial intelligence; Information retrieval; History; Machine translation; Philosophy","score_opus":0.04572153790965464,"score_gpt":0.29345114168665953,"score_spread":0.2477296037770049,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3029889931","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20184046,0.0035190403,0.0070909476,0.0016580212,0.0012886835,0.00056106,0.6098959,0.001563343,0.17258255],"genre_scores_gemma":[0.2664446,0.0020319698,0.018978352,0.0005498966,0.00056886167,0.0014424484,0.64400786,0.0017550401,0.064220935],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.998987,0.000280591,0.00015222114,0.00013518405,0.0003599264,0.00008511085],"domain_scores_gemma":[0.9962239,0.0012925752,0.0002337776,0.00065624976,0.0012055004,0.00038785482],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010440353,0.00050288433,0.00047750966,0.009435523,0.0024899822,0.0018894847,0.00072999985,0.00062356837,0.053078637],"category_scores_gemma":[0.006478014,0.00028087696,0.00016295395,0.010030854,0.0011041999,0.0011863298,0.0023655738,0.00082779356,0.025947822],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00063549454,0.00020049044,0.0079096,0.002174381,0.00003442351,0.0010203221,0.011500407,0.00048402665,0.013147167,0.01575643,0.75519085,0.19194643],"study_design_scores_gemma":[0.00010772734,0.000057691548,0.052345827,0.00051518134,0.00003638779,0.00066208845,0.0057077613,0.0007425279,0.005066743,0.001955146,0.9327561,0.00004677055],"about_ca_topic_score_codex":0.014629666,"about_ca_topic_score_gemma":0.030219562,"teacher_disagreement_score":0.053078637,"about_ca_system_score_codex":0.0009811205,"about_ca_system_score_gemma":0.0026501743,"threshold_uncertainty_score":0.17756575},"labels":[],"label_agreement":null},{"id":"W3029985558","doi":"","title":"The Nunavut Hansard Inuktitut-English Parallel Corpus 3.0 with Preliminary Machine Translation Results.","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Sentence; Machine translation; Indigenous; Natural language processing; Indigenous language; Linguistics; Artificial intelligence; Speech recognition","score_opus":0.021495609075024055,"score_gpt":0.266804596939113,"score_spread":0.24530898786408895,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3029985558","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.23886031,0.01079709,0.045088362,0.0033896961,0.0029717367,0.0037742986,0.48888126,0.026990797,0.17924646],"genre_scores_gemma":[0.19331539,0.0015148235,0.08629465,0.0005749202,0.00026042687,0.0028971455,0.6639236,0.006602052,0.04461699],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9966558,0.0015751676,0.00026034165,0.0006720349,0.00054772926,0.00028887024],"domain_scores_gemma":[0.995404,0.0012170307,0.00013294669,0.00076404336,0.0020804713,0.0004015454],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032643853,0.0014922242,0.0013098557,0.004440198,0.0039448505,0.0026185962,0.0018982975,0.0010983951,0.04886709],"category_scores_gemma":[0.010710484,0.0007707741,0.0005438368,0.0047834236,0.0009296555,0.002248779,0.0040594917,0.0013280001,0.027044136],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0044780304,0.0011471915,0.0065202015,0.0051946435,0.00041680204,0.0024659205,0.005251723,0.004453098,0.035370547,0.014673732,0.6433319,0.2766962],"study_design_scores_gemma":[0.0018201751,0.00060397125,0.03017794,0.0010080598,0.0005303608,0.0022773647,0.0041374443,0.012494155,0.045275025,0.0064273127,0.89499134,0.0002567512],"about_ca_topic_score_codex":0.072437026,"about_ca_topic_score_gemma":0.0986783,"teacher_disagreement_score":0.072437026,"about_ca_system_score_codex":0.0017997066,"about_ca_system_score_gemma":0.0063329083,"threshold_uncertainty_score":0.16347665},"labels":[],"label_agreement":null},{"id":"W3030307641","doi":"","title":"An Analysis of Massively Multilingual Neural Machine Translation for Low-Resource Languages.","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Machine translation; Natural language processing; Artificial intelligence; Set (abstract data type); Resource (disambiguation); Massively parallel; Programming language","score_opus":0.028529584331923822,"score_gpt":0.33911117094096893,"score_spread":0.3105815866090451,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3030307641","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.812707,0.006692375,0.13668266,0.0024830827,0.00039279927,0.00032965676,0.0062006484,0.008270397,0.026241386],"genre_scores_gemma":[0.94799876,0.00049998175,0.040068656,0.00016683621,0.000095505755,0.00013201268,0.006889428,0.00042215118,0.0037265907],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9980459,0.00097258633,0.0001147329,0.00020766769,0.00049798784,0.00016104638],"domain_scores_gemma":[0.99081296,0.006304179,0.00026977193,0.000719206,0.0017598582,0.0001340282],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032445306,0.0006848348,0.00062654103,0.0014891231,0.00084761705,0.0012757229,0.0010275381,0.0007936074,0.0038095831],"category_scores_gemma":[0.014657015,0.00026044133,0.0004849064,0.0017564453,0.00050133216,0.0020461583,0.00082488917,0.0007674697,0.0011915432],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0047535747,0.001309321,0.022440057,0.0021672866,0.0009268579,0.0020070665,0.00064731494,0.3705645,0.03284,0.03933513,0.055981405,0.46702746],"study_design_scores_gemma":[0.000052623327,0.00024248876,0.006318212,0.00003834103,0.000106382475,0.00024779123,0.00017897367,0.9655091,0.00983749,0.013507703,0.003937971,0.000022932496],"about_ca_topic_score_codex":0.008634785,"about_ca_topic_score_gemma":0.013641089,"teacher_disagreement_score":0.008634785,"about_ca_system_score_codex":0.0012828943,"about_ca_system_score_gemma":0.0012143487,"threshold_uncertainty_score":0.017169058},"labels":[],"label_agreement":null},{"id":"W3030379882","doi":"","title":"On the Creation of a Corpus for Coherence Evaluation of Discursive Units","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University; Bank of Canada; National Bank of Canada","funders":"","keywords":"Computer science; Coherence (philosophical gambling strategy); Natural language processing; Artificial intelligence; Sentence; Argument (complex analysis); Focus (optics); Classifier (UML); Linguistics; Textual entailment; Logical consequence; Mathematics","score_opus":0.05399198577407965,"score_gpt":0.33059146662545824,"score_spread":0.2765994808513786,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3030379882","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20092176,0.0009618101,0.7403103,0.0020966234,0.00040937433,0.0038293994,0.00969297,0.007953777,0.03382402],"genre_scores_gemma":[0.2721308,0.0003204328,0.7033212,0.0002181776,0.00011998366,0.002702215,0.013053351,0.001549172,0.006584631],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9863339,0.008650459,0.0011307396,0.0011134103,0.002434774,0.00033674],"domain_scores_gemma":[0.9444702,0.033751573,0.0014920282,0.005883655,0.013068661,0.0013338375],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012335106,0.00076419435,0.000889957,0.0064268326,0.0030776747,0.003905706,0.0022068669,0.0016343232,0.0094846105],"category_scores_gemma":[0.047721807,0.00077330973,0.00046984767,0.0036237217,0.0023958622,0.0069885342,0.005346199,0.0019935889,0.0036392775],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008691405,0.0013138934,0.00950704,0.0014274669,0.000120015924,0.00077853695,0.011883001,0.0114542395,0.0596853,0.07109675,0.05890156,0.772963],"study_design_scores_gemma":[0.0007250828,0.0013874253,0.031472094,0.0010283954,0.00033756468,0.0021642186,0.014517463,0.4315942,0.21971002,0.071592495,0.22490448,0.00056656444],"about_ca_topic_score_codex":0.008010175,"about_ca_topic_score_gemma":0.012790616,"teacher_disagreement_score":0.012335106,"about_ca_system_score_codex":0.0015974978,"about_ca_system_score_gemma":0.003278793,"threshold_uncertainty_score":0.06523508},"labels":[],"label_agreement":null},{"id":"W3030650355","doi":"","title":"DiMLex-Bangla: A Lexicon of Bangla Discourse Connectives.","year":2020,"lang":"en","type":"article","venue":"Åbo Akademi University Research Portal","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Bengali; Lexicon; Computer science; Treebank; Natural language processing; Artificial intelligence; Linguistics; German; Annotation","score_opus":0.057589311312586586,"score_gpt":0.3440530868068982,"score_spread":0.2864637754943116,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3030650355","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05745279,0.006360099,0.264126,0.002144336,0.0017582793,0.0017593753,0.36158583,0.07349089,0.23132242],"genre_scores_gemma":[0.32555676,0.0023106462,0.2942252,0.0010716775,0.00024474476,0.001712268,0.31227052,0.009171315,0.053436946],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9996044,0.00009216349,0.000099955454,0.00012639315,0.00005541612,0.000021719068],"domain_scores_gemma":[0.9992716,0.00030839664,0.00007880326,0.00009480745,0.00017397081,0.0000724999],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004111426,0.00077085564,0.00064311625,0.0023537446,0.0010317508,0.003318557,0.00075255707,0.0005064621,0.031080356],"category_scores_gemma":[0.0015178127,0.0007775668,0.00041095336,0.0021528413,0.00052438665,0.0028387483,0.0017785279,0.0007744362,0.021390019],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010198895,0.00024929966,0.0078062066,0.0064248214,0.00016503145,0.0020827393,0.0059504127,0.001443177,0.04746311,0.11368741,0.4257367,0.38797116],"study_design_scores_gemma":[0.000043346823,0.000033207216,0.0024096596,0.00016085191,0.000029402294,0.0010023478,0.00071676896,0.0020405548,0.005576732,0.0057899677,0.98216045,0.000036674686],"about_ca_topic_score_codex":0.0018831466,"about_ca_topic_score_gemma":0.00398773,"teacher_disagreement_score":0.031080356,"about_ca_system_score_codex":0.0010997363,"about_ca_system_score_gemma":0.0015746774,"threshold_uncertainty_score":0.10397416},"labels":[],"label_agreement":null},{"id":"W3031229455","doi":"","title":"SEDAR: a Large Scale French-English Financial Domain Parallel Corpus","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Machine translation; Computer science; Domain (mathematical analysis); Preprocessor; Natural language processing; Sentence; Translation (biology); Artificial intelligence; Scale (ratio); Speech recognition; Chemistry","score_opus":0.011982349944706733,"score_gpt":0.26276082289982794,"score_spread":0.25077847295512123,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3031229455","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.35992762,0.004966531,0.053512886,0.0040016077,0.0013086969,0.0013637644,0.5016189,0.019606413,0.05369358],"genre_scores_gemma":[0.27266982,0.0011434921,0.061005443,0.0008267665,0.0003155982,0.0012218576,0.6451708,0.0020670972,0.015579072],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9985623,0.00055246404,0.00013152753,0.00033256054,0.0002925288,0.0001287136],"domain_scores_gemma":[0.99505156,0.0021218685,0.00017363389,0.0005859228,0.0017391042,0.000327863],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017739303,0.0012410708,0.0007534519,0.00438736,0.0021277901,0.0016771251,0.0014399004,0.0014330586,0.022219501],"category_scores_gemma":[0.0070475866,0.0004436807,0.0006326935,0.0031112318,0.001014873,0.001985593,0.0018569325,0.0013230087,0.009744931],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0024387655,0.0015369855,0.013759375,0.0045031663,0.00059857377,0.00497691,0.0031154943,0.008787898,0.05996235,0.021161687,0.60642755,0.27273127],"study_design_scores_gemma":[0.0012913563,0.0005878602,0.070052415,0.00045189916,0.00053775986,0.0044065216,0.00410619,0.03391825,0.033356875,0.008594587,0.84239507,0.00030119286],"about_ca_topic_score_codex":0.04281013,"about_ca_topic_score_gemma":0.044907823,"teacher_disagreement_score":0.04281013,"about_ca_system_score_codex":0.0014551517,"about_ca_system_score_gemma":0.0034204018,"threshold_uncertainty_score":0.08512193},"labels":[],"label_agreement":null},{"id":"W3031973162","doi":"","title":"NLP Scholar: A Dataset for Examining the State of NLP Research.","year":2020,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Metadata; Artificial intelligence; Information retrieval; Natural language processing; Citation; Focus (optics); Named entity; World Wide Web","score_opus":0.12193293461524737,"score_gpt":0.37562701290002976,"score_spread":0.2536940782847824,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3031973162","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0032127623,0.001416893,0.00086586055,0.0007096549,0.00014998435,0.00010343084,0.9853247,0.00096158247,0.0072550736],"genre_scores_gemma":[0.0030896214,0.00061465794,0.0029731009,0.00014472273,0.00006232407,0.00018246872,0.9914403,0.00012371852,0.0013689606],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99351853,0.0010409585,0.0016443228,0.0009558341,0.002420264,0.00042005375],"domain_scores_gemma":[0.97913796,0.0089787515,0.0035262713,0.0024609899,0.004452509,0.0014435411],"candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.0032870611,0.0013086468,0.001202925,0.030032061,0.0020140777,0.004114831,0.0023520219,0.0024885705,0.014449298],"category_scores_gemma":[0.023651721,0.0004757541,0.0012185844,0.03896389,0.0007311793,0.0035566604,0.0034364427,0.0016780923,0.01820155],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001894471,0.00011271934,0.007987071,0.005166865,0.00012000056,0.00045042823,0.0006480373,0.0006633102,0.0018173676,0.008138498,0.9508665,0.023839848],"study_design_scores_gemma":[0.00011194089,0.000028043625,0.00945285,0.00037202824,0.000040402316,0.0002693106,0.00051835855,0.0009601485,0.001134818,0.0028887577,0.98418427,0.000039084567],"about_ca_topic_score_codex":0.018458918,"about_ca_topic_score_gemma":0.041911084,"teacher_disagreement_score":0.9967129,"about_ca_system_score_codex":0.002924765,"about_ca_system_score_gemma":0.00523096,"threshold_uncertainty_score":0.048337698},"labels":[],"label_agreement":null},{"id":"W3032201847","doi":"","title":"SpiCE: A New Open-Access Corpus of Conversational Bilingual Speech in Cantonese and English.","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Transcription (linguistics); Spice; Sentence; Annotation; Natural language processing; Speech corpus; Storyboard; Phonetic transcription; Speech recognition; Linguistics; Artificial intelligence; Speech synthesis; Multimedia; Engineering","score_opus":0.04815606913529093,"score_gpt":0.3652228133420278,"score_spread":0.31706674420673686,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3032201847","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.40316015,0.0030706744,0.015206153,0.0011278667,0.0005689238,0.0016713955,0.53317827,0.0050600506,0.036956467],"genre_scores_gemma":[0.33814454,0.0006790278,0.01744675,0.00025583402,0.00015976129,0.0026527701,0.62764704,0.00082365406,0.012190576],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99856573,0.00044234341,0.00018234558,0.00032609949,0.00032100308,0.00016244932],"domain_scores_gemma":[0.9952591,0.0015402322,0.0002556547,0.0006179137,0.0017743856,0.00055265776],"candidate_categories":["open_science"],"consensus_categories":[],"category_scores_codex":[0.0014105969,0.0011014863,0.0007409334,0.003325559,0.0017928317,0.0014021785,0.0013202407,0.00087692705,0.01745474],"category_scores_gemma":[0.006742451,0.00034594527,0.00029785704,0.0027798554,0.0008624537,0.0016210782,0.0029110142,0.0010025463,0.004922333],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.003967754,0.0014017834,0.04658138,0.007127294,0.00040796265,0.0042429096,0.019991465,0.00279955,0.12522815,0.008758069,0.43091056,0.34858322],"study_design_scores_gemma":[0.00090133515,0.0005803974,0.36788765,0.0011038587,0.00040405127,0.0036140655,0.01548167,0.009502767,0.027733969,0.0031942884,0.5692571,0.00033883445],"about_ca_topic_score_codex":0.06782387,"about_ca_topic_score_gemma":0.102579564,"teacher_disagreement_score":0.99867976,"about_ca_system_score_codex":0.0011679083,"about_ca_system_score_gemma":0.003678437,"threshold_uncertainty_score":0.13485819},"labels":[],"label_agreement":null},{"id":"W3032216439","doi":"","title":"Fine-grained Morphosyntactic Analysis and Generation Tools for More Than One Thousand Languages.","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Natural language processing; Security token; Metric (unit); Artificial intelligence; Machine translation; Parallel corpora; Linguistics; Engineering","score_opus":0.04463166772679519,"score_gpt":0.3218636676018755,"score_spread":0.2772319998750803,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3032216439","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.18066983,0.00363957,0.55680275,0.0012183308,0.0006891248,0.0014023639,0.044170093,0.18337521,0.028032757],"genre_scores_gemma":[0.3336333,0.0007437355,0.54982096,0.00040786783,0.00009308986,0.0010968086,0.09082494,0.009916182,0.013463074],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9967332,0.0010008742,0.00040569223,0.0006314017,0.0010218875,0.00020709017],"domain_scores_gemma":[0.9929865,0.0029451572,0.00024434744,0.002053443,0.00144296,0.00032762656],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0039643203,0.001301012,0.00075440604,0.002481219,0.0008222492,0.0017491116,0.001954132,0.0010906454,0.013137224],"category_scores_gemma":[0.010400278,0.0007572334,0.001064951,0.0022704222,0.00069203135,0.004430672,0.0030985773,0.0013901739,0.0068687205],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015034997,0.0007974874,0.0074149254,0.001710637,0.00043932433,0.00067120406,0.0016557008,0.011785039,0.07274408,0.018870786,0.11556646,0.7668409],"study_design_scores_gemma":[0.0012694267,0.0013775029,0.023802804,0.0006670393,0.00071872596,0.0016302576,0.0021072216,0.24682459,0.2704061,0.06490397,0.38587004,0.00042233954],"about_ca_topic_score_codex":0.005240927,"about_ca_topic_score_gemma":0.006514921,"teacher_disagreement_score":0.013137224,"about_ca_system_score_codex":0.00090172986,"about_ca_system_score_gemma":0.0018542582,"threshold_uncertainty_score":0.043948412},"labels":[],"label_agreement":null},{"id":"W3032491810","doi":"","title":"GM-RKB WikiText Error Correction Task and Baselines.","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Error detection and correction; Artificial intelligence; Natural language processing; Task (project management); Language model; Ground truth; Speech recognition; Information retrieval; Machine learning; Algorithm","score_opus":0.021723023275288137,"score_gpt":0.2949398695453348,"score_spread":0.27321684627004666,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3032491810","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.28748575,0.0072319712,0.08066499,0.0037463824,0.007037636,0.006965436,0.29613188,0.212249,0.09848691],"genre_scores_gemma":[0.25482303,0.0008243613,0.11357129,0.0021436512,0.0005430844,0.006159883,0.56284237,0.0129292095,0.046163145],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9870745,0.004811199,0.0017039669,0.002943043,0.0027793534,0.000688035],"domain_scores_gemma":[0.96526104,0.013067847,0.0014044415,0.010214271,0.008017174,0.002035249],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010597548,0.0035350733,0.0019420682,0.0033187147,0.0021810762,0.0036866704,0.0046061273,0.004208244,0.02160893],"category_scores_gemma":[0.053720277,0.0009634225,0.001313136,0.0027590797,0.001208548,0.005665122,0.0072748796,0.00477811,0.033268522],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0052235345,0.0038126973,0.0076753683,0.004758002,0.0007832855,0.000609575,0.0009531707,0.0057441215,0.025251532,0.0020099734,0.6445594,0.2986194],"study_design_scores_gemma":[0.007347206,0.004299547,0.09471348,0.0019245917,0.0019675263,0.004745079,0.003159251,0.1546687,0.16746335,0.013714513,0.5449161,0.001080669],"about_ca_topic_score_codex":0.013923359,"about_ca_topic_score_gemma":0.017720465,"teacher_disagreement_score":0.02160893,"about_ca_system_score_codex":0.0011118863,"about_ca_system_score_gemma":0.0034115305,"threshold_uncertainty_score":0.07228905},"labels":[],"label_agreement":null},{"id":"W3032645833","doi":"10.18653/v1/2020.sigmorphon-1.3","title":"The SIGMORPHON 2020 Shared Task on Unsupervised Morphological Paradigm Completion","year":2020,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Task (project management); Pipeline (software); Natural language processing; Surprise; Baseline (sea); Artificial intelligence; Modular design; Field (mathematics); Lemma (botany); Programming language; Psychology; Mathematics; Communication","score_opus":0.0371723396303211,"score_gpt":0.2820476178599801,"score_spread":0.24487527822965902,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3032645833","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.24403381,0.008138536,0.2798647,0.0092943795,0.011166759,0.0088126315,0.15458316,0.19549374,0.0886123],"genre_scores_gemma":[0.2283407,0.00068886176,0.314188,0.0031348038,0.0012929944,0.0076774247,0.38059697,0.01771487,0.046365455],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9795221,0.008508424,0.001257379,0.004722453,0.0040786583,0.0019109508],"domain_scores_gemma":[0.962206,0.012247503,0.0011686337,0.012395922,0.0076048323,0.00437712],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.020237021,0.0063097314,0.0031273968,0.0029765605,0.003132104,0.004533912,0.0056779278,0.006001123,0.027552871],"category_scores_gemma":[0.044299673,0.0015275177,0.002929661,0.0022445729,0.0021438254,0.008541233,0.013592427,0.0051597604,0.02956378],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0025226257,0.0014770963,0.0028969927,0.0025474739,0.00033816329,0.0006995846,0.0014134349,0.004826052,0.024597367,0.0043454044,0.72508526,0.22925054],"study_design_scores_gemma":[0.0031484247,0.002946826,0.021396682,0.00050940795,0.0003140072,0.0029713176,0.0023181331,0.0958003,0.071172014,0.034290735,0.7644419,0.0006902189],"about_ca_topic_score_codex":0.009252081,"about_ca_topic_score_gemma":0.014305966,"teacher_disagreement_score":0.027552871,"about_ca_system_score_codex":0.0032520848,"about_ca_system_score_gemma":0.007652175,"threshold_uncertainty_score":0.10702485},"labels":[],"label_agreement":null},{"id":"W3033194858","doi":"10.46504/12201700ch","title":"Does Reading SoTL Matter?: Difficult Questions of Impact","year":2017,"lang":"en","type":"article","venue":"InSight A Journal of Scholarly Teaching","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Reading (process); Psychology; Philosophy; Linguistics","score_opus":0.014962534779849134,"score_gpt":0.31487733250429795,"score_spread":0.2999147977244488,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3033194858","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.24211793,0.044471763,0.0031852368,0.55716664,0.0064184787,0.0001837674,0.0029262912,0.00019212237,0.14333773],"genre_scores_gemma":[0.9643266,0.008334606,0.0010268281,0.014576916,0.0046476326,0.00009938962,0.00053445116,0.00024398632,0.0062096626],"study_design_codex":"design_other","study_design_gemma":"qualitative","domain_scores_codex":[0.9820312,0.009862889,0.00089535135,0.0016426637,0.0041364315,0.0014314804],"domain_scores_gemma":[0.7937739,0.14796117,0.018872062,0.007468938,0.020664789,0.011259122],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.026542407,0.00060670625,0.0016564376,0.0038259556,0.002560899,0.009816892,0.0024480869,0.0029151165,0.04210663],"category_scores_gemma":[0.22401215,0.00033651543,0.001029361,0.0061069583,0.008516604,0.021584034,0.004274716,0.00437622,0.0053515467],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0071729315,0.001930744,0.17211302,0.0049841204,0.0021078316,0.0004123906,0.019162506,0.00062387355,0.0011323425,0.12278141,0.13716985,0.5304089],"study_design_scores_gemma":[0.0012169007,0.0015129732,0.332215,0.006515081,0.0030225131,0.0004772956,0.07626566,0.0021730745,0.0026868125,0.4408582,0.13279045,0.0002660736],"about_ca_topic_score_codex":0.0061107716,"about_ca_topic_score_gemma":0.00660065,"teacher_disagreement_score":0.9734576,"about_ca_system_score_codex":0.0028896306,"about_ca_system_score_gemma":0.0051927296,"threshold_uncertainty_score":0.14086068},"labels":[],"label_agreement":null},{"id":"W3033346395","doi":"10.18653/v1/2020.ngt-1.20","title":"Growing Together: Modeling Human Language Learning With n-Best Multi-Checkpoint Machine Translation","year":2020,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Compute Canada","keywords":"Paraphrase; Computer science; Fluency; Macro; Machine translation; Artificial intelligence; Task (project management); Natural language processing; Language model; Translation (biology); Portuguese; Programming language; Linguistics","score_opus":0.047052995056586576,"score_gpt":0.3118409787660523,"score_spread":0.26478798370946577,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3033346395","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19257988,0.0015802267,0.7908281,0.0016231883,0.00026858127,0.00024987062,0.0014178305,0.0065429495,0.004909361],"genre_scores_gemma":[0.799979,0.00032302455,0.18774977,0.0005375887,0.00018972725,0.00043324835,0.0027445315,0.0007653822,0.0072778324],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99877113,0.00058689323,0.000039292594,0.00040845308,0.00010208601,0.0000922185],"domain_scores_gemma":[0.9969517,0.0020046195,0.00014335799,0.00046436515,0.0002857307,0.00015023015],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028679618,0.0014136726,0.0011984817,0.00084304635,0.000762332,0.0016113597,0.002387868,0.0022074634,0.0029972787],"category_scores_gemma":[0.009908143,0.0009809233,0.0012713972,0.0011084002,0.0010613737,0.0024294425,0.0020242415,0.0026660615,0.0014564554],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003250084,0.00013021925,0.001829133,0.000077653945,0.00012594674,0.00011974187,0.00030707373,0.8988906,0.0012969126,0.0032509118,0.0049077007,0.08873912],"study_design_scores_gemma":[0.000014995962,0.000039886283,0.00019884761,0.00000571998,0.0000110536075,0.000013503569,0.000020065878,0.99306846,0.00051369035,0.0055197566,0.00058732624,0.0000066877615],"about_ca_topic_score_codex":0.013740763,"about_ca_topic_score_gemma":0.019275784,"teacher_disagreement_score":0.013740763,"about_ca_system_score_codex":0.001167945,"about_ca_system_score_gemma":0.001139971,"threshold_uncertainty_score":0.027321577},"labels":[],"label_agreement":null},{"id":"W3035351075","doi":"10.24963/ijcai.2020/512","title":"Unsupervised Multilingual Alignment using Wasserstein Barycenter","year":2020,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Collège Boréal; Vector Institute; University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada; Mitacs","keywords":"Computer science; Pairwise comparison; Natural language processing; Transitive relation; Artificial intelligence; Translation (biology); Word (group theory); Machine translation; Process (computing); Quality (philosophy); Programming language; Linguistics; Mathematics","score_opus":0.04874839497133945,"score_gpt":0.3162621032281758,"score_spread":0.26751370825683635,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3035351075","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007940146,0.00049553585,0.9860037,0.00013528707,0.00008480829,0.000051253093,0.00019050828,0.003846311,0.0012523667],"genre_scores_gemma":[0.21860018,0.0005971403,0.766851,0.00038826748,0.0002918363,0.00025233888,0.003610576,0.0030786148,0.006330039],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99703777,0.00093001546,0.00015313132,0.0010105438,0.0006324732,0.00023617767],"domain_scores_gemma":[0.9968893,0.0012519398,0.00042435213,0.00066369365,0.0006112862,0.00015940364],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019873532,0.0025163922,0.002863885,0.0027898545,0.0013940618,0.0023794882,0.0031634758,0.0019004743,0.0039045305],"category_scores_gemma":[0.006911538,0.001224673,0.0016707726,0.004130389,0.0017259504,0.0042528664,0.0029773533,0.0027128637,0.0034914066],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004910046,0.00024813373,0.0013571243,0.00045567323,0.00034435344,0.00037027462,0.00032904875,0.5147204,0.022013579,0.045039028,0.017184498,0.39744684],"study_design_scores_gemma":[0.000028857856,0.000043319575,0.00017024585,0.000015709986,0.000018593904,0.000065401626,0.000025394695,0.95803326,0.0062881196,0.03209534,0.003185383,0.000030396746],"about_ca_topic_score_codex":0.008427948,"about_ca_topic_score_gemma":0.013319978,"teacher_disagreement_score":0.008427948,"about_ca_system_score_codex":0.0018131792,"about_ca_system_score_gemma":0.002345587,"threshold_uncertainty_score":0.016757786},"labels":[],"label_agreement":null},{"id":"W3035668914","doi":"","title":"PoKED: A Semi-Supervised System for Word Sense Disambiguation","year":2020,"lang":"en","type":"article","venue":"International Conference on Machine Learning","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Word-sense disambiguation; Computer science; Natural language processing; SemEval; Artificial intelligence; Word (group theory); Linguistics; WordNet; Engineering","score_opus":0.043391661373164574,"score_gpt":0.30795606690180205,"score_spread":0.26456440552863747,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3035668914","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.026261909,0.0013235192,0.7350302,0.0005507581,0.0014806901,0.0006257761,0.018781938,0.21104534,0.0048999083],"genre_scores_gemma":[0.109294504,0.000480355,0.82320815,0.00072652596,0.0002853524,0.0008905327,0.05011077,0.006275239,0.008728491],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9975318,0.00055488915,0.0004012438,0.0009336634,0.0004564605,0.000121893856],"domain_scores_gemma":[0.9964489,0.0013285342,0.00025780234,0.00075510214,0.00096939324,0.00024029499],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021048486,0.0024772836,0.0018991864,0.004446163,0.002075815,0.0024149145,0.0025069725,0.001973465,0.011932032],"category_scores_gemma":[0.006883409,0.001194317,0.0014022592,0.002644866,0.0007674924,0.0053881337,0.005019866,0.001754738,0.014214392],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017653665,0.00059082377,0.003470843,0.0021804452,0.0005514509,0.0011391488,0.0008710874,0.003902398,0.071710676,0.008693431,0.1985875,0.70653677],"study_design_scores_gemma":[0.0008732226,0.0009079398,0.007759606,0.00041840837,0.000602035,0.0027843514,0.0013699171,0.44269574,0.16488071,0.07621975,0.30092055,0.00056784105],"about_ca_topic_score_codex":0.002551638,"about_ca_topic_score_gemma":0.0053536384,"teacher_disagreement_score":0.011932032,"about_ca_system_score_codex":0.0005809327,"about_ca_system_score_gemma":0.0024367024,"threshold_uncertainty_score":0.039916694},"labels":[],"label_agreement":null},{"id":"W3036116903","doi":"10.1162/tacl_a_00316","title":"Learning Lexical Subspaces in a Distributional Vector Space","year":2020,"lang":"en","type":"article","venue":"Transactions of the Association for Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Mila - Quebec Artificial Intelligence Institute","funders":"","keywords":"Computer science; Distributional semantics; Linear subspace; Artificial intelligence; Natural language processing; Similarity (geometry); Vector space; Word (group theory); Space (punctuation); Relation (database); Semantics (computer science); Semantic similarity; Suite; Code (set theory); Linguistics; Programming language; Data mining; Mathematics","score_opus":0.013875671878323446,"score_gpt":0.2681135972948576,"score_spread":0.25423792541653417,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3036116903","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.055032782,0.0007160115,0.9394506,0.0004169782,0.000048225847,0.00007728327,0.0007656198,0.0016939209,0.0017987245],"genre_scores_gemma":[0.5963921,0.0008294194,0.38916337,0.00047889806,0.00016655552,0.00034509588,0.007380394,0.00040524727,0.004838897],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99813145,0.0006467518,0.0001342734,0.00064173125,0.00030551146,0.00014030767],"domain_scores_gemma":[0.99808264,0.0008137062,0.00021507099,0.0003656897,0.0003592963,0.00016355871],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017422653,0.0011632977,0.0013780601,0.002530649,0.0006995851,0.002564076,0.0013695849,0.0011629409,0.002647024],"category_scores_gemma":[0.0055560893,0.00052513665,0.0011599815,0.0024954777,0.0010757794,0.0060747922,0.0031297496,0.0020186738,0.0016383686],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005051931,0.00053411076,0.017444262,0.0004795365,0.00031198096,0.0003299372,0.0010401017,0.12514746,0.009618644,0.14221428,0.015902992,0.6864715],"study_design_scores_gemma":[0.000024378796,0.00013504457,0.001144367,0.000052317115,0.000021308386,0.000109905275,0.00032571342,0.77694535,0.0012299297,0.21574076,0.0042360565,0.00003492108],"about_ca_topic_score_codex":0.0024928297,"about_ca_topic_score_gemma":0.0041198484,"teacher_disagreement_score":0.002647024,"about_ca_system_score_codex":0.0008138457,"about_ca_system_score_gemma":0.0009682217,"threshold_uncertainty_score":0.009214044},"labels":[],"label_agreement":null},{"id":"W3036873463","doi":"10.1007/978-3-030-50420-5_29","title":"An Empirical Evaluation of Attention and Pointer Networks for Paraphrase Generation","year":2020,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Paraphrase; Computer science; Pointer (user interface); Machine translation; Artificial intelligence; Natural language processing; Sequence (biology); Artificial neural network; Similarity (geometry); Image (mathematics)","score_opus":0.04518612985403888,"score_gpt":0.3375778526078879,"score_spread":0.29239172275384906,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3036873463","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.93505204,0.005590931,0.03501914,0.00077628606,0.00023885128,0.00096600904,0.0037827878,0.0038171657,0.014756805],"genre_scores_gemma":[0.9596015,0.0008129361,0.029042676,0.00015028445,0.00013046445,0.0003572969,0.0059276163,0.00035788244,0.003619407],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9933476,0.004222411,0.00047363498,0.0008730472,0.0009271028,0.00015623063],"domain_scores_gemma":[0.84820503,0.13717198,0.0031505183,0.005940594,0.0043393187,0.001192547],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008932009,0.0012827775,0.0007643171,0.0026814116,0.00072934903,0.001976374,0.0024550075,0.0019574054,0.008991842],"category_scores_gemma":[0.08430547,0.0004929923,0.00056263723,0.0019675724,0.0008174595,0.0056439796,0.0021567456,0.0017334156,0.00222116],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0139813535,0.0062500667,0.03654167,0.0030954685,0.00083171064,0.00045540964,0.0015820897,0.051238537,0.014457632,0.00397809,0.01578378,0.8518042],"study_design_scores_gemma":[0.0024977876,0.008096129,0.05267587,0.00044533104,0.0014317252,0.0011409054,0.0015579042,0.884675,0.025212962,0.012138548,0.009983112,0.00014475475],"about_ca_topic_score_codex":0.0051175975,"about_ca_topic_score_gemma":0.0047360724,"teacher_disagreement_score":0.008991842,"about_ca_system_score_codex":0.0012049554,"about_ca_system_score_gemma":0.0008347506,"threshold_uncertainty_score":0.047237515},"labels":[],"label_agreement":null},{"id":"W3037066510","doi":"10.1002/meet.2009.14504603102","title":"Universal abstracting","year":2009,"lang":"en","type":"article","venue":"Proceedings of the American Society for Information Science and Technology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Lingua franca; Linguistics; Machine translation; Natural language processing; Natural language; Universal Networking Language; Semantics (computer science); Representation (politics); Grammar; Universal grammar; Artificial intelligence; Programming language; Comprehension approach; Generative grammar","score_opus":0.006941895526748819,"score_gpt":0.2580001996303356,"score_spread":0.25105830410358676,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3037066510","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0072835544,0.0014793813,0.72734904,0.001968298,0.0021709295,0.0015944674,0.0176287,0.056615505,0.18391016],"genre_scores_gemma":[0.14815189,0.0029977884,0.5575389,0.0018669745,0.0015791992,0.002008461,0.048041705,0.020091126,0.21772398],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99315286,0.00124801,0.0015366729,0.0013718073,0.0021992666,0.00049147016],"domain_scores_gemma":[0.9838328,0.002850949,0.001041958,0.006877281,0.00485503,0.00054206455],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0053567807,0.0017299553,0.001235519,0.0054357992,0.0017265561,0.006858988,0.0027628357,0.0012608494,0.09497783],"category_scores_gemma":[0.02472169,0.0009063805,0.0020614031,0.004235047,0.0018776897,0.009279747,0.008067557,0.001651276,0.043673314],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00027892375,0.00008922526,0.0011279897,0.0018149075,0.000080072736,0.00045270965,0.0025551617,0.0012477782,0.009422291,0.21412705,0.263366,0.50543797],"study_design_scores_gemma":[0.000027795479,0.000042862608,0.00042366216,0.00023538784,0.00005125953,0.00037112134,0.00039265834,0.0019463318,0.0081971185,0.05488265,0.933368,0.00006107985],"about_ca_topic_score_codex":0.002531205,"about_ca_topic_score_gemma":0.0020551963,"teacher_disagreement_score":0.09497783,"about_ca_system_score_codex":0.0019614953,"about_ca_system_score_gemma":0.004585888,"threshold_uncertainty_score":0.31773245},"labels":[],"label_agreement":null},{"id":"W3037085112","doi":"10.18653/v1/2020.repl4nlp-1.23","title":"Supertagging with CCG primitives","year":2020,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada; Compute Canada","keywords":"Lexicalization; Computer science; Natural language processing; Artificial intelligence; Parsing; Word (group theory); Part of speech; Rule-based machine translation; Task (project management); Sentence; Linguistics","score_opus":0.013937173942154117,"score_gpt":0.23193301783085177,"score_spread":0.21799584388869764,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3037085112","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019754717,0.00014081693,0.9634935,0.00028585858,0.0001337201,0.00017307504,0.00061121094,0.010630395,0.004776678],"genre_scores_gemma":[0.35868862,0.0001482381,0.6288727,0.00069883326,0.00010071,0.00035893038,0.0020827404,0.0032847074,0.005764622],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9981377,0.0005642439,0.00012199895,0.0005527842,0.00045617926,0.0001670992],"domain_scores_gemma":[0.99382895,0.0023882708,0.00022743673,0.002781498,0.00061236945,0.00016142883],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018704401,0.0007340714,0.0006660488,0.001565633,0.00083049916,0.0012151136,0.0019519699,0.0011810985,0.0063225855],"category_scores_gemma":[0.0070988485,0.00060435414,0.0009609935,0.002087702,0.0021532932,0.0032566322,0.0032614914,0.0023618955,0.0029472583],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005478881,0.00020629128,0.004439441,0.00051130453,0.000080694364,0.00091708585,0.002477653,0.052582476,0.06312086,0.29622155,0.029050691,0.549844],"study_design_scores_gemma":[0.00006030575,0.00008980682,0.00084459444,0.00007969847,0.00004770686,0.00036245288,0.00022752098,0.41804558,0.032606594,0.48831925,0.059240956,0.00007541425],"about_ca_topic_score_codex":0.0037059372,"about_ca_topic_score_gemma":0.007816987,"teacher_disagreement_score":0.0063225855,"about_ca_system_score_codex":0.00095997,"about_ca_system_score_gemma":0.0018888923,"threshold_uncertainty_score":0.021151185},"labels":[],"label_agreement":null},{"id":"W3037167856","doi":"10.18653/v1/2020.sigmorphon-1.12","title":"Low-Resource G2P and P2G Conversion with Synthetic Training Data","year":2020,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Innovates","keywords":"Grapheme; Computer science; Task (project management); Resource (disambiguation); Training set; Transduction (biophysics); Artificial intelligence; Machine learning; Computer network; Engineering; Systems engineering","score_opus":0.035921468993388675,"score_gpt":0.24982338500397946,"score_spread":0.21390191601059078,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3037167856","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7614678,0.0025681027,0.1265576,0.002158221,0.0009024725,0.0015504287,0.031495303,0.048165254,0.025134776],"genre_scores_gemma":[0.7862161,0.000488736,0.13581996,0.0006834161,0.00012717623,0.0008647406,0.066763595,0.0019265288,0.007109748],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9967443,0.0014141827,0.00017641067,0.00091154105,0.0004448591,0.00030867875],"domain_scores_gemma":[0.99263144,0.0045804526,0.0001630408,0.0017598317,0.0006316734,0.00023356453],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029605702,0.0027704688,0.001139076,0.0013270191,0.0012835719,0.0017405503,0.003059653,0.0032546702,0.00581545],"category_scores_gemma":[0.013464051,0.0007216164,0.0010697204,0.002739949,0.0015198607,0.0027791644,0.0020453248,0.0032048023,0.0034648755],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0050137145,0.0034541192,0.007459609,0.0020271877,0.0005997787,0.002210597,0.00073240354,0.46506065,0.022170164,0.0037215387,0.109268524,0.37828177],"study_design_scores_gemma":[0.0010157216,0.0011067402,0.0067263665,0.00010395768,0.00015259003,0.00070177426,0.00092916325,0.9174742,0.03912164,0.010167795,0.022334948,0.00016510296],"about_ca_topic_score_codex":0.03012417,"about_ca_topic_score_gemma":0.03195006,"teacher_disagreement_score":0.03012417,"about_ca_system_score_codex":0.001323311,"about_ca_system_score_gemma":0.001477077,"threshold_uncertainty_score":0.0598976},"labels":[],"label_agreement":null},{"id":"W3037443502","doi":"10.1609/aiide.v13i2.12968","title":"Deep Learning for Speech Accent Detection in Videogames","year":2017,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence and Interactive Digital Entertainment","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Social Sciences and Humanities Research Council of Canada; Alberta Biodiversity Monitoring Institute; Alberta Conservation Association; Nvidia","keywords":"Stress (linguistics); Pronunciation; Construct (python library); World Englishes; Psychology; Linguistics; Computer science; Speech recognition","score_opus":0.03794799496025514,"score_gpt":0.31484250824125337,"score_spread":0.27689451328099823,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3037443502","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7160592,0.0011775582,0.27105534,0.0006462811,0.00017458004,0.00012393744,0.00094428303,0.0031222533,0.006696538],"genre_scores_gemma":[0.94831663,0.00021978813,0.04664638,0.0001315712,0.000033714216,0.000041857307,0.001063014,0.000049787624,0.0034972157],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99959224,0.0001424992,0.00002069198,0.00010769336,0.000050864484,0.00008593501],"domain_scores_gemma":[0.99925417,0.0004442071,0.000053243504,0.000050057108,0.0001523153,0.00004599477],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011286003,0.000931253,0.00044814128,0.0007958224,0.00030947337,0.0008991945,0.00073634327,0.0006377724,0.0012690496],"category_scores_gemma":[0.0026670254,0.00030577424,0.0004386835,0.00044891707,0.00037323195,0.0010237113,0.00075173785,0.0012963099,0.000666566],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011231272,0.0006302366,0.016115457,0.00025028636,0.00018038905,0.00041813747,0.000906843,0.16533615,0.04183934,0.0030447966,0.0060469634,0.7641083],"study_design_scores_gemma":[0.0000143865855,0.00009437792,0.0049825013,0.0000205974,0.000021036723,0.000038746217,0.00025618443,0.98052406,0.010335518,0.0024912092,0.0012070378,0.000014313301],"about_ca_topic_score_codex":0.008175296,"about_ca_topic_score_gemma":0.010035315,"teacher_disagreement_score":0.008175296,"about_ca_system_score_codex":0.0008435068,"about_ca_system_score_gemma":0.00038966292,"threshold_uncertainty_score":0.016255379},"labels":[],"label_agreement":null},{"id":"W3037446006","doi":"10.18653/v1/2020.sigmorphon-1.16","title":"One Model to Pronounce Them All: Multilingual Grapheme-to-Phoneme Conversion With a Transformer Ensemble","year":2020,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Compute Canada","keywords":"Grapheme; Transformer; Computer science; Word error rate; Task (project management); Speech recognition; Artificial intelligence; Language model; Natural language processing; Speech synthesis; Engineering; Voltage","score_opus":0.05299286893416044,"score_gpt":0.2905230790569408,"score_spread":0.23753021012278036,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3037446006","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05921626,0.001081249,0.9049506,0.0014688999,0.0009304585,0.00011320979,0.0014843153,0.019504545,0.0112504875],"genre_scores_gemma":[0.7270693,0.00081267644,0.23931664,0.0010969202,0.00034078004,0.00017289585,0.005146155,0.001991995,0.02405257],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995371,0.00010081693,0.000019928759,0.00021532996,0.00007044206,0.000056402358],"domain_scores_gemma":[0.9994524,0.00015760341,0.000023853356,0.00016847745,0.000138893,0.00005878671],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008167754,0.0018838877,0.0008794165,0.00064029684,0.00055408815,0.0011719692,0.001673365,0.0010943401,0.0051807007],"category_scores_gemma":[0.0016471776,0.0005187725,0.0012069035,0.0005945505,0.0005182333,0.002618209,0.0017969103,0.0034190381,0.005621825],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008704458,0.000488148,0.0035182708,0.0002586203,0.00053714594,0.0007642509,0.0005145982,0.18141526,0.044300955,0.015860738,0.043778367,0.7076932],"study_design_scores_gemma":[0.000041077543,0.00010963187,0.00042594477,0.00002055624,0.00013231293,0.00027528848,0.00009840204,0.960033,0.01505354,0.015447162,0.008321952,0.000041091804],"about_ca_topic_score_codex":0.0053810775,"about_ca_topic_score_gemma":0.010615532,"teacher_disagreement_score":0.0053810775,"about_ca_system_score_codex":0.00042700596,"about_ca_system_score_gemma":0.0010112957,"threshold_uncertainty_score":0.017331123},"labels":[],"label_agreement":null},{"id":"W3040594194","doi":"10.1007/978-3-030-51825-7_3","title":"Clause Size Reduction with all-UIP Learning","year":2020,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Reduction (mathematics); Artificial intelligence; Computer security; Mathematics","score_opus":0.013602347541774378,"score_gpt":0.2500519772948739,"score_spread":0.2364496297530995,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3040594194","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015723525,0.00084168784,0.9289714,0.0015890058,0.00055254967,0.00029108016,0.0019394179,0.019311173,0.030780198],"genre_scores_gemma":[0.18941149,0.00040134252,0.7704091,0.0010689762,0.00035327912,0.0003588313,0.0065592546,0.0036080154,0.027829664],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9983365,0.0004248908,0.00009511467,0.0003415427,0.0006056686,0.00019638179],"domain_scores_gemma":[0.99590474,0.0018795802,0.000093678354,0.0015420569,0.0004756404,0.00010433089],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001107837,0.0012413167,0.0011316178,0.0014122347,0.00090262113,0.0013693946,0.003954137,0.0010653184,0.031988107],"category_scores_gemma":[0.0066340263,0.00081034255,0.0020652509,0.0019538566,0.0009511839,0.004219357,0.0033713325,0.0047376733,0.0060586077],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028874227,0.00025570684,0.0005753026,0.0004061037,0.000067695895,0.00014965773,0.000096617725,0.030116145,0.0067245737,0.04093312,0.06686145,0.85352486],"study_design_scores_gemma":[0.00020307169,0.00018645074,0.000668978,0.000117954,0.00014730744,0.0003520583,0.00020951808,0.6285268,0.031404443,0.29963052,0.038507197,0.00004570558],"about_ca_topic_score_codex":0.0028276343,"about_ca_topic_score_gemma":0.0069343615,"teacher_disagreement_score":0.031988107,"about_ca_system_score_codex":0.0009466891,"about_ca_system_score_gemma":0.0018026009,"threshold_uncertainty_score":0.1070109},"labels":[],"label_agreement":null},{"id":"W3040907860","doi":"10.13053/cys-24-2-3335","title":"A Multilingual Study of Multi-Sentence Compression using Word Vertex-Labeled Graphs and Integer Linear Programming","year":2020,"lang":"en","type":"article","venue":"PolyPublie (École Polytechnique de Montréal)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Agence Nationale de la Recherche","keywords":"Grammaticality; Computer science; Automatic summarization; Sentence; Natural language processing; Integer programming; Artificial intelligence; Word (group theory); Graph; Algorithm; Theoretical computer science; Grammar; Mathematics; Linguistics","score_opus":0.028397684383551905,"score_gpt":0.28940798069330853,"score_spread":0.26101029630975664,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3040907860","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17521557,0.0033195766,0.80923355,0.0012036179,0.00016619204,0.00018173819,0.00054594374,0.0020590827,0.008074716],"genre_scores_gemma":[0.64846116,0.0010462897,0.3443208,0.00021560707,0.00031820478,0.00013263103,0.0014782442,0.0005884139,0.0034385948],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99729,0.0015645127,0.00011588199,0.00042799226,0.0004699099,0.00013160435],"domain_scores_gemma":[0.98384446,0.013041485,0.00079575914,0.0007348555,0.0013472749,0.00023617408],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002595317,0.0007450471,0.0006163139,0.0022107037,0.0005737104,0.0015287705,0.0009536325,0.00059672765,0.0024920083],"category_scores_gemma":[0.014612854,0.00035285615,0.0006852938,0.0026166504,0.000961304,0.0029322486,0.00093922904,0.0012942823,0.0005084959],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010833022,0.00084449205,0.0073242495,0.001108299,0.0003509429,0.0006660437,0.0009095537,0.31549513,0.01366727,0.045261122,0.0076759523,0.60561365],"study_design_scores_gemma":[0.00004532662,0.00030635603,0.0017381029,0.000024708324,0.000058090904,0.00018868594,0.0002157244,0.9705064,0.008158247,0.014695171,0.0040198676,0.00004328616],"about_ca_topic_score_codex":0.00906061,"about_ca_topic_score_gemma":0.009779105,"teacher_disagreement_score":0.00906061,"about_ca_system_score_codex":0.0014918341,"about_ca_system_score_gemma":0.0009816521,"threshold_uncertainty_score":0.018015742},"labels":[],"label_agreement":null},{"id":"W3041138137","doi":"","title":"Developing LexO : a Collaborative Editor of Multilingual Lexica and Termino-Ontological Resources in the Humanities","year":2017,"lang":"en","type":"article","venue":"CNR ExploRA","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Deutsche Forschungsgemeinschaft; Ministère de l'Économie, de la Science et de l'Innovation - Québec; Canarie","keywords":"Ontology; Digital humanities; Philosophy; Humanities; Epistemology; Computer science; Sociology; Library science","score_opus":0.06526546304006949,"score_gpt":0.3347112972559999,"score_spread":0.26944583421593044,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3041138137","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011095173,0.0006982222,0.91251224,0.001938771,0.0015520022,0.00051266846,0.005162055,0.035106663,0.031422254],"genre_scores_gemma":[0.07649499,0.00086882646,0.8262465,0.00096942903,0.00091859227,0.0007554571,0.014344007,0.01902314,0.060379054],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99705374,0.0011089672,0.00032669687,0.0006376689,0.00076338014,0.00010946606],"domain_scores_gemma":[0.98812383,0.006643775,0.00047218063,0.0023807855,0.0014892381,0.00089007366],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005861739,0.0010838077,0.0011290304,0.0031744507,0.0013034516,0.0061140102,0.0018618725,0.0012741995,0.030242229],"category_scores_gemma":[0.018399714,0.00078572624,0.00090466696,0.001889473,0.0011187126,0.008698456,0.0062572765,0.002567051,0.013053812],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009950983,0.00035089126,0.002727516,0.0030201867,0.00027330022,0.002600566,0.010053003,0.003809196,0.03172695,0.14388281,0.21822937,0.5823312],"study_design_scores_gemma":[0.00009238356,0.00007152137,0.0004191339,0.00025188128,0.00006109972,0.0006746803,0.0011486963,0.009612011,0.0096644545,0.023978155,0.9539272,0.00009874552],"about_ca_topic_score_codex":0.0006173171,"about_ca_topic_score_gemma":0.0013102544,"teacher_disagreement_score":0.030242229,"about_ca_system_score_codex":0.0006646275,"about_ca_system_score_gemma":0.0022143996,"threshold_uncertainty_score":0.10117036},"labels":[],"label_agreement":null},{"id":"W3041698080","doi":"10.71781/9831","title":"Towards learning sentence representation with self-supervision","year":2019,"lang":"en","type":"dissertation","venue":"Open MIND","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Compute Canada; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs; Canadian Institute for Advanced Research","keywords":"Self representation; Sentence; Representation (politics); Psychology; Artificial intelligence; Computer science; Natural language processing; Cognitive science; Art; Political science; Humanities","score_opus":0.021653391500389654,"score_gpt":0.33349456018342943,"score_spread":0.3118411686830398,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3041698080","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02662409,0.0012540078,0.95876557,0.00055039575,0.00020887906,0.00013175709,0.0005827322,0.009611176,0.0022713637],"genre_scores_gemma":[0.40155548,0.000980206,0.5787948,0.00072308606,0.00040841234,0.00030820628,0.004088163,0.0007751642,0.01236651],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989724,0.0003582838,0.00005980384,0.00033028467,0.00018984174,0.00008940683],"domain_scores_gemma":[0.99740666,0.0013335659,0.00018205111,0.00045011824,0.00053056213,0.0000971116],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001685745,0.0014283524,0.0011059426,0.0012225629,0.0004235612,0.0012921023,0.0020717117,0.0016002472,0.0042504882],"category_scores_gemma":[0.005714592,0.0007070771,0.0013498815,0.0009276426,0.00061383774,0.0029199813,0.0013099356,0.0022912247,0.0023926082],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00040724693,0.00025472688,0.0016942226,0.00035002577,0.00021574953,0.00019940615,0.00046303624,0.08297849,0.020362789,0.0058496348,0.016332692,0.8708919],"study_design_scores_gemma":[0.000021841983,0.000098273034,0.00041248734,0.00002921866,0.000039225048,0.000061059276,0.00004289204,0.9833999,0.0066596465,0.00620813,0.0030140956,0.000013245204],"about_ca_topic_score_codex":0.0066272486,"about_ca_topic_score_gemma":0.009427286,"teacher_disagreement_score":0.0066272486,"about_ca_system_score_codex":0.00089168857,"about_ca_system_score_gemma":0.0013118885,"threshold_uncertainty_score":0.014219344},"labels":[],"label_agreement":null},{"id":"W3043085860","doi":"10.17613/34jx-fw32","title":"Seeing the Heiltsuk orthography from font encoding through to Unicode: A case study using convertextract","year":2018,"lang":"en","type":"article","venue":"Humanities Commons CORE (Modern Language Association / Columbia University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Unicode; Computer science; World Wide Web; Transliteration; Encoding (memory); Font; Orthography; Writing system; Rendering (computer graphics); Natural language processing; Artificial intelligence; Linguistics","score_opus":0.0426305788822605,"score_gpt":0.26976166655299105,"score_spread":0.22713108767073054,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3043085860","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.92905945,0.0005888471,0.009684894,0.0029926728,0.00015742982,0.00015938446,0.00018837265,0.00025709358,0.05691192],"genre_scores_gemma":[0.9213325,0.0010854504,0.023208339,0.00091248006,0.00003922741,0.00009736871,0.0001746535,0.0005720892,0.052577954],"study_design_codex":"qualitative","study_design_gemma":"not_applicable","domain_scores_codex":[0.9979759,0.0009502996,0.00014723728,0.00027219113,0.00045803149,0.00019628616],"domain_scores_gemma":[0.9975102,0.0015889737,0.00019319945,0.00030815118,0.00022798091,0.00017146682],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016096468,0.0006507858,0.00039898633,0.0010856056,0.008029916,0.004472155,0.0014288069,0.002633448,0.005889633],"category_scores_gemma":[0.0072972733,0.00048264727,0.00039360445,0.0016371219,0.004653058,0.004187164,0.0029004372,0.0037244041,0.0012787094],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018866359,0.00063413964,0.009905721,0.0004698957,0.00002146993,0.091533124,0.7223507,0.0010022989,0.009990013,0.0335531,0.010272215,0.1200787],"study_design_scores_gemma":[0.000036468653,0.0003687472,0.014924402,0.0005136956,0.00006432788,0.09022726,0.5426828,0.004153156,0.029207725,0.009638157,0.30797905,0.00020417193],"about_ca_topic_score_codex":0.02710075,"about_ca_topic_score_gemma":0.08784703,"teacher_disagreement_score":0.02710075,"about_ca_system_score_codex":0.0032483155,"about_ca_system_score_gemma":0.00232851,"threshold_uncertainty_score":0.053885996},"labels":[],"label_agreement":null},{"id":"W3044096252","doi":"10.3758/s13423-020-01769-w","title":"Prenominal adjective order is such a fat big deal because adjectives are ordered by likely need","year":2020,"lang":"en","type":"review","venue":"Psychonomic Bulletin & Review","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Adjective; Psychology; Order (exchange); Linguistics; Cognitive psychology; Noun","score_opus":0.032204521797818485,"score_gpt":0.32033137225805575,"score_spread":0.2881268504602373,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3044096252","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.001857453,0.93465585,0.019433899,0.0058366507,0.0032386202,0.00006837702,0.00040985915,0.00024746775,0.03425189],"genre_scores_gemma":[0.019872895,0.9374354,0.019869393,0.005786781,0.0017597222,0.000118751675,0.00074560056,0.00017224356,0.014239118],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9994947,0.00010194673,0.000052666142,0.00013177056,0.00019384263,0.000025128666],"domain_scores_gemma":[0.99790084,0.0012806708,0.00013281943,0.00015829314,0.0004898479,0.000037471844],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012615018,0.00070119015,0.0011672259,0.0013845671,0.000357148,0.0015031399,0.00092018116,0.0006786867,0.0063789207],"category_scores_gemma":[0.0032672612,0.0003265858,0.0005788289,0.0022080692,0.0021433125,0.0039859354,0.0008895501,0.0020559474,0.0053440873],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007744969,0.000040311177,0.00036856413,0.0063424646,0.00006272079,0.00007573505,0.00013713387,0.00010578286,0.0026600976,0.048899125,0.03066698,0.9105637],"study_design_scores_gemma":[0.000015277818,0.000039161972,0.0011856761,0.0010163982,0.000092125345,0.0003785036,0.00012638913,0.000087378634,0.0010990566,0.022940459,0.9729988,0.000020841137],"about_ca_topic_score_codex":0.0028038947,"about_ca_topic_score_gemma":0.0048889234,"teacher_disagreement_score":0.0063789207,"about_ca_system_score_codex":0.0007991783,"about_ca_system_score_gemma":0.0028816166,"threshold_uncertainty_score":0.021339595},"labels":[],"label_agreement":null},{"id":"W3046646387","doi":"","title":"Digital Linguistic Landscape: the multilingual system of Ottawa.","year":2020,"lang":"en","type":"article","venue":"DH","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Linguistic landscape; Linguistics; Computer science; Geography","score_opus":0.011296498897944377,"score_gpt":0.2512260856020263,"score_spread":0.23992958670408193,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3046646387","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.058133267,0.0018397177,0.022752926,0.0018969133,0.0003724313,0.00051923207,0.7128696,0.051103342,0.15051259],"genre_scores_gemma":[0.32106268,0.002609934,0.08424762,0.00045052564,0.000078138604,0.00091993867,0.40248448,0.008953707,0.17919305],"study_design_codex":"not_applicable","study_design_gemma":"observational","domain_scores_codex":[0.9995202,0.000063440835,0.00006011558,0.000120031655,0.00015982539,0.00007635135],"domain_scores_gemma":[0.99839395,0.0002036494,0.00008270205,0.00031253218,0.0007064784,0.00030061917],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00066828215,0.0005987946,0.00040914406,0.0048326696,0.002411365,0.0040098457,0.0011677719,0.0004738961,0.03783349],"category_scores_gemma":[0.0035180552,0.00042499154,0.00026120475,0.006504793,0.00082035956,0.0033346196,0.002781523,0.0006485948,0.012327307],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010419232,0.00008626037,0.020641474,0.0013196652,0.00012349394,0.000869759,0.009292128,0.0022623325,0.009396685,0.018878957,0.6218355,0.31425175],"study_design_scores_gemma":[0.000026639496,0.000018562938,0.027145803,0.00024713646,0.00009062833,0.00018606281,0.0037114953,0.0027930883,0.0052238298,0.0033434902,0.95706,0.00015326915],"about_ca_topic_score_codex":0.8305873,"about_ca_topic_score_gemma":0.90420157,"teacher_disagreement_score":0.16941267,"about_ca_system_score_codex":0.006984524,"about_ca_system_score_gemma":0.0117765,"threshold_uncertainty_score":0.34082073},"labels":[],"label_agreement":null},{"id":"W3047612559","doi":"10.4000/linx.6671","title":"Deux dictionnaires informatisés de Jean Dubois et Françoise Dubois-Charlier, leurs ultimes travaux","year":2020,"lang":"fr","type":"article","venue":"Linx","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Humanities; Art; Philosophy","score_opus":0.042595126078079026,"score_gpt":0.2917173231674895,"score_spread":0.24912219708941047,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3047612559","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.319285,0.040445615,0.23509607,0.02677583,0.0069607515,0.00047192522,0.025986439,0.0038691224,0.34110928],"genre_scores_gemma":[0.55276793,0.018451897,0.15537977,0.003898681,0.000991493,0.000347341,0.01874547,0.003394753,0.24602269],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9981761,0.00053797016,0.0002453498,0.00035867823,0.0005907815,0.00009115577],"domain_scores_gemma":[0.9949951,0.002163384,0.00027983138,0.00049580185,0.0018639946,0.00020182447],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018167746,0.00082210905,0.00044957592,0.0042372374,0.0022583394,0.003613798,0.00037156604,0.0008300885,0.017803673],"category_scores_gemma":[0.0064054383,0.00040612393,0.00037815166,0.0041958913,0.0017662883,0.0022291108,0.0013971033,0.0018485596,0.004492215],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00066764606,0.00010814009,0.02151815,0.0019866216,0.00012194981,0.0018523681,0.06400006,0.0019056322,0.027629344,0.18324593,0.13900338,0.5579608],"study_design_scores_gemma":[0.00000953538,0.00002092025,0.0075112814,0.00024709542,0.000014114678,0.0005020017,0.0032850415,0.00031880214,0.0033953243,0.001448443,0.9831976,0.000049746635],"about_ca_topic_score_codex":0.07019844,"about_ca_topic_score_gemma":0.089408934,"teacher_disagreement_score":0.07019844,"about_ca_system_score_codex":0.0030378697,"about_ca_system_score_gemma":0.0035159595,"threshold_uncertainty_score":0.13957971},"labels":[],"label_agreement":null},{"id":"W3047780583","doi":"10.31234/osf.io/xzer9_v1","title":"Degrees of Separation in Semantic and Syntactic Relationships","year":2025,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"Natural Sciences and Engineering Research Council of Canada; National Science Foundation","keywords":"Natural language processing; Computer science; Artificial intelligence; Linguistics; Philosophy","score_opus":0.03353790272009815,"score_gpt":0.3262894078272938,"score_spread":0.29275150510719566,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3047780583","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9098376,0.000284172,0.07512344,0.0005811861,0.000020484924,0.000044442015,0.00018527187,0.0002415373,0.0136819],"genre_scores_gemma":[0.99270743,0.00006703663,0.0064659626,0.00004514804,0.000007609517,0.000023443523,0.00009362228,0.000022234484,0.0005675955],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99742633,0.0007708067,0.0002898948,0.00053610286,0.00071059086,0.00026626405],"domain_scores_gemma":[0.9875403,0.006219022,0.0014616005,0.0035299517,0.00065705576,0.00059217605],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025628423,0.00055694685,0.00044139664,0.0012650411,0.0007423985,0.0036119372,0.0009080103,0.0010662159,0.004069873],"category_scores_gemma":[0.020757837,0.0006682185,0.00048181502,0.00086647423,0.0031206608,0.0075518964,0.0054707634,0.0019039352,0.00053200865],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020353752,0.0005838724,0.08884732,0.0005607628,0.00040702845,0.00059838983,0.008646423,0.04896192,0.1523207,0.5228421,0.0010545057,0.17314157],"study_design_scores_gemma":[0.000100523175,0.00027479083,0.02811196,0.00005863481,0.00017286884,0.00082806405,0.0026527264,0.06648955,0.026431728,0.86801654,0.0067173084,0.00014527199],"about_ca_topic_score_codex":0.00038584412,"about_ca_topic_score_gemma":0.0005000343,"teacher_disagreement_score":0.004069873,"about_ca_system_score_codex":0.0006025597,"about_ca_system_score_gemma":0.00047562076,"threshold_uncertainty_score":0.013615072},"labels":[],"label_agreement":null},{"id":"W3047949969","doi":"","title":"Data Processing Techniques for the TAC-KBP 2019 EDL Task.","year":2019,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Task (project management); Computer science; Engineering","score_opus":0.012221810131879693,"score_gpt":0.2933966171739274,"score_spread":0.2811748070420477,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3047949969","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013464948,0.0016294905,0.80671406,0.0026985505,0.0005014423,0.0011293637,0.054638837,0.09294847,0.026274921],"genre_scores_gemma":[0.09771325,0.00075441366,0.79109365,0.00062335626,0.00014439627,0.0010838622,0.0936972,0.005358905,0.009531096],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.996354,0.0010394598,0.00063515943,0.0006530712,0.001087412,0.00023094847],"domain_scores_gemma":[0.99331176,0.0028965673,0.00024765564,0.0017636529,0.0015853806,0.00019499652],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003547678,0.0014238803,0.00087274006,0.0043315245,0.001630652,0.0032642176,0.002629342,0.0018150297,0.023027081],"category_scores_gemma":[0.014854777,0.00075837184,0.00148954,0.0043155546,0.00058375456,0.006009973,0.0039507304,0.0022632752,0.017054059],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00075130805,0.00034446924,0.0020220014,0.0022885331,0.000153509,0.0009954117,0.0010167161,0.007429893,0.016093317,0.035226468,0.36522186,0.56845653],"study_design_scores_gemma":[0.00039777183,0.0002795319,0.0030875476,0.000657729,0.00016299011,0.0016032957,0.0017993533,0.22018582,0.050705954,0.14938961,0.5715444,0.0001859858],"about_ca_topic_score_codex":0.0067645465,"about_ca_topic_score_gemma":0.009251132,"teacher_disagreement_score":0.023027081,"about_ca_system_score_codex":0.0012408438,"about_ca_system_score_gemma":0.0023836053,"threshold_uncertainty_score":0.07703328},"labels":[],"label_agreement":null},{"id":"W3048188704","doi":"10.1162/ling_a_00401","title":"Statistical Evidence for Learnable Lexical Subclasses in Japanese","year":2020,"lang":"en","type":"article","venue":"Linguistic Inquiry","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; Mila - Quebec Artificial Intelligence Institute","funders":"","keywords":"Phonotactics; Lexicon; Learnability; Linguistics; Computer science; Natural language processing; Phonology; Psychology; Artificial intelligence; Philosophy","score_opus":0.1347884067309446,"score_gpt":0.3918053928929698,"score_spread":0.2570169861620252,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3048188704","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9959111,0.00007803074,0.002814628,0.000050761864,0.000003333662,0.0000048876022,0.00006904482,0.00003449985,0.0010337408],"genre_scores_gemma":[0.9988931,0.000043423395,0.0006305369,0.000015202149,0.0000052972714,0.0000073375795,0.00022532885,0.00001672543,0.0001632955],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99930644,0.00015598281,0.000088212524,0.0003031807,0.00008886886,0.000057352776],"domain_scores_gemma":[0.98863214,0.006331464,0.0014168416,0.0018353565,0.0012402807,0.0005438726],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001053056,0.00024410733,0.0004881956,0.0019090315,0.00093385595,0.0013078301,0.00055398437,0.00047409016,0.0020028644],"category_scores_gemma":[0.009094245,0.000458129,0.00032462706,0.0013543746,0.0018817774,0.001223277,0.001078531,0.00063578266,0.00032147823],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010553402,0.0002994521,0.75848216,0.0005475041,0.00040034906,0.0009836829,0.010616791,0.011362945,0.09113906,0.014092133,0.001518617,0.109502],"study_design_scores_gemma":[0.00003696006,0.00020087948,0.92215085,0.00002907421,0.00015348868,0.0004321318,0.0032697588,0.04731173,0.007261821,0.017580293,0.0014882147,0.00008478479],"about_ca_topic_score_codex":0.0085607795,"about_ca_topic_score_gemma":0.011694756,"teacher_disagreement_score":0.0085607795,"about_ca_system_score_codex":0.00044059785,"about_ca_system_score_gemma":0.00044068482,"threshold_uncertainty_score":0.017021894},"labels":[],"label_agreement":null},{"id":"W3048942436","doi":"10.15760/honors.947","title":"An Introductory Overview of the Koyukon (Athabaskan) Verb","year":2020,"lang":"en","type":"dissertation","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Verb; Linguistics; Theme (computing); Meaning (existential); Language family; Psychology; Computer science; Philosophy; World Wide Web","score_opus":0.018037646916586034,"score_gpt":0.3039962798752337,"score_spread":0.28595863295864765,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3048942436","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.030448452,0.25610343,0.03325367,0.0045924196,0.0019064728,0.0003172504,0.0025745104,0.00039686228,0.6704069],"genre_scores_gemma":[0.25016093,0.41541833,0.067958534,0.0029918144,0.00161432,0.0007214506,0.00838592,0.00078955287,0.2519592],"study_design_codex":"design_other","study_design_gemma":"qualitative","domain_scores_codex":[0.99988484,0.000024140094,0.000019944517,0.000030502439,0.000022923312,0.000017670704],"domain_scores_gemma":[0.99990904,0.000041587424,0.000008776021,0.000005677205,0.000023060935,0.00001181105],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00018932763,0.000606959,0.00040816917,0.0021435327,0.0017269679,0.0020232506,0.00035225303,0.000825407,0.012518132],"category_scores_gemma":[0.00034623066,0.0003364471,0.0002445954,0.0028732137,0.0010880636,0.0027803646,0.001028373,0.0012744913,0.0039356523],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010636912,0.00015214691,0.00466823,0.004385936,0.000022866989,0.0013982558,0.030289495,0.0010533833,0.010043005,0.34816846,0.09552782,0.5041841],"study_design_scores_gemma":[0.0000015968077,0.000022178223,0.004947812,0.0006506708,0.000005831396,0.00078923727,0.0019884394,0.00015626913,0.0002719587,0.009824953,0.9813277,0.0000132908835],"about_ca_topic_score_codex":0.009024816,"about_ca_topic_score_gemma":0.017198756,"teacher_disagreement_score":0.012518132,"about_ca_system_score_codex":0.0014854225,"about_ca_system_score_gemma":0.0013252124,"threshold_uncertainty_score":0.04187733},"labels":[],"label_agreement":null},{"id":"W3050298219","doi":"10.1007/978-3-030-55814-7_2","title":"Extraction of a Knowledge Graph from French Cultural Heritage Documents","year":2020,"lang":"en","type":"book-chapter","venue":"Communications in computer and information science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Cultural heritage; Knowledge graph; Information extraction; Graph; Christian ministry; Relationship extraction; Knowledge extraction; Information retrieval; Context (archaeology); World Wide Web; Natural language processing; Artificial intelligence; Geography; Archaeology; Political science","score_opus":0.03249528905846527,"score_gpt":0.3237258388079702,"score_spread":0.2912305497495049,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3050298219","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.24769041,0.008299039,0.5944307,0.0027893235,0.0004987797,0.0012139927,0.06788064,0.02716357,0.05003363],"genre_scores_gemma":[0.32673067,0.0042752675,0.5554861,0.00033461233,0.00014892606,0.00026925138,0.09076196,0.00092938123,0.021063844],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997527,0.000036166257,0.000022926579,0.000092456175,0.0000688779,0.000026870533],"domain_scores_gemma":[0.9994578,0.00029342988,0.000035702546,0.000051756073,0.00013258541,0.00002873287],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00018879781,0.00091474055,0.0006273453,0.008083963,0.0010371034,0.0016897178,0.0006590874,0.0009185405,0.004599318],"category_scores_gemma":[0.001177349,0.00040090238,0.0010270308,0.0051699895,0.00035427386,0.0011831614,0.00060644787,0.00056651246,0.0024040549],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00030176443,0.0002934471,0.0058719795,0.0016129259,0.00024823827,0.0036892686,0.001092535,0.010201037,0.051182225,0.015498508,0.048341982,0.8616662],"study_design_scores_gemma":[0.00017098224,0.0004732051,0.047155943,0.0012584431,0.0015635327,0.006323657,0.0032005548,0.25334817,0.12112884,0.04756227,0.5176004,0.00021390972],"about_ca_topic_score_codex":0.03496833,"about_ca_topic_score_gemma":0.044479225,"teacher_disagreement_score":0.03496833,"about_ca_system_score_codex":0.0010439577,"about_ca_system_score_gemma":0.001979954,"threshold_uncertainty_score":0.06952953},"labels":[],"label_agreement":null},{"id":"W30536900","doi":"10.1111/risa.13248","title":"Data-driven computational linguistics at FaMAF-UNC, Argentina","year":2010,"lang":"en","type":"article","venue":"North American Chapter of the Association for Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Computational linguistics; Computer science; Applied linguistics; Linguistics; Language technology; Language and Communication Technologies; Natural language; Natural language processing; Data science; Philosophy","score_opus":0.02033526140629643,"score_gpt":0.28923574419932174,"score_spread":0.2689004827930253,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W30536900","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15314628,0.046915255,0.23508203,0.09323102,0.004262451,0.0008120768,0.08538916,0.017817825,0.36334383],"genre_scores_gemma":[0.5287258,0.019477123,0.23023011,0.0021630833,0.0010689422,0.0014509899,0.060651407,0.007401015,0.14883156],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9948338,0.0028256837,0.0002480086,0.0010758055,0.0007909596,0.00022571236],"domain_scores_gemma":[0.9887129,0.00753114,0.000402211,0.00091914367,0.001954317,0.0004803114],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0051890165,0.00067341834,0.00086794095,0.0022762485,0.002668364,0.0051373565,0.00084636884,0.0014601091,0.023013491],"category_scores_gemma":[0.015931094,0.0005345237,0.0007315933,0.0035628781,0.0016881162,0.0021104547,0.0023573348,0.0013912023,0.007413831],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00062485976,0.00025381724,0.018244332,0.0015040992,0.00013876492,0.0013109107,0.007436552,0.01163536,0.0029849308,0.21660247,0.35053366,0.38873023],"study_design_scores_gemma":[0.00006935431,0.000040815958,0.012053675,0.0009571611,0.000027839214,0.0003126755,0.0018854038,0.026272207,0.0017803172,0.078235775,0.8783075,0.00005727051],"about_ca_topic_score_codex":0.055790715,"about_ca_topic_score_gemma":0.04460387,"teacher_disagreement_score":0.055790715,"about_ca_system_score_codex":0.006513893,"about_ca_system_score_gemma":0.007582611,"threshold_uncertainty_score":0.11093199},"labels":[],"label_agreement":null},{"id":"W307283492","doi":"10.17161/iallt.v11i2.8979","title":"The Evolution of the Language Laboratory","year":2019,"lang":"en","type":"article","venue":"IALLT Journal of Language Learning Technologies","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Appalachian State University; University of South Florida; East Carolina University; Dartmouth College; University of Kansas; University of Waterloo; Brigham Young University; Auburn University; U.S. Department of State; Harvard University; Princeton University; North Carolina State University; Purdue University; University of Southern California; Wayne State University; Rice University; Yale University; Emory University; Boston College","keywords":"Computer science; Linguistics; Astrobiology; Biology; Philosophy","score_opus":0.0035752908435706865,"score_gpt":0.2383473010200142,"score_spread":0.2347720101764435,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W307283492","genre_codex":"other","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07797237,0.012693229,0.13784893,0.20840956,0.0037161505,0.00036838162,0.0006577147,0.0036005303,0.55473316],"genre_scores_gemma":[0.69825983,0.0052984064,0.096410476,0.025593404,0.0027707259,0.00074555154,0.00055686204,0.0021329043,0.16823186],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9873153,0.0063664354,0.0005307724,0.0027392001,0.0020950465,0.0009532129],"domain_scores_gemma":[0.95434517,0.02048081,0.0014834672,0.007853746,0.010132738,0.0057040206],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.022972994,0.00081788836,0.00076574675,0.004725794,0.0051103025,0.015008098,0.0043201055,0.0059472597,0.030100504],"category_scores_gemma":[0.035760038,0.00102661,0.0010135522,0.0019502373,0.02857307,0.024894744,0.01185273,0.009501653,0.0062650973],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011303489,0.00018873179,0.00089101255,0.00007684169,0.0000135940045,0.00009599232,0.002665926,0.0006756142,0.0009591422,0.9196345,0.010680527,0.06400494],"study_design_scores_gemma":[0.000092117625,0.00023515477,0.0014461143,0.00031248058,0.000015151108,0.000258943,0.0023575376,0.004627162,0.003232395,0.30533862,0.6819707,0.00011365618],"about_ca_topic_score_codex":0.00806911,"about_ca_topic_score_gemma":0.005990009,"teacher_disagreement_score":0.030100504,"about_ca_system_score_codex":0.015101324,"about_ca_system_score_gemma":0.013471292,"threshold_uncertainty_score":0.12149423},"labels":[],"label_agreement":null},{"id":"W3072958547","doi":"10.1177/0023830920932955","title":"Beyond Plain and Extra-Grammatical Morphology: Echo-Pairs in Hungarian","year":2020,"lang":"en","type":"article","venue":"Language and Speech","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Context (archaeology); Similarity (geometry); Generalization; Grammar; Linguistics; Echo (communications protocol); Lexicon; Contrast (vision); Set (abstract data type); Computer science; Natural language processing; Artificial intelligence; Psychology; Mathematics; Geography","score_opus":0.010469820437185424,"score_gpt":0.25114089791927113,"score_spread":0.24067107748208572,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3072958547","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9937443,0.00008868001,0.0029537475,0.000053627136,0.000011409724,0.000017097522,0.00022145333,0.00003492105,0.0028749073],"genre_scores_gemma":[0.99761546,0.00003518677,0.0014127728,0.000028989982,0.000004845378,0.000015411946,0.00039858551,0.00003626251,0.00045250502],"study_design_codex":"observational","study_design_gemma":"not_applicable","domain_scores_codex":[0.9991744,0.00035803718,0.00009862685,0.000205537,0.00010791632,0.000055491208],"domain_scores_gemma":[0.993494,0.0046028323,0.0005808471,0.000801201,0.00034695028,0.00017435926],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011838773,0.0002708931,0.00038895925,0.00064906786,0.0007604057,0.0009163604,0.00042540763,0.00068976765,0.00516335],"category_scores_gemma":[0.008503854,0.00023819762,0.00015347691,0.0008970303,0.0014368041,0.0018292817,0.0014929881,0.0008034544,0.0009010784],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.003894688,0.00078977435,0.30409563,0.002239685,0.0003166051,0.013055933,0.21494131,0.002604294,0.1368771,0.033598687,0.0070430283,0.28054336],"study_design_scores_gemma":[0.00016790236,0.0005165908,0.8541913,0.00012592215,0.00012764813,0.012626926,0.04877085,0.0065024537,0.02687477,0.020006211,0.029844524,0.00024489994],"about_ca_topic_score_codex":0.0012077534,"about_ca_topic_score_gemma":0.002048084,"teacher_disagreement_score":0.00516335,"about_ca_system_score_codex":0.00024305319,"about_ca_system_score_gemma":0.00016088663,"threshold_uncertainty_score":0.017273128},"labels":[],"label_agreement":null},{"id":"W3082303080","doi":"","title":"Vous avez dit \"proéminence\" ?","year":2006,"lang":"fr","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Variation (astronomy); Prosody; Identification (biology); Computer science; Task (project management); Coding (social sciences); Speech recognition; Natural language processing; Artificial intelligence; Psychology; Statistics; Mathematics; Engineering; Biology","score_opus":0.014598428524994836,"score_gpt":0.24091363948093725,"score_spread":0.22631521095594243,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3082303080","genre_codex":"commentary","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0116423685,0.025025245,0.014832522,0.7460093,0.055382445,0.000053466854,0.00042321932,0.0004728044,0.14615867],"genre_scores_gemma":[0.24590713,0.023432272,0.014551104,0.16790575,0.025333544,0.00027766594,0.0009967482,0.0016585663,0.5199373],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.990703,0.0034237544,0.0002761424,0.0019036392,0.002683329,0.0010101593],"domain_scores_gemma":[0.9903442,0.0020906215,0.00046417973,0.001305098,0.0034653062,0.002330426],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00894554,0.00081853435,0.0011446227,0.0011350472,0.006073505,0.015518378,0.0023377135,0.0052916147,0.048082344],"category_scores_gemma":[0.025529612,0.00040270417,0.0005649807,0.001487732,0.007652777,0.022763329,0.007904431,0.0075025666,0.01795098],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020503883,0.000059791837,0.0019450993,0.00036429765,0.000040802144,0.0002575655,0.009416811,0.00018496961,0.0013093976,0.40297958,0.41921195,0.16402471],"study_design_scores_gemma":[0.000007515194,0.000020660742,0.00057550555,0.00017397154,0.0000081082735,0.00017303649,0.0068737557,0.00016529943,0.00033561245,0.07917567,0.9124699,0.000020913298],"about_ca_topic_score_codex":0.003830092,"about_ca_topic_score_gemma":0.0072209644,"teacher_disagreement_score":0.048082344,"about_ca_system_score_codex":0.0040237526,"about_ca_system_score_gemma":0.0060699373,"threshold_uncertainty_score":0.16085148},"labels":[],"label_agreement":null},{"id":"W3082687447","doi":"10.5539/ijel.v10n6p118","title":"Corpus Pattern Analysis of of-Construction Phrase Transformations to the Genitive","year":2020,"lang":"en","type":"article","venue":"International Journal of English Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"Japan Society for the Promotion of Science","keywords":"Genitive case; Phrase; Face (sociological concept); Linguistics; Focus (optics); Determiner phrase; Meaning (existential); Noun phrase; Computer science; Artificial intelligence; Natural language processing; Psychology; Philosophy","score_opus":0.013243082120111833,"score_gpt":0.2782057377063923,"score_spread":0.26496265558628046,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3082687447","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8584627,0.0015404733,0.047456395,0.00091966917,0.00038545125,0.0010184704,0.02212916,0.00084180053,0.067245826],"genre_scores_gemma":[0.90314436,0.00095036544,0.056700677,0.00017997208,0.00007708686,0.0015916149,0.028401485,0.00065717264,0.008297298],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99777883,0.0007290453,0.00021555183,0.0006190473,0.00055813306,0.00009942555],"domain_scores_gemma":[0.98894393,0.006232356,0.00084983226,0.0016164831,0.0022231392,0.00013415533],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014189119,0.00038241112,0.00041287267,0.0031698185,0.0013970255,0.0014947577,0.00061514403,0.00044929874,0.006847785],"category_scores_gemma":[0.01322917,0.00033027926,0.00040651497,0.0072736815,0.00112864,0.0012905784,0.0015510995,0.0013474367,0.001468793],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010801476,0.0003555483,0.053738054,0.0037233883,0.00023049794,0.004256479,0.06930892,0.004218325,0.09831419,0.07723074,0.057466958,0.63007677],"study_design_scores_gemma":[0.00019720389,0.0004108671,0.36547703,0.00075506454,0.0002907509,0.0065909517,0.044075992,0.038889084,0.065357774,0.025330357,0.45238385,0.0002410518],"about_ca_topic_score_codex":0.008055977,"about_ca_topic_score_gemma":0.010959505,"teacher_disagreement_score":0.008055977,"about_ca_system_score_codex":0.0011915386,"about_ca_system_score_gemma":0.0012627713,"threshold_uncertainty_score":0.022908151},"labels":[],"label_agreement":null},{"id":"W3082920524","doi":"10.11606/d.45.2004.tde-20210729-142722","title":"Aprendizado de regras de substituição para normatização de textos históricos","year":2004,"lang":"pt","type":"dissertation","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Regina","funders":"","keywords":"Spelling; Lexicon; Linguistics; Computer science; Portuguese; Natural language processing; Computation; Artificial intelligence; Humanities; Art; Philosophy; Algorithm","score_opus":0.02037653076917156,"score_gpt":0.31081618177886605,"score_spread":0.2904396510096945,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3082920524","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.25582427,0.0022476895,0.6419799,0.0016327366,0.00079949596,0.00092826155,0.002270872,0.02113383,0.073182985],"genre_scores_gemma":[0.48437798,0.0012706405,0.48509812,0.00025416483,0.00021128655,0.00027968013,0.0022911753,0.0027549,0.023462037],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99796605,0.0006809603,0.00018182985,0.0004936962,0.00050489034,0.00017269503],"domain_scores_gemma":[0.9924839,0.0022165792,0.00052833615,0.0029066568,0.0016617844,0.00020259916],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003443709,0.0007403959,0.00083604804,0.0020523171,0.0013635241,0.0028076367,0.0010552664,0.0005816907,0.011469344],"category_scores_gemma":[0.011531818,0.0004295517,0.0011317321,0.0018889551,0.0016839267,0.0037466218,0.002000165,0.0013668558,0.0036044146],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014801876,0.00028959406,0.0080124065,0.0012247933,0.000110165274,0.000835093,0.008181303,0.0033495643,0.06177686,0.07011279,0.013070074,0.8315571],"study_design_scores_gemma":[0.00039488616,0.0011105919,0.022277404,0.0009087671,0.0006931777,0.0023805655,0.011547988,0.077763245,0.20566374,0.10480723,0.572187,0.00026545988],"about_ca_topic_score_codex":0.004454711,"about_ca_topic_score_gemma":0.008535743,"teacher_disagreement_score":0.011469344,"about_ca_system_score_codex":0.0009898777,"about_ca_system_score_gemma":0.0015188688,"threshold_uncertainty_score":0.03836876},"labels":[],"label_agreement":null},{"id":"W3085940072","doi":"10.17619/unipb/1-980","title":"Knowledge Graphs for Multilingual Language Translation and Generation","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"Deutscher Akademischer Austauschdienst","keywords":"Machine translation; Computer science; Artificial intelligence; Natural language processing; Proper noun; Noun; Natural language generation; Inference; Quality (philosophy); Artificial neural network; Natural language; Linguistics; Philosophy","score_opus":0.11756955366207564,"score_gpt":0.24957436377142353,"score_spread":0.1320048101093479,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3085940072","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0030867853,0.0035695503,0.9663567,0.0015557447,0.00036813167,0.00020135222,0.0031084432,0.00866869,0.0130846575],"genre_scores_gemma":[0.1432167,0.006089296,0.8196124,0.0009357084,0.00036000228,0.00061190285,0.01771587,0.0025455903,0.008912623],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99793065,0.00075202744,0.00017984191,0.00056173885,0.00046242535,0.00011326295],"domain_scores_gemma":[0.99543744,0.0023530493,0.00028392655,0.0012865469,0.0005305109,0.00010842995],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017407567,0.0013621062,0.0008573695,0.003669851,0.0011010235,0.003015104,0.0020286785,0.0016878004,0.019071246],"category_scores_gemma":[0.011023906,0.00085299124,0.0019590568,0.0050104316,0.0014895004,0.005817295,0.0045179436,0.0024865225,0.008392703],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013375178,0.000105731306,0.00058448844,0.0012105859,0.00017609648,0.00056396995,0.00057876424,0.051277284,0.0031448582,0.30315056,0.056016833,0.583057],"study_design_scores_gemma":[0.00003676032,0.000035704532,0.00035795482,0.00034797285,0.000076734796,0.00028284598,0.00020037954,0.13689947,0.0033043695,0.71189964,0.14650686,0.000051315918],"about_ca_topic_score_codex":0.004998807,"about_ca_topic_score_gemma":0.006707516,"teacher_disagreement_score":0.019071246,"about_ca_system_score_codex":0.0016809962,"about_ca_system_score_gemma":0.0017634588,"threshold_uncertainty_score":0.06379968},"labels":[],"label_agreement":null},{"id":"W3087976089","doi":"","title":"Proceedings 13th Joint ISO-ACL Workshop on Interoperable Semantic Annotation (ISA-13)","year":2017,"lang":"en","type":"article","venue":"Data Archiving and Networked Services (DANS)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Schweizerischer Nationalfonds zur Förderung der Wissenschaftlichen Forschung; Agence Nationale de la Recherche; European Regional Development Fund; Deutsche Forschungsgemeinschaft; Trinity College Dublin; City University of Hong Kong; Science Foundation Ireland; Ministère de l'Économie, de la Science et de l'Innovation - Québec; National Natural Science Foundation of China; Canarie","keywords":"Interoperability; Computer science; Joint (building); Annotation; Semantic annotation; Semantic interoperability; Information retrieval; World Wide Web; Artificial intelligence; Engineering","score_opus":0.030897701603482527,"score_gpt":0.2831676811640467,"score_spread":0.2522699795605642,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3087976089","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013769861,0.010656199,0.81169635,0.02547556,0.030027622,0.0015183084,0.008044944,0.015060906,0.08375029],"genre_scores_gemma":[0.070438944,0.011855367,0.6428466,0.0074485564,0.0049986956,0.0018711699,0.07058137,0.0129891755,0.17697012],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9873825,0.0056885453,0.0014689219,0.0015007394,0.0029042112,0.001054974],"domain_scores_gemma":[0.9781123,0.004388082,0.00052993157,0.0058449483,0.008776475,0.0023482684],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.031298887,0.0022552651,0.0031414153,0.0045469296,0.0037629497,0.014950681,0.0055375905,0.005122142,0.04429139],"category_scores_gemma":[0.024554225,0.0018745575,0.0035557495,0.0040620933,0.0041477736,0.016161552,0.013436818,0.008697625,0.02260283],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009961461,0.001531013,0.0018502872,0.0011320474,0.00027938193,0.0006931056,0.002667014,0.0033845687,0.0067419913,0.093504384,0.5847236,0.3024965],"study_design_scores_gemma":[0.000108099885,0.00014933782,0.0014870263,0.0011222925,0.00023918481,0.0005000123,0.0015809995,0.014940622,0.009467881,0.077443875,0.89283395,0.00012684404],"about_ca_topic_score_codex":0.025831686,"about_ca_topic_score_gemma":0.025542242,"teacher_disagreement_score":0.04429139,"about_ca_system_score_codex":0.0039422316,"about_ca_system_score_gemma":0.013912606,"threshold_uncertainty_score":0.16552633},"labels":[],"label_agreement":null},{"id":"W3088355435","doi":"","title":"PACTE: A colloaborative platform for textual annotation","year":2017,"lang":"en","type":"article","venue":"Meeting of the Association for Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; École de Technologie Supérieure","funders":"","keywords":"Annotation; Computer science; Natural language processing; Artificial intelligence; Information retrieval; World Wide Web","score_opus":0.021316635145724953,"score_gpt":0.31386733643635356,"score_spread":0.2925507012906286,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3088355435","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004116265,0.0003063716,0.7559756,0.0010190788,0.0009929384,0.00088864705,0.037135758,0.17682534,0.022740033],"genre_scores_gemma":[0.061878968,0.00053053506,0.6996862,0.0011725603,0.00064587785,0.0030739687,0.14652331,0.03994776,0.0465408],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9959992,0.0012128222,0.00048164767,0.00091347424,0.0011663664,0.00022650884],"domain_scores_gemma":[0.98639363,0.005316324,0.000699389,0.0041597863,0.0025709353,0.000859965],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004145683,0.002620617,0.0013358493,0.0056023807,0.0030371188,0.004033516,0.0035881395,0.0025858618,0.07197965],"category_scores_gemma":[0.017119762,0.0016698646,0.0014128542,0.004223833,0.001462979,0.010915282,0.011706512,0.0034491317,0.05544557],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017932795,0.0003818393,0.0017097377,0.0025522243,0.00019177621,0.0016269467,0.0043280134,0.00276745,0.049995255,0.082830794,0.6072373,0.24458542],"study_design_scores_gemma":[0.0001719643,0.00012944128,0.001226023,0.00026888345,0.000097540025,0.00066037383,0.0010595107,0.050408784,0.032321356,0.054008417,0.8594098,0.00023784433],"about_ca_topic_score_codex":0.005808033,"about_ca_topic_score_gemma":0.008825713,"teacher_disagreement_score":0.07197965,"about_ca_system_score_codex":0.00097249245,"about_ca_system_score_gemma":0.0035592725,"threshold_uncertainty_score":0.24079591},"labels":[],"label_agreement":null},{"id":"W3088525328","doi":"10.1109/compe49325.2020.9200135","title":"A Machine Learning Approach to Anaphora Resolution in Nepali Language","year":2020,"lang":"en","type":"article","venue":"2020 International Conference on Computational Performance Evaluation (ComPE)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University of Edmonton; University of Alberta","funders":"","keywords":"Nepali; Computer science; Anaphora (linguistics); Natural language processing; Chunking (psychology); Artificial intelligence; Resolution (logic); Machine learning; Linguistics","score_opus":0.06216091291302274,"score_gpt":0.33512641266006815,"score_spread":0.2729654997470454,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3088525328","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.045178004,0.0012010554,0.94294906,0.0016357283,0.000066961315,0.00020722067,0.00041390792,0.0010943025,0.007253798],"genre_scores_gemma":[0.40502933,0.0008372646,0.58524823,0.0005804016,0.00018148834,0.0003728401,0.0009675229,0.00008981308,0.006693165],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99850625,0.0006475526,0.00010118429,0.00037878475,0.00030264654,0.00006349712],"domain_scores_gemma":[0.9982425,0.0011106543,0.0001224865,0.00019763634,0.00029124535,0.000035534],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018397833,0.00040869828,0.0006343528,0.0019262936,0.0012590362,0.001770101,0.0014405869,0.0009314341,0.0021223358],"category_scores_gemma":[0.0051069316,0.0002827195,0.00064724585,0.0016726746,0.00067028595,0.002575635,0.0010317847,0.0017403308,0.00079393183],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022916899,0.0006583819,0.009127326,0.0005580681,0.00020948001,0.0008378888,0.0015594239,0.10850915,0.016938817,0.06290727,0.0061599994,0.79230505],"study_design_scores_gemma":[0.000023558281,0.00015226487,0.003663126,0.000060577706,0.00005838176,0.0006847997,0.0003097725,0.9049523,0.01887444,0.054040976,0.017129824,0.00004998888],"about_ca_topic_score_codex":0.002615395,"about_ca_topic_score_gemma":0.0030287434,"teacher_disagreement_score":0.002615395,"about_ca_system_score_codex":0.0009738025,"about_ca_system_score_gemma":0.0007729016,"threshold_uncertainty_score":0.009729862},"labels":[],"label_agreement":null},{"id":"W3089254180","doi":"","title":"Using the Nunavut Hansard Data for Experiments in Morphological Analysis and Machine Translation","year":2018,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Morpheme; Natural language processing; Computer science; Machine translation; Artificial intelligence; Linguistics; Speech recognition","score_opus":0.16138758710692508,"score_gpt":0.3971125203453311,"score_spread":0.23572493323840604,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3089254180","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7125325,0.0016068539,0.02698346,0.0011522517,0.00054367125,0.0022559543,0.12465399,0.004315282,0.12595594],"genre_scores_gemma":[0.6084531,0.0007759263,0.11581261,0.00047169946,0.00010314955,0.002435369,0.22545214,0.0019976099,0.044498425],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9974497,0.00075805903,0.00021758227,0.00052087073,0.00085002853,0.00020369636],"domain_scores_gemma":[0.9904231,0.0032649068,0.00027925466,0.002160352,0.0035495698,0.00032289885],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013158758,0.0006339212,0.0005688445,0.003347702,0.0044553326,0.0016524815,0.0012266722,0.00082983234,0.012003578],"category_scores_gemma":[0.0070453407,0.0003678621,0.0005878088,0.006253726,0.001383477,0.0010402015,0.0012323316,0.0011847601,0.005355254],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0049344054,0.0031066223,0.07057544,0.0024655738,0.0002982511,0.0033879648,0.024933964,0.014135017,0.102614015,0.033022013,0.20198865,0.5385381],"study_design_scores_gemma":[0.00043211487,0.00053270144,0.22930282,0.00027551368,0.00018826802,0.0016506042,0.013632633,0.020101277,0.106536075,0.0043786718,0.62256163,0.0004077144],"about_ca_topic_score_codex":0.496925,"about_ca_topic_score_gemma":0.67778164,"teacher_disagreement_score":0.503075,"about_ca_system_score_codex":0.0044439808,"about_ca_system_score_gemma":0.0042320876,"threshold_uncertainty_score":0.9880651},"labels":[],"label_agreement":null},{"id":"W3089480095","doi":"","title":"M is for Maple: A Canadian Alphabet","year":2002,"lang":"en","type":"article","venue":"Journal of Childhood Studies","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Maple; Alphabet; Mathematics; Biology; Linguistics; Botany; Philosophy","score_opus":0.03545884682221414,"score_gpt":0.29953509293446395,"score_spread":0.26407624611224984,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3089480095","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0039069504,0.001629697,0.019354245,0.019758264,0.012459456,0.00020285562,0.02825412,0.012908733,0.9015257],"genre_scores_gemma":[0.106330074,0.0030787203,0.04871911,0.00448602,0.0015442948,0.0002750118,0.017406587,0.0073466543,0.81081355],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9990658,0.00010349755,0.00006322843,0.00016917157,0.0003426391,0.00025567494],"domain_scores_gemma":[0.9972881,0.0002892692,0.00007908036,0.00033965622,0.0016356659,0.0003682613],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00067490217,0.0009871926,0.0008607108,0.0030634983,0.0073235203,0.0064876974,0.0016920753,0.0014889062,0.24722295],"category_scores_gemma":[0.00666393,0.00044246134,0.0004801576,0.005568964,0.0020266618,0.0036271408,0.0019798393,0.0028646814,0.0656298],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010227591,0.00000924144,0.00020988655,0.00010539698,0.0000040303885,0.00012977979,0.0005264454,0.0001680677,0.0004208781,0.09211643,0.8551817,0.051025894],"study_design_scores_gemma":[0.0000038002895,0.0000026492826,0.00013561416,0.000027239765,0.000002216717,0.00003975836,0.00018312094,0.000095266725,0.00016098299,0.004616773,0.9947213,0.000011306338],"about_ca_topic_score_codex":0.5789286,"about_ca_topic_score_gemma":0.6780104,"teacher_disagreement_score":0.4210714,"about_ca_system_score_codex":0.011160794,"about_ca_system_score_gemma":0.029056778,"threshold_uncertainty_score":0.84710234},"labels":[],"label_agreement":null},{"id":"W3091525649","doi":"10.18653/v1/2020.aacl-main.67","title":"Neural RST-based Evaluation of Discourse Coherence","year":2020,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Coherence (philosophical gambling strategy); Computer science; Parsing; Benchmark (surveying); Artificial intelligence; Natural language processing; Rhetorical question; State (computer science); Algorithm; Mathematics; Linguistics; Statistics","score_opus":0.059396477788238373,"score_gpt":0.384690862164279,"score_spread":0.32529438437604064,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3091525649","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.76973045,0.013506587,0.17940648,0.0014493609,0.0005629851,0.00036719118,0.0072882413,0.0057935193,0.02189521],"genre_scores_gemma":[0.9642547,0.0004002136,0.027140824,0.000047802438,0.00012003615,0.000090093454,0.00398744,0.00013054615,0.0038283693],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985253,0.00051407784,0.00012163383,0.00039378894,0.0003039818,0.00014112516],"domain_scores_gemma":[0.993972,0.0036734443,0.000386314,0.0003784356,0.0012955932,0.0002942624],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025250388,0.00080964016,0.00087225944,0.0026917835,0.0005258772,0.0014700118,0.0009886854,0.001181973,0.0046438775],"category_scores_gemma":[0.010710413,0.00023794454,0.00035142922,0.0011075721,0.00029997135,0.001981701,0.0015367476,0.0009290937,0.0013617948],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00667082,0.00060527335,0.022498153,0.0008098811,0.0006317667,0.0002569575,0.0006153282,0.055414572,0.03428351,0.0043689916,0.013795553,0.86004925],"study_design_scores_gemma":[0.00014429535,0.00066202006,0.025160445,0.0000998973,0.00020760208,0.000129879,0.0003501044,0.94688815,0.018197471,0.004722641,0.0033844004,0.000053090076],"about_ca_topic_score_codex":0.0057775257,"about_ca_topic_score_gemma":0.009545824,"teacher_disagreement_score":0.0057775257,"about_ca_system_score_codex":0.0009660439,"about_ca_system_score_gemma":0.0007087346,"threshold_uncertainty_score":0.015535355},"labels":[],"label_agreement":null},{"id":"W3092034297","doi":"10.1109/access.2021.3063715","title":"Machine Translation of Mathematical Text","year":2021,"lang":"en","type":"preprint","venue":"IEEE Access","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Transformer; Backup; Artificial intelligence; Perplexity; Natural language processing; BLEU; Machine translation; Polymath; Glossary; Parsing; Language model; Database; Linguistics","score_opus":0.04082429320500572,"score_gpt":0.3449060266599361,"score_spread":0.3040817334549304,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3092034297","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11364682,0.0011173743,0.71897817,0.0012456073,0.0009513414,0.0006242626,0.023311483,0.078079484,0.06204552],"genre_scores_gemma":[0.33109003,0.0007987119,0.57271045,0.00037671492,0.00027361856,0.00043111644,0.061793305,0.00814528,0.024380786],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99863726,0.0003760398,0.00014338989,0.00039755367,0.00034972897,0.00009603368],"domain_scores_gemma":[0.9968622,0.0011255441,0.00012002338,0.0007496556,0.001071699,0.000070951566],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00094863126,0.0011070258,0.0007233535,0.0014599271,0.0008531132,0.001665912,0.0010175675,0.0005151898,0.023571664],"category_scores_gemma":[0.006913614,0.00046985463,0.0008883435,0.0013853908,0.0005001871,0.0022803608,0.0015778261,0.0012461339,0.016399147],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00044866564,0.00033238088,0.0019992574,0.0018437995,0.000115871735,0.0012538936,0.0013461258,0.015276184,0.09396354,0.03561976,0.108861595,0.73893887],"study_design_scores_gemma":[0.0002122464,0.00042040306,0.005236865,0.0002336711,0.00012313003,0.0022959851,0.0008249119,0.21859807,0.26425278,0.036353927,0.47127187,0.00017613289],"about_ca_topic_score_codex":0.001581539,"about_ca_topic_score_gemma":0.002183588,"teacher_disagreement_score":0.023571664,"about_ca_system_score_codex":0.00082741975,"about_ca_system_score_gemma":0.0012788317,"threshold_uncertainty_score":0.0788551},"labels":[],"label_agreement":null},{"id":"W3092614733","doi":"10.18653/v1/2020.findings-emnlp.195","title":"Participatory Research for Low-resourced Machine Translation: A Case Study in African Languages","year":2020,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Machine translation; Citizen journalism; Scalability; Participatory action research; Process (computing); Artificial intelligence; Focus (optics); Diversity (politics); Data science; Translation (biology); Code (set theory); Languages of Africa; Natural language processing; World Wide Web; Political science; Sociology; Programming language; Linguistics; Set (abstract data type); Database","score_opus":0.24803139057706,"score_gpt":0.4603048529427606,"score_spread":0.21227346236570058,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3092614733","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.920128,0.002159058,0.011619454,0.016207382,0.0002950472,0.0009802581,0.00005813493,0.000061568484,0.048491098],"genre_scores_gemma":[0.98415434,0.0008700927,0.0045715356,0.0013048386,0.000035551533,0.0005701807,0.000022252863,0.00004737016,0.008423788],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9714624,0.02496039,0.00034286673,0.00068942166,0.0007831285,0.0017617417],"domain_scores_gemma":[0.96753335,0.026585428,0.0011601837,0.0010550539,0.00110794,0.0025580418],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02182882,0.0007642119,0.0006557404,0.0012150013,0.022856941,0.0050177467,0.0019527188,0.003835496,0.007927246],"category_scores_gemma":[0.025644382,0.00058485725,0.0006381001,0.0023049552,0.009396842,0.005455649,0.008668545,0.0036119316,0.00121711],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014864508,0.00040153813,0.0033980254,0.00041699,0.000009911547,0.010754401,0.93765926,0.00022307757,0.0017196698,0.010757391,0.0031730218,0.03133813],"study_design_scores_gemma":[0.00004136463,0.0002346505,0.0018154422,0.00054889615,0.000010279752,0.0021190832,0.9418405,0.0004634225,0.00073554413,0.0042257705,0.047934912,0.000030109686],"about_ca_topic_score_codex":0.003454702,"about_ca_topic_score_gemma":0.008150685,"teacher_disagreement_score":0.022856941,"about_ca_system_score_codex":0.0032287117,"about_ca_system_score_gemma":0.00796756,"threshold_uncertainty_score":0.11544317},"labels":[],"label_agreement":null},{"id":"W3093808828","doi":"10.18653/v1/2021.emnlp-main.740","title":"BARThez: a Skilled Pretrained French Sequence-to-Sequence Model","year":2021,"lang":"en","type":"preprint","venue":"Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Centre National de la Recherche Scientifique","keywords":"Automatic summarization; Discriminative model; Computer science; Generative grammar; Benchmark (surveying); Transfer of learning; Artificial intelligence; Natural language processing; Sequence (biology); Code (set theory); Generative model; Field (mathematics); Language model; Machine learning; Programming language; Cartography","score_opus":0.08687262609204593,"score_gpt":0.4277476868274459,"score_spread":0.34087506073539997,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3093808828","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09347856,0.0027205902,0.7354111,0.0026084161,0.0012499756,0.00053853326,0.023181658,0.10772115,0.033090025],"genre_scores_gemma":[0.5475124,0.001001092,0.31223044,0.0024439795,0.0004040922,0.0011211239,0.068813704,0.005630249,0.060842942],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9996749,0.00007127652,0.000012151337,0.00013981029,0.00004663186,0.0000552225],"domain_scores_gemma":[0.99946266,0.00027585492,0.000020964382,0.00008421552,0.00012565273,0.000030587067],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006625992,0.0017134658,0.0006130204,0.00080256414,0.00069071655,0.001179443,0.0024354593,0.001802896,0.013419496],"category_scores_gemma":[0.0022681626,0.0006697215,0.0013425185,0.0006744096,0.0005022141,0.0015283347,0.0009989153,0.0027352923,0.0066099893],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00063182216,0.00026011962,0.0032881165,0.00033551216,0.00028984115,0.00053131924,0.00026649318,0.5248369,0.013813981,0.016055811,0.13977604,0.29991406],"study_design_scores_gemma":[0.000041738884,0.000072207105,0.00046321974,0.000021935062,0.000025824977,0.00009028849,0.00004054637,0.9770685,0.0028330248,0.004732363,0.014582627,0.00002770729],"about_ca_topic_score_codex":0.045783427,"about_ca_topic_score_gemma":0.08969801,"teacher_disagreement_score":0.045783427,"about_ca_system_score_codex":0.0013445319,"about_ca_system_score_gemma":0.0018733605,"threshold_uncertainty_score":0.091033876},"labels":[],"label_agreement":null},{"id":"W3094618215","doi":"10.4018/978-1-7998-3468-7.ch008","title":"A State-of-the-Art Review of Nigerian Languages Natural Language Processing Research","year":2020,"lang":"en","type":"review","venue":"Advances in IT standards and standardization research (AISSR) book series/Advances in IT standards and standardization research series","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Yoruba; Hausa; Computer science; Igbo; Languages of Africa; Resource (disambiguation); State (computer science); Natural language processing; Representation (politics); Artificial intelligence; Linguistics; Programming language; Political science","score_opus":0.02938039977579118,"score_gpt":0.46767479431809894,"score_spread":0.4382943945423078,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3094618215","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00016177967,0.9978676,0.00013543984,0.000260158,0.00017946887,0.000007796233,0.00003599833,0.0000075566854,0.0013440802],"genre_scores_gemma":[0.0006197398,0.9986078,0.00022276519,0.0001179384,0.00007668904,0.000007224251,0.00003743793,0.0000015919627,0.00030881964],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9995123,0.000077531215,0.00010800605,0.00008245822,0.00019064994,0.000029108161],"domain_scores_gemma":[0.9983943,0.0009641864,0.00017212839,0.000027917024,0.00039064747,0.000050801933],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009705824,0.00069484994,0.0011241785,0.005382954,0.00044047085,0.0012866717,0.0007175141,0.00078553346,0.0045372704],"category_scores_gemma":[0.0021800469,0.0004740637,0.0006465819,0.006201242,0.0004568432,0.0020594355,0.00057045545,0.0009905513,0.0016302632],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00004714607,0.000051435018,0.00033686936,0.06692589,0.00007960556,0.00015246219,0.00021679996,0.00023905742,0.0013352181,0.0033249238,0.030185457,0.8971053],"study_design_scores_gemma":[0.000005521496,0.000062866195,0.0017214622,0.021608038,0.00020403933,0.0005971667,0.0001880152,0.00007751302,0.0004545358,0.0011015267,0.97396016,0.000019131807],"about_ca_topic_score_codex":0.0025619734,"about_ca_topic_score_gemma":0.0051710736,"teacher_disagreement_score":0.005382954,"about_ca_system_score_codex":0.0007040953,"about_ca_system_score_gemma":0.0026474418,"threshold_uncertainty_score":0.01517874},"labels":[],"label_agreement":null},{"id":"W3094991512","doi":"10.1002/9781118788516.sem132","title":"Matrix and Embedded Presuppositions","year":2020,"lang":"en","type":"other","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Presupposition; Sentence; Generalization; Projection (relational algebra); Matrix (chemical analysis); Filter (signal processing); Linguistics; Mathematics; Epistemology; Computer science; Philosophy; Algorithm; Computer vision","score_opus":0.00697181740032761,"score_gpt":0.27522945683016503,"score_spread":0.26825763942983744,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3094991512","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11647597,0.003025044,0.6051759,0.004985478,0.00030621397,0.00010287713,0.0003982586,0.0010952326,0.26843497],"genre_scores_gemma":[0.9178347,0.00084961904,0.06420278,0.00054759934,0.00014195242,0.000060164704,0.000248225,0.00026481008,0.01585017],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99783486,0.0007729255,0.00012655853,0.0005476975,0.0005524492,0.00016552705],"domain_scores_gemma":[0.996405,0.0018000487,0.0003348274,0.00077673054,0.00052186375,0.00016145229],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019789792,0.0005352969,0.00037877384,0.0012646978,0.001415928,0.0041940766,0.0014214612,0.0017265849,0.01693608],"category_scores_gemma":[0.009820902,0.00046015173,0.0007375256,0.0010435669,0.007184194,0.014878367,0.0036212313,0.002460816,0.001760924],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005385949,0.000011615762,0.00033741377,0.00007200917,0.0000144486075,0.0001393342,0.0015081151,0.00040989305,0.0017332741,0.9800265,0.00053128164,0.015162351],"study_design_scores_gemma":[0.000018866103,0.000033744367,0.0012398219,0.0000638785,0.000028733348,0.00029640453,0.00095459126,0.0059847557,0.0050093625,0.9666501,0.019693052,0.000026767666],"about_ca_topic_score_codex":0.002466888,"about_ca_topic_score_gemma":0.0022012964,"teacher_disagreement_score":0.01693608,"about_ca_system_score_codex":0.0017866367,"about_ca_system_score_gemma":0.0012182663,"threshold_uncertainty_score":0.056656897},"labels":[],"label_agreement":null},{"id":"W3096839736","doi":"10.5430/elr.v9n4p1","title":"Machine Translation in Foreign Language Learning Classroom-Learners’ Indiscriminate Use or Instructors’ Discriminate Stance","year":2020,"lang":"en","type":"article","venue":"English Linguistics Research","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Dilemma; Mathematics education; Curriculum; Foreign language; Computer science; Pedagogy; Psychology","score_opus":0.1012692666463389,"score_gpt":0.3662089561596121,"score_spread":0.2649396895132732,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3096839736","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5606032,0.034854863,0.044373203,0.06969591,0.0018314722,0.00010578657,0.0000753169,0.00045525644,0.28800505],"genre_scores_gemma":[0.9713605,0.005220396,0.005647246,0.0028346973,0.000181008,0.00004063313,0.000022641532,0.000084247455,0.014608578],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99492764,0.003063304,0.0002624141,0.0005091966,0.0009266172,0.0003107392],"domain_scores_gemma":[0.9926224,0.0047721444,0.0006156068,0.00049734325,0.0009932419,0.0004993091],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0043515307,0.0002270057,0.00028154513,0.00076223287,0.0017986319,0.0059245024,0.00045630545,0.0013774375,0.002867411],"category_scores_gemma":[0.01074709,0.00014439087,0.00020216704,0.00056253216,0.0037227005,0.0044393716,0.0025259664,0.0020867526,0.0012141562],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000085322696,0.00035163155,0.018750275,0.0011590493,0.000026697133,0.0015847425,0.27982044,0.00027652687,0.011450585,0.17943414,0.015097956,0.49196267],"study_design_scores_gemma":[0.000034146757,0.00037339583,0.018748054,0.0023134882,0.000078269964,0.0037196283,0.20543668,0.001998874,0.024821255,0.086340666,0.6560503,0.0000851198],"about_ca_topic_score_codex":0.00050465955,"about_ca_topic_score_gemma":0.00090194895,"teacher_disagreement_score":0.0059245024,"about_ca_system_score_codex":0.001245103,"about_ca_system_score_gemma":0.0022551795,"threshold_uncertainty_score":0.023013353},"labels":[],"label_agreement":null},{"id":"W3098210275","doi":"10.37213/cjal.2020.30649","title":"Investigating the Alignment Between the CELPIP-General Reading Test and the Canadian Language Benchmarks: A Content Validation Study","year":2020,"lang":"en","type":"article","venue":"Canadian Journal of Applied Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Test (biology); Computer science; Content validity; Language assessment; Language proficiency; Scale (ratio); Test validity; Index (typography); Psychology; Natural language processing; Mathematics education; Psychometrics; World Wide Web","score_opus":0.030153507772761788,"score_gpt":0.25192937680072625,"score_spread":0.22177586902796445,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3098210275","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.96286255,0.00033248652,0.0061602416,0.0011473855,0.00010186034,0.0021176762,0.0020163916,0.0000997181,0.025161779],"genre_scores_gemma":[0.9809761,0.00018481708,0.011483875,0.0004033409,0.000019686955,0.0022939937,0.002159877,0.00008038702,0.0023978462],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.955479,0.008976103,0.0030193257,0.0036730508,0.026616799,0.0022356194],"domain_scores_gemma":[0.79247,0.04913483,0.014314423,0.01138501,0.12768716,0.005008542],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04761833,0.0005368569,0.0007330226,0.00648854,0.005743155,0.0044433945,0.0037598175,0.00093552173,0.0018142755],"category_scores_gemma":[0.20382524,0.00052874465,0.00071980315,0.009112138,0.005678259,0.00198396,0.004404314,0.002296749,0.00033185363],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007341724,0.0011496278,0.68821627,0.00080570386,0.00027662204,0.00041547095,0.09544756,0.0021852064,0.0033062994,0.015117149,0.010036628,0.18230928],"study_design_scores_gemma":[0.00009474813,0.00040025314,0.94736403,0.00035633284,0.00011220436,0.00008791162,0.024933923,0.002551902,0.0028832164,0.0014184217,0.019670174,0.00012696051],"about_ca_topic_score_codex":0.86266685,"about_ca_topic_score_gemma":0.8894689,"teacher_disagreement_score":0.13733315,"about_ca_system_score_codex":0.05214524,"about_ca_system_score_gemma":0.103319995,"threshold_uncertainty_score":0.37834197},"labels":[],"label_agreement":null},{"id":"W3099330995","doi":"10.18653/v1/2020.emnlp-main.332","title":"Improving Word Sense Disambiguation with Translations","year":2020,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Alberta Machine Intelligence Institute; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Word-sense disambiguation; Computer science; Natural language processing; Word (group theory); SemEval; Artificial intelligence; Sense (electronics); Linguistics; WordNet; Engineering; Philosophy","score_opus":0.01577025542673025,"score_gpt":0.24118658396042925,"score_spread":0.225416328533699,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3099330995","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09415999,0.0020124104,0.8829782,0.000415909,0.00040240263,0.00021860775,0.0008390879,0.013765969,0.0052074348],"genre_scores_gemma":[0.24765591,0.0009186256,0.74331635,0.00031011383,0.00012127669,0.00011898385,0.0032190997,0.0010731769,0.0032664223],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99616385,0.0013282411,0.000515136,0.0010745515,0.0007607468,0.00015751628],"domain_scores_gemma":[0.996148,0.0013382811,0.0003609166,0.0009060515,0.0011445376,0.00010231802],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031666516,0.0021551442,0.0017322972,0.0046951375,0.0014660347,0.002381383,0.0014079073,0.001016292,0.0033255415],"category_scores_gemma":[0.009990274,0.00078709127,0.0012389472,0.0051769004,0.0010223604,0.004179119,0.0034081764,0.0012992604,0.004542296],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005858816,0.0003861733,0.006337281,0.001109574,0.00037713858,0.00081413984,0.001369736,0.037274778,0.1123183,0.01671505,0.015154918,0.807557],"study_design_scores_gemma":[0.00026503264,0.00048005013,0.005225324,0.0001639842,0.00050440454,0.0015176946,0.0017326957,0.5987723,0.28309506,0.048253708,0.05975684,0.00023298706],"about_ca_topic_score_codex":0.0027272913,"about_ca_topic_score_gemma":0.004069931,"teacher_disagreement_score":0.0046951375,"about_ca_system_score_codex":0.00049844605,"about_ca_system_score_gemma":0.0018936907,"threshold_uncertainty_score":0.016746998},"labels":[],"label_agreement":null},{"id":"W3099658661","doi":"10.18653/v1/2020.emnlp-main.744","title":"An information theoretic view on selecting linguistic probes","year":2020,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute; University of Toronto","funders":"Canadian Institute for Advanced Research","keywords":"Classifier (UML); Artificial intelligence; Modulo; Computer science; Information gain; Natural language processing; Selection (genetic algorithm); Machine learning; Mathematics; Discrete mathematics","score_opus":0.010903307717429967,"score_gpt":0.26873315603708536,"score_spread":0.2578298483196554,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3099658661","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009387254,0.0007036564,0.97304416,0.005503797,0.00007179267,0.00012541765,0.00018431316,0.00024574558,0.010733921],"genre_scores_gemma":[0.63033,0.0014283581,0.35464847,0.004437899,0.0007863904,0.0016810794,0.00057919795,0.0003204754,0.0057880236],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.98432124,0.00887585,0.0006220834,0.0023342513,0.0032553892,0.0005911494],"domain_scores_gemma":[0.9120121,0.07341896,0.00402215,0.006200398,0.0033012147,0.001045284],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.018561082,0.0019656331,0.002191571,0.0052821557,0.0017778939,0.007207692,0.004801463,0.005640959,0.008002097],"category_scores_gemma":[0.092121735,0.0013879647,0.0018464009,0.0027991307,0.015156592,0.020184673,0.0055218916,0.005923197,0.0011609386],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002261905,0.00009168437,0.0010001793,0.00025403575,0.00009114821,0.000069895286,0.00045678535,0.019545743,0.0017258285,0.93713194,0.0019460813,0.03746047],"study_design_scores_gemma":[0.000051184885,0.00007828588,0.00020484971,0.000035570323,0.000023831257,0.00004521895,0.000064223255,0.0442394,0.00088420743,0.9530787,0.0012627185,0.0000318325],"about_ca_topic_score_codex":0.0012049209,"about_ca_topic_score_gemma":0.00072191114,"teacher_disagreement_score":0.018561082,"about_ca_system_score_codex":0.0036144245,"about_ca_system_score_gemma":0.0017856769,"threshold_uncertainty_score":0.09816152},"labels":[],"label_agreement":null},{"id":"W3099716694","doi":"10.18653/v1/2020.codi-1.17","title":"Coreference for Discourse Parsing: A Neural Approach","year":2020,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Coreference; Parsing; Computer science; Natural language processing; Resolver; Artificial intelligence; Resolution (logic); Parser combinator","score_opus":0.06118390513021559,"score_gpt":0.3126236598979384,"score_spread":0.2514397547677228,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3099716694","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.071061395,0.00095707894,0.90791196,0.0010846216,0.00012737403,0.0002191441,0.00070217287,0.0069924565,0.010943833],"genre_scores_gemma":[0.5329367,0.0004774423,0.45705232,0.00042738835,0.00013426151,0.0002591139,0.0011812181,0.00035764574,0.0071740113],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9982754,0.00080524274,0.00007057066,0.00048241013,0.00022076699,0.00014561113],"domain_scores_gemma":[0.99597836,0.0025208658,0.00020627683,0.00062354753,0.000579074,0.00009188069],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003524239,0.0011217924,0.0007061276,0.0016239639,0.00094722863,0.0020320949,0.002601554,0.0019306771,0.0069741216],"category_scores_gemma":[0.01027585,0.00060183334,0.00097113376,0.0011408731,0.00066394656,0.004514001,0.0017199862,0.002763159,0.0015784949],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00068812526,0.00042820408,0.0026121414,0.00041026482,0.00026474998,0.00020772399,0.00054849626,0.14687994,0.03518867,0.019950131,0.0059718774,0.7868496],"study_design_scores_gemma":[0.000026569136,0.000072865805,0.0011432242,0.000034574783,0.00009404524,0.000079639365,0.00010025012,0.9567615,0.017597996,0.021495825,0.0025650624,0.000028529732],"about_ca_topic_score_codex":0.006716192,"about_ca_topic_score_gemma":0.014149979,"teacher_disagreement_score":0.0069741216,"about_ca_system_score_codex":0.0015410028,"about_ca_system_score_gemma":0.0016370977,"threshold_uncertainty_score":0.023330808},"labels":[],"label_agreement":null},{"id":"W3099859597","doi":"10.18653/v1/2020.coling-main.515","title":"An Analysis of Dataset Overlap on Winograd-Style Tasks","year":2020,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research; Microsoft Research","keywords":"Schema (genetic algorithms); Computer science; Pronoun; Artificial intelligence; Natural language processing; Style (visual arts); Parsing; Language model; Machine learning; Linguistics","score_opus":0.027144826076210892,"score_gpt":0.32995580318585577,"score_spread":0.3028109771096449,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3099859597","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.94295746,0.004829251,0.010409439,0.0011393719,0.00033133,0.0001783726,0.025305754,0.002765909,0.012082994],"genre_scores_gemma":[0.80636895,0.0008169387,0.012782522,0.00047544326,0.0001442746,0.00048263418,0.17399575,0.0014396961,0.0034938338],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9808333,0.0057748063,0.0019813697,0.005237517,0.00478021,0.0013928143],"domain_scores_gemma":[0.9628924,0.020501023,0.0018405833,0.008768944,0.004528223,0.0014688089],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01142087,0.00113439,0.001531326,0.0032244024,0.0021553084,0.0032864315,0.0021373234,0.001537333,0.002559764],"category_scores_gemma":[0.0495263,0.0006084841,0.0013803279,0.005352527,0.0018662359,0.004036996,0.0056854864,0.0021397257,0.0017037148],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.007973743,0.0021203395,0.30777383,0.003201814,0.002722466,0.0028101725,0.0075248326,0.060255498,0.027104123,0.012664842,0.18922396,0.37662446],"study_design_scores_gemma":[0.00094331306,0.002813128,0.45988345,0.00077903265,0.0010043181,0.0074450076,0.010899494,0.22446012,0.052768495,0.023294572,0.21523778,0.00047130694],"about_ca_topic_score_codex":0.0061267694,"about_ca_topic_score_gemma":0.009543615,"teacher_disagreement_score":0.01142087,"about_ca_system_score_codex":0.0013434304,"about_ca_system_score_gemma":0.0015270195,"threshold_uncertainty_score":0.06040007},"labels":[],"label_agreement":null},{"id":"W3099902934","doi":"","title":"Low-Resource NMT: an Empirical Study on the Effect of Rich Morphological Word Segmentation on Inuktitut","year":2020,"lang":"en","type":"article","venue":"Conference of the Association for Machine Translation in the Americas","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Computer science; Word (group theory); Linguistics; Philosophy","score_opus":0.050208405779885217,"score_gpt":0.3467333131491154,"score_spread":0.29652490736923015,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3099902934","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99048257,0.00019464093,0.0013363323,0.00021035229,0.000021078693,0.000034661964,0.0006821826,0.00020619636,0.0068319463],"genre_scores_gemma":[0.9936613,0.00009265544,0.0023471075,0.0001394186,0.000021106971,0.000052684893,0.0016581848,0.00026408624,0.0017634972],"study_design_codex":"observational","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99527514,0.0028954102,0.00044233448,0.00061990187,0.00051609834,0.00025109007],"domain_scores_gemma":[0.86574113,0.11598708,0.00570965,0.0072385394,0.0037530675,0.0015705281],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00539381,0.00055330654,0.0006023943,0.0008633507,0.0017004326,0.0024652937,0.0015248823,0.0011925863,0.010870683],"category_scores_gemma":[0.08177237,0.00040629396,0.00046548434,0.0021425856,0.00095075724,0.0038661833,0.0021838872,0.0020901305,0.0030759063],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.021776436,0.008627536,0.5120561,0.0025514846,0.0007219719,0.004394057,0.026173063,0.022277994,0.04943322,0.012115089,0.019014414,0.3208586],"study_design_scores_gemma":[0.0004955472,0.003182684,0.83881366,0.00031831439,0.00079985696,0.0033577813,0.01869296,0.07317069,0.025119137,0.012562048,0.023223517,0.00026384057],"about_ca_topic_score_codex":0.012234595,"about_ca_topic_score_gemma":0.015539262,"teacher_disagreement_score":0.012234595,"about_ca_system_score_codex":0.0010513258,"about_ca_system_score_gemma":0.0012420736,"threshold_uncertainty_score":0.036366105},"labels":[],"label_agreement":null},{"id":"W3101745556","doi":"10.18653/v1/2020.wmt-1.42","title":"Translating Similar Languages: Role of Mutual Intelligibility in Multilingual Transformers","year":2020,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Compute Canada","keywords":"Computer science; Natural language processing; Jaccard index; Transformer; Artificial intelligence; Machine translation; Marathi; Intelligibility (philosophy); Speech recognition; Linguistics; Pattern recognition (psychology)","score_opus":0.023033928901316704,"score_gpt":0.3216688303297862,"score_spread":0.2986349014284695,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3101745556","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6354527,0.0015010752,0.336398,0.0012226787,0.00022401843,0.0001913369,0.0013127207,0.0055954237,0.01810202],"genre_scores_gemma":[0.9647027,0.00018800583,0.03202548,0.00008048877,0.00003778849,0.000036388523,0.0009387704,0.00049035903,0.001499957],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9953993,0.0022819075,0.00033402815,0.0010014225,0.0006786896,0.00030463046],"domain_scores_gemma":[0.98942965,0.005989201,0.00050844403,0.0024527016,0.0012610529,0.0003590391],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006329681,0.0014193685,0.0011221156,0.0015011537,0.0007932867,0.003732674,0.0011253769,0.0009860684,0.004248543],"category_scores_gemma":[0.019266361,0.00049984845,0.0010816504,0.0013979198,0.001384854,0.0063107633,0.004102091,0.0021323196,0.0017251315],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0047895242,0.00097512937,0.04143477,0.0016081927,0.0011363913,0.0018099115,0.0053640995,0.28108284,0.08671723,0.038729418,0.008964225,0.5273884],"study_design_scores_gemma":[0.00018397417,0.0011470923,0.007529361,0.00011686864,0.00052326464,0.0012373348,0.0014108608,0.8504403,0.077764556,0.04917173,0.010289395,0.00018520556],"about_ca_topic_score_codex":0.0024413166,"about_ca_topic_score_gemma":0.003909979,"teacher_disagreement_score":0.006329681,"about_ca_system_score_codex":0.00072661147,"about_ca_system_score_gemma":0.0011749913,"threshold_uncertainty_score":0.03347498},"labels":[],"label_agreement":null},{"id":"W3102213339","doi":"","title":"Getallen vermenigvuldigen in O(n log n) stappen.","year":2020,"lang":"nl","type":"article","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Mathematics","score_opus":0.015217006278382832,"score_gpt":0.2354261967228938,"score_spread":0.22020919044451098,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3102213339","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04442168,0.00698227,0.4410442,0.0141311735,0.008170748,0.0007390743,0.021593202,0.04107694,0.4218408],"genre_scores_gemma":[0.2346604,0.003423601,0.2796304,0.0039049846,0.0014661094,0.0011222938,0.03507217,0.013508309,0.42721173],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9977124,0.00037270458,0.00012731565,0.00059682253,0.0005721486,0.00061860186],"domain_scores_gemma":[0.99817526,0.0006198865,0.000062443825,0.0008122233,0.00016396513,0.00016624492],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007811174,0.0026002899,0.0019077492,0.0017346307,0.0015351672,0.0029075285,0.0020065801,0.0015453632,0.22098081],"category_scores_gemma":[0.00513966,0.0006882364,0.0017829558,0.0019200442,0.0009674942,0.005818509,0.0061382256,0.0021742212,0.12070623],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016777636,0.00039377337,0.001517864,0.0013907863,0.00017174751,0.0006312538,0.0003013432,0.0056073386,0.018835198,0.038178697,0.32327804,0.60801625],"study_design_scores_gemma":[0.00063475076,0.0003206124,0.002713648,0.00045113434,0.0002884367,0.001854384,0.00069300475,0.041919548,0.018677402,0.32127187,0.61104035,0.00013488313],"about_ca_topic_score_codex":0.0021359946,"about_ca_topic_score_gemma":0.007929111,"teacher_disagreement_score":0.22098081,"about_ca_system_score_codex":0.0015074693,"about_ca_system_score_gemma":0.0014866926,"threshold_uncertainty_score":0.7392545},"labels":[],"label_agreement":null},{"id":"W3103735191","doi":"10.18653/v1/2020.eval4nlp-1.11","title":"Grammaticality and Language Modelling","year":2020,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Grammaticality; Variable (mathematics); Computer science; Language model; Correlation; Natural language processing; Linguistics; Artificial intelligence; Cola (plant); Point (geometry); Psychology; Cognitive psychology; Grammar; Mathematics","score_opus":0.02606263882137646,"score_gpt":0.2678293875160296,"score_spread":0.24176674869465312,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3103735191","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.033421576,0.020104941,0.7419737,0.02806487,0.0012987588,0.00014833252,0.0015339705,0.0015982038,0.17185564],"genre_scores_gemma":[0.77992904,0.009897912,0.16741295,0.0038568946,0.0021850306,0.00029240167,0.0023935845,0.0013569079,0.032675233],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9964179,0.0017576197,0.00021497186,0.0008539576,0.0005635338,0.00019202167],"domain_scores_gemma":[0.99477047,0.0033844258,0.00045793658,0.0007617293,0.0004738498,0.00015153384],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003684195,0.0010639512,0.0010221578,0.0028266527,0.0012458253,0.0063897404,0.0019283987,0.0025140655,0.01116491],"category_scores_gemma":[0.014885644,0.00045835812,0.0019491452,0.001554552,0.0093082795,0.008014613,0.0032815556,0.0035767897,0.004057859],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000018218247,0.000014002745,0.0009894397,0.00012248586,0.00004716107,0.00009905169,0.0010064191,0.0072610555,0.00033549333,0.9605257,0.002849657,0.026731346],"study_design_scores_gemma":[0.000005305561,0.000010999597,0.00032872963,0.00005396644,0.000009131649,0.00011895493,0.00009508122,0.0083393175,0.00011904697,0.9736192,0.017280199,0.000019986923],"about_ca_topic_score_codex":0.0040372415,"about_ca_topic_score_gemma":0.0017624375,"teacher_disagreement_score":0.01116491,"about_ca_system_score_codex":0.0023468693,"about_ca_system_score_gemma":0.0014110175,"threshold_uncertainty_score":0.037350297},"labels":[],"label_agreement":null},{"id":"W3105626986","doi":"","title":"A Dictionary Lookup Strategy for Translating of Discontinuous Phrases","year":2006,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Linguistics; Philosophy","score_opus":0.015021293070837783,"score_gpt":0.2736535588009394,"score_spread":0.2586322657301016,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3105626986","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0070180483,0.00016969674,0.9835392,0.00014623054,0.00010392434,0.00016615838,0.00049729215,0.0046535414,0.0037059642],"genre_scores_gemma":[0.058974314,0.00031090248,0.93219185,0.00012907058,0.000059986258,0.00020450962,0.001598153,0.0013474087,0.0051837866],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992035,0.00022943605,0.00012975058,0.00022686426,0.00015772338,0.000052796313],"domain_scores_gemma":[0.9984321,0.00051389675,0.00011239641,0.00055806804,0.00032912343,0.000054471944],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007288696,0.0009212672,0.0010254547,0.0012181748,0.0008853198,0.0021738955,0.0012297953,0.0010210162,0.012185279],"category_scores_gemma":[0.0032143248,0.0006526199,0.0006156394,0.0015609502,0.0010474379,0.0029337073,0.001785128,0.0013263432,0.009548331],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005164861,0.0001812768,0.00051726203,0.00088080537,0.00006768394,0.00082533265,0.0015310211,0.004186659,0.13517956,0.14437856,0.02228437,0.689451],"study_design_scores_gemma":[0.0003019048,0.0007958332,0.0009638421,0.00025405386,0.00016884695,0.0037197024,0.0019692995,0.17088102,0.33354843,0.19425917,0.29282588,0.00031211053],"about_ca_topic_score_codex":0.0005407195,"about_ca_topic_score_gemma":0.0010789154,"teacher_disagreement_score":0.012185279,"about_ca_system_score_codex":0.0004498055,"about_ca_system_score_gemma":0.0007888656,"threshold_uncertainty_score":0.040763795},"labels":[],"label_agreement":null},{"id":"W3105864207","doi":"10.18653/v1/2020.emnlp-main.71","title":"Word class flexibility: A deep contextualized approach","year":2020,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"St. Michael's Hospital; Vector Institute; University of Toronto","funders":"Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada","keywords":"Flexibility (engineering); Computer science; Class (philosophy); Noun; Verb; Word (group theory); Natural language processing; Linguistics; Artificial intelligence; Part of speech; Variation (astronomy); Linguistic typology; Typology; Phenomenon; Mathematics; Sociology","score_opus":0.04017475278434007,"score_gpt":0.286791765619195,"score_spread":0.24661701283485493,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3105864207","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14415298,0.00044431348,0.84776336,0.0007228094,0.000057184356,0.00009608294,0.00096613803,0.0008840136,0.00491314],"genre_scores_gemma":[0.81891984,0.00023249935,0.17787215,0.00017862048,0.00004792472,0.00019280666,0.0009057727,0.00021274974,0.0014376341],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99918205,0.00028972153,0.000049505135,0.00031245925,0.0000794489,0.00008678095],"domain_scores_gemma":[0.9977775,0.0009647254,0.00030595408,0.0006424934,0.0001966683,0.00011268676],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00084890216,0.0006436583,0.00059105933,0.001957016,0.00079123967,0.0021340663,0.0011452979,0.0009869457,0.003657434],"category_scores_gemma":[0.0047881496,0.00061382196,0.0010437707,0.001708274,0.0019986758,0.0048999763,0.0029084077,0.0019683929,0.00058012735],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00053231337,0.00031525863,0.036454096,0.0004760962,0.00030470156,0.0006988084,0.005590982,0.16426489,0.019055538,0.3731907,0.0047961324,0.39432052],"study_design_scores_gemma":[0.000017228194,0.00007570378,0.0053807185,0.000063604384,0.000058652135,0.00019436063,0.00084904215,0.51644564,0.0021035841,0.46777976,0.006983444,0.000048334903],"about_ca_topic_score_codex":0.0035203728,"about_ca_topic_score_gemma":0.006280232,"teacher_disagreement_score":0.003657434,"about_ca_system_score_codex":0.001031145,"about_ca_system_score_gemma":0.0007522237,"threshold_uncertainty_score":0.012235343},"labels":[],"label_agreement":null},{"id":"W3106545386","doi":"10.18653/v1/2020.inlg-1.3","title":"Generating Intelligible Plumitifs Descriptions: Use Case Application with Ethical Considerations","year":2020,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Université Laval","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Code (set theory); Jurisdiction; Identity (music); Simple (philosophy); Architecture; Source code; World Wide Web; Computer security; Law; Programming language; Political science; Epistemology","score_opus":0.06695531563910474,"score_gpt":0.31934327767771775,"score_spread":0.252387962038613,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3106545386","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2602225,0.00048526534,0.66433734,0.0031432423,0.00026520752,0.0024084598,0.005067759,0.027572494,0.03649772],"genre_scores_gemma":[0.4514556,0.0003010524,0.5244596,0.00042961258,0.000059300302,0.0008439553,0.0053747664,0.0028917864,0.014184314],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99396473,0.0038497525,0.00029668695,0.0004179923,0.0012805851,0.00019015202],"domain_scores_gemma":[0.9770329,0.018089117,0.00052349264,0.0020473802,0.0020679403,0.00023918033],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0054473733,0.00086281623,0.0003281031,0.0014601828,0.0008225733,0.0023522826,0.0014019588,0.0019368946,0.0113986125],"category_scores_gemma":[0.030338297,0.0004562069,0.00067624263,0.00081645907,0.0011229315,0.0023268855,0.0024132896,0.0012552771,0.003077217],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014864098,0.0014688811,0.011139254,0.0028006337,0.00016991586,0.01860281,0.030253747,0.07908739,0.056340817,0.09092096,0.09577276,0.6119563],"study_design_scores_gemma":[0.00054124446,0.00067611225,0.004144803,0.00062332663,0.00013762618,0.0056727948,0.009969622,0.4239782,0.1333188,0.058393624,0.36229956,0.00024433638],"about_ca_topic_score_codex":0.0034731687,"about_ca_topic_score_gemma":0.0035206303,"teacher_disagreement_score":0.0113986125,"about_ca_system_score_codex":0.0012011783,"about_ca_system_score_gemma":0.0010436882,"threshold_uncertainty_score":0.03813213},"labels":[],"label_agreement":null},{"id":"W3107236768","doi":"","title":"Analysis of correctness in adverb use in the Japanese composition support system Nutmeg","year":2020,"lang":"en","type":"article","venue":"Znanstvena založba Filozofske fakultete","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Tellabs (Canada)","funders":"","keywords":"Lexis; Linguistics; Adverb; Computer science; Natural language processing; Acronym; Syntax; Artificial intelligence; Correctness; Noun; Programming language","score_opus":0.01971027654275415,"score_gpt":0.2557873318361732,"score_spread":0.23607705529341905,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3107236768","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9913288,0.00022632131,0.0049501006,0.000065145294,0.000021275004,0.000045308916,0.00043672696,0.00041650506,0.0025099206],"genre_scores_gemma":[0.99046665,0.000092070426,0.0055485712,0.00003196794,0.000008329621,0.000036286587,0.001364312,0.00033531885,0.0021163318],"study_design_codex":"observational","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99696964,0.00067705306,0.00041618152,0.00053295953,0.0011758149,0.00022836446],"domain_scores_gemma":[0.9819763,0.012443274,0.0014822865,0.00088001316,0.0030265385,0.0001916507],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002744949,0.00032229206,0.00047236224,0.0013567196,0.0010150693,0.0015270291,0.00051078247,0.0005808436,0.0017110548],"category_scores_gemma":[0.01887911,0.00039360294,0.00029940618,0.0014853581,0.0008649457,0.0025793505,0.0010492272,0.0005985487,0.0007638828],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0032567193,0.00030752932,0.5992619,0.001161455,0.00013924406,0.0024616462,0.026186608,0.0061665676,0.11550613,0.0058077546,0.002695953,0.23704852],"study_design_scores_gemma":[0.000076457305,0.0011572067,0.73276836,0.00011174745,0.00034314804,0.0022510896,0.016543861,0.10066178,0.121557266,0.0039451905,0.020428348,0.0001555608],"about_ca_topic_score_codex":0.009227404,"about_ca_topic_score_gemma":0.012337674,"teacher_disagreement_score":0.009227404,"about_ca_system_score_codex":0.00067826867,"about_ca_system_score_gemma":0.0008541169,"threshold_uncertainty_score":0.018347383},"labels":[],"label_agreement":null},{"id":"W3108045241","doi":"10.54590/pop.2020.012","title":"How Can We Broaden and Diversify Humanities Knowledge Translation?","year":2020,"lang":"en","type":"article","venue":"Pop! Public Open Participatory","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Premise; Digital humanities; Work (physics); Sociology; Translation studies; Humanities; Knowledge management; Engineering ethics; Epistemology; Computer science; Linguistics; Philosophy; Engineering","score_opus":0.2632176122636743,"score_gpt":0.3482732232520639,"score_spread":0.0850556109883896,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3108045241","genre_codex":"commentary","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014429219,0.021035045,0.21388882,0.66129553,0.004149445,0.0006490341,0.00063430285,0.0033750967,0.08054359],"genre_scores_gemma":[0.26369464,0.030843474,0.5427487,0.12013716,0.006644912,0.0021641098,0.0027535553,0.003110024,0.027903408],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9474273,0.035985846,0.0024952227,0.0046152603,0.0060182977,0.0034580186],"domain_scores_gemma":[0.79524374,0.11219667,0.0069950274,0.052446086,0.024398405,0.008720027],"candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.11845643,0.0015683747,0.0021068316,0.01256111,0.007844743,0.02299248,0.0060453615,0.010190265,0.022026671],"category_scores_gemma":[0.1691662,0.0013070798,0.0017437929,0.010055742,0.021304201,0.08270998,0.04263854,0.013439909,0.01595683],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018199661,0.00032742872,0.0030388103,0.0021503537,0.00014511142,0.00029087055,0.018536849,0.0021625035,0.0026302258,0.4032828,0.05780662,0.50944644],"study_design_scores_gemma":[0.000059483395,0.00007670155,0.001406226,0.0026029889,0.00005244937,0.00027006285,0.018329214,0.001864278,0.0010631001,0.63025033,0.3439429,0.00008216101],"about_ca_topic_score_codex":0.0059212483,"about_ca_topic_score_gemma":0.0063170935,"teacher_disagreement_score":0.9770075,"about_ca_system_score_codex":0.005603542,"about_ca_system_score_gemma":0.021181298,"threshold_uncertainty_score":0.62646496},"labels":[],"label_agreement":null},{"id":"W3109404979","doi":"10.7202/1073654ar","title":"Meng, Ji and Oakes, Michael, eds. (2019): Advances in Empirical Translation Studies: Developing Translation Resources and Technologies. Cambridge: Cambridge University Press, 270 p.","year":2020,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Interpretation (philosophy); Translation studies; Translation (biology); Sociology; Library science; Linguistics; Philosophy; Computer science","score_opus":0.05646573460462809,"score_gpt":0.2953681354532795,"score_spread":0.23890240084865141,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3109404979","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00015081097,0.9728689,0.004525574,0.013571751,0.003597998,0.000020106054,0.00029907768,0.00009438688,0.0048713796],"genre_scores_gemma":[0.0030250982,0.9623431,0.01598667,0.0028157837,0.0037374366,0.00010567475,0.0006841298,0.00018567132,0.011116554],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9963632,0.0014887087,0.00050770986,0.00037160865,0.001127816,0.0001409174],"domain_scores_gemma":[0.9818143,0.013907083,0.0011019133,0.00053977943,0.0019158684,0.0007209562],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009758431,0.0023531197,0.0024253146,0.009936479,0.0013149775,0.007448373,0.0018780036,0.0041663097,0.020683976],"category_scores_gemma":[0.020448228,0.0025156108,0.001029312,0.01173935,0.0039382717,0.016315918,0.0026136504,0.005596067,0.014762676],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008587095,0.000020655583,0.00049857877,0.0037215296,0.00006071569,0.00009275075,0.0014515829,0.00020786571,0.00036199647,0.014561785,0.6011775,0.3777591],"study_design_scores_gemma":[0.000023713495,0.000028467814,0.002011647,0.005261049,0.00008669066,0.00041133477,0.00079571747,0.00023912151,0.0005130972,0.027178666,0.9633855,0.00006499422],"about_ca_topic_score_codex":0.008363736,"about_ca_topic_score_gemma":0.018782554,"teacher_disagreement_score":0.020683976,"about_ca_system_score_codex":0.0029192178,"about_ca_system_score_gemma":0.0063061183,"threshold_uncertainty_score":0.06919479},"labels":[],"label_agreement":null},{"id":"W3110442023","doi":"10.1121/1.5147766","title":"Developing and maintaining the <i>Phonological CorpusTools</i> software","year":2020,"lang":"en","type":"article","venue":"The Journal of the Acoustical Society of America","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Python (programming language); Software; Workflow; Graphical user interface; Cross-platform; Upload; Variety (cybernetics); Human–computer interaction; Software engineering; World Wide Web; Artificial intelligence; Programming language; Database","score_opus":0.0217773411865529,"score_gpt":0.2609649231940568,"score_spread":0.2391875820075039,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3110442023","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01049982,0.00014058292,0.47612262,0.00058825675,0.00046597244,0.0017738328,0.043640137,0.44559056,0.021178268],"genre_scores_gemma":[0.034592055,0.00020479305,0.6807138,0.0007852956,0.00016383598,0.0057992823,0.09894394,0.15410864,0.024688363],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9969823,0.0004020216,0.00042268215,0.0008353043,0.0011371584,0.00022065436],"domain_scores_gemma":[0.9883996,0.0036241885,0.0005778342,0.003086955,0.0036389665,0.00067244587],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005474599,0.0014739345,0.0012270468,0.0040687886,0.0015431598,0.0042177704,0.0045551755,0.0009833828,0.05466364],"category_scores_gemma":[0.019769814,0.002174514,0.0014232845,0.0027659761,0.0012026237,0.0049083345,0.005056236,0.0038701429,0.04038929],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005853854,0.00018543415,0.004415574,0.0018060533,0.00018812357,0.00065265235,0.002898447,0.0040745013,0.038220186,0.015392159,0.5144443,0.41713715],"study_design_scores_gemma":[0.00022887276,0.00014266164,0.008058201,0.0003972343,0.00009490933,0.00087994867,0.000780809,0.032976765,0.074151404,0.018854735,0.8629908,0.00044361013],"about_ca_topic_score_codex":0.0065250616,"about_ca_topic_score_gemma":0.0077798986,"teacher_disagreement_score":0.05466364,"about_ca_system_score_codex":0.0012141909,"about_ca_system_score_gemma":0.004146122,"threshold_uncertainty_score":0.18286812},"labels":[],"label_agreement":null},{"id":"W3110874656","doi":"","title":"Methodological Tools for Linguistic Description and Typology.","year":2019,"lang":"en","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Social Sciences and Humanities Research Council of Canada; Centre National de la Recherche Scientifique; Max Planck Instituut voor Psycholinguïstiek; Fondation Fyssen; Russian Foundation for Basic Research; Agence Nationale de la Recherche","keywords":"Typology; Linguistics; Linguistic typology; Computer science; Sociology; Philosophy; Anthropology","score_opus":0.08937190619513008,"score_gpt":0.32150447098380924,"score_spread":0.23213256478867916,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3110874656","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0012595826,0.0017195352,0.9681427,0.0037756197,0.0003930003,0.00040451763,0.0007689995,0.0008246644,0.022711378],"genre_scores_gemma":[0.100281894,0.0023434586,0.8767534,0.0010376596,0.00091911125,0.0038180123,0.0038594366,0.0009576847,0.010029394],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9414349,0.042877592,0.0066270055,0.0041146087,0.004212031,0.00073380885],"domain_scores_gemma":[0.8579428,0.09112134,0.0055587473,0.029484568,0.014041399,0.0018511856],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.045809153,0.0019703158,0.0020900648,0.013707607,0.0044755237,0.016843358,0.007167775,0.0035211823,0.023567725],"category_scores_gemma":[0.11043086,0.0021172452,0.0029025327,0.009762952,0.019042458,0.031053785,0.01106216,0.0067633768,0.0066476436],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000014190527,0.000027076934,0.00025681333,0.00035212009,0.00004086998,0.000082454986,0.0018120343,0.00038587739,0.0001646396,0.96683973,0.006540264,0.023483887],"study_design_scores_gemma":[0.000011286868,0.000007728647,0.00013309972,0.00035666034,0.000019659507,0.00010115163,0.0015688546,0.0021424321,0.00023852942,0.9615346,0.033870336,0.000015706051],"about_ca_topic_score_codex":0.0038596035,"about_ca_topic_score_gemma":0.0032891768,"teacher_disagreement_score":0.045809153,"about_ca_system_score_codex":0.006334574,"about_ca_system_score_gemma":0.006755562,"threshold_uncertainty_score":0.2422648},"labels":[],"label_agreement":null},{"id":"W3110879416","doi":"","title":"An Ontological Representation of Sex and Gender Information","year":2020,"lang":"en","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Sherbrooke","funders":"","keywords":"Computer science; Focus (optics); Representation (politics); Information retrieval; Natural language processing; Data science; Politics; Political science","score_opus":0.028993220827394844,"score_gpt":0.2832517562046626,"score_spread":0.2542585353772677,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3110879416","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.039180864,0.0012892326,0.84804213,0.0069300644,0.0008662249,0.00027370718,0.014785475,0.0020324804,0.08659989],"genre_scores_gemma":[0.51845175,0.0021676426,0.42955834,0.001352019,0.0006176876,0.0003570527,0.019974438,0.000858621,0.026662473],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9983315,0.0003899767,0.00019311692,0.00043967564,0.00045173033,0.00019398982],"domain_scores_gemma":[0.9978897,0.00062110636,0.00014374132,0.0005245363,0.000628995,0.00019188396],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001983436,0.0006415955,0.00066264655,0.0046245595,0.0026312836,0.004492038,0.0016830539,0.0015510765,0.008865428],"category_scores_gemma":[0.00462292,0.00062917784,0.001725694,0.005026911,0.0020170473,0.010911098,0.0028749986,0.0022556195,0.0019551243],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008061308,0.000048900318,0.0020638392,0.00019371444,0.000030979354,0.00034494855,0.0025661415,0.0015168851,0.002307546,0.92788905,0.011124287,0.05183301],"study_design_scores_gemma":[0.00002042632,0.00002807417,0.002564796,0.00032334522,0.00016968958,0.00063484086,0.0026911828,0.022044066,0.0031689568,0.64492905,0.32335705,0.000068465204],"about_ca_topic_score_codex":0.019021409,"about_ca_topic_score_gemma":0.02019139,"teacher_disagreement_score":0.019021409,"about_ca_system_score_codex":0.0025783621,"about_ca_system_score_gemma":0.00250604,"threshold_uncertainty_score":0.037821412},"labels":[],"label_agreement":null},{"id":"W3111509402","doi":"10.1017/langcog.2023.11","title":"A learning perspective on the emergence of abstractions: the curious case of phone(me)s","year":2023,"lang":"en","type":"article","venue":"Language and Cognition","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Social Sciences and Humanities Research Council of Canada; Leverhulme Trust","keywords":"Computer science; Generalization; Perspective (graphical); Abstraction; Operationalization; Phone; Artificial intelligence; Consistency (knowledge bases); Process (computing); Simple (philosophy); Natural language processing; Linguistics; Programming language; Mathematics","score_opus":0.013650387022682023,"score_gpt":0.2992132991199458,"score_spread":0.2855629120972638,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3111509402","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.32471687,0.0002901699,0.651079,0.0037951136,0.000069995134,0.00004618593,0.00008871018,0.00039321926,0.019520724],"genre_scores_gemma":[0.95634776,0.000078095094,0.041962016,0.00014771207,0.000021277021,0.00002664418,0.000029768978,0.000040142022,0.001346586],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9986494,0.00062053476,0.00005366459,0.0003384367,0.0002271447,0.00011084864],"domain_scores_gemma":[0.99206334,0.0048080655,0.00055448123,0.0019132887,0.0003918701,0.00026877696],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021002656,0.00035598408,0.00044936425,0.000760576,0.00074335496,0.002788922,0.0014225196,0.0014030207,0.0028858904],"category_scores_gemma":[0.012654131,0.0003899918,0.00076828484,0.00041139114,0.009906221,0.007965269,0.0027160123,0.0032135916,0.0003003926],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015963691,0.00006478309,0.00782553,0.0001387791,0.000055925866,0.00075298914,0.009199411,0.04291779,0.009894038,0.8828677,0.0008316746,0.045291685],"study_design_scores_gemma":[0.000015177499,0.00008577435,0.0021900535,0.000036948022,0.000017749544,0.0004641311,0.000994169,0.13126427,0.0045449166,0.85667527,0.0036662465,0.000045251105],"about_ca_topic_score_codex":0.001485175,"about_ca_topic_score_gemma":0.00090784114,"teacher_disagreement_score":0.0028858904,"about_ca_system_score_codex":0.00070925424,"about_ca_system_score_gemma":0.0004695354,"threshold_uncertainty_score":0.011107385},"labels":[],"label_agreement":null},{"id":"W3111920988","doi":"10.5539/ijel.v11n1p166","title":"A Survey of the Mixed Use of He/She for Chinese Freshmen in English Majors","year":2020,"lang":"en","type":"article","venue":"International Journal of English Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Pronunciation; First language; Psychology; Negative transfer; Linguistics","score_opus":0.028888954036851263,"score_gpt":0.3005862048470512,"score_spread":0.2716972508101999,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3111920988","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99958044,0.000066506116,0.000018294208,0.0000557272,0.000002230653,0.000011590474,0.00007254278,0.0000011598506,0.00019145863],"genre_scores_gemma":[0.9984829,0.00043919287,0.00009995182,0.00012157983,0.0000073535043,0.000034203054,0.00014891486,0.0000020154941,0.0006638028],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99912983,0.00019596812,0.00016297119,0.00010645644,0.00024387093,0.00016093877],"domain_scores_gemma":[0.9974693,0.00042817168,0.0009164257,0.00008697575,0.0005723415,0.0005267993],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013910666,0.0002268802,0.00032096528,0.0013819587,0.0010900145,0.00059651554,0.00040221427,0.00041626705,0.0018792669],"category_scores_gemma":[0.003230139,0.00031129443,0.000338877,0.0011590581,0.0005404529,0.0008830438,0.00071647164,0.0005037205,0.0002951965],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003007135,0.00007561307,0.9665077,0.00011234223,0.000020224621,0.0004869571,0.024322921,0.000011348532,0.0009359274,0.000042545817,0.00032015413,0.0071341754],"study_design_scores_gemma":[0.000002114767,0.0001815825,0.93721235,0.000038715938,0.000015620397,0.00046349617,0.06050292,0.00007751303,0.00022482936,0.00001202323,0.0012529898,0.00001585806],"about_ca_topic_score_codex":0.0201354,"about_ca_topic_score_gemma":0.026639776,"teacher_disagreement_score":0.0201354,"about_ca_system_score_codex":0.0009155745,"about_ca_system_score_gemma":0.001206826,"threshold_uncertainty_score":0.04003638},"labels":[],"label_agreement":null},{"id":"W3112745663","doi":"10.1002/alz.037526","title":"Multilingual text normalization for computer‐based detection of Alzheimer’s disease","year":2020,"lang":"en","type":"article","venue":"Alzheimer s & Dementia","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure","funders":"","keywords":"Normalization (sociology); Computer science; Natural language processing; Pipeline (software); Artificial intelligence; Scalability; Task (project management); Context (archaeology); Correlation; Database; Programming language","score_opus":0.030760376588456356,"score_gpt":0.28869444840297215,"score_spread":0.2579340718145158,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3112745663","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.38210982,0.0047136317,0.5196738,0.0013238031,0.0009949434,0.0015775879,0.02848293,0.051384524,0.009738958],"genre_scores_gemma":[0.42044213,0.0009535292,0.5238958,0.00026729234,0.00043182526,0.0016999994,0.044346813,0.0015220848,0.0064404854],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99667114,0.00083389616,0.00045591054,0.001129299,0.0007193331,0.00019056816],"domain_scores_gemma":[0.9915195,0.0034983708,0.00087851833,0.0010376798,0.002810036,0.00025587878],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002415025,0.0013682871,0.00088156376,0.00492703,0.00083341886,0.0016016489,0.000970331,0.0006448793,0.0061715944],"category_scores_gemma":[0.011257695,0.00031849652,0.0008338904,0.0024407173,0.00054513064,0.0014758995,0.0014297168,0.00084008125,0.004974907],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012803075,0.0003539502,0.017703872,0.001454132,0.00026439785,0.000606437,0.00087200967,0.0038804288,0.14087777,0.0011179177,0.023596501,0.8079923],"study_design_scores_gemma":[0.00018590062,0.0010707206,0.17113307,0.0003981614,0.0006503381,0.002801592,0.0027833006,0.24804983,0.43910185,0.006690881,0.12675537,0.00037900312],"about_ca_topic_score_codex":0.0028167297,"about_ca_topic_score_gemma":0.0037745386,"teacher_disagreement_score":0.0061715944,"about_ca_system_score_codex":0.0008609704,"about_ca_system_score_gemma":0.0012099878,"threshold_uncertainty_score":0.020645976},"labels":[],"label_agreement":null},{"id":"W3113236257","doi":"10.1109/smc42975.2020.9282949","title":"Question-Answering System with Linguistic Terms over RDF Knowledge Graphs","year":2020,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"RDF; Computer science; SPARQL; RDF Schema; Linked data; Simple Knowledge Organization System; Question answering; Information retrieval; World Wide Web; Interface (matter); Semantic Web","score_opus":0.010222396850008619,"score_gpt":0.2579736058153748,"score_spread":0.2477512089653662,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3113236257","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017400287,0.0002888487,0.909451,0.0014183024,0.0001092076,0.0007244474,0.004644635,0.055796355,0.010166909],"genre_scores_gemma":[0.1758416,0.0004086387,0.7917326,0.0012391867,0.0001394117,0.0007512859,0.015589023,0.0015536415,0.012744636],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9982324,0.0005596034,0.0002925831,0.00050601433,0.0003186496,0.00009066092],"domain_scores_gemma":[0.9978703,0.0011784125,0.000108476976,0.0003035617,0.00043658767,0.00010266388],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024728114,0.00075431186,0.0008732541,0.002454892,0.0012199699,0.0024838215,0.0018226148,0.0019676154,0.009312243],"category_scores_gemma":[0.00643957,0.00043825485,0.0011611488,0.0014502475,0.00071331253,0.004916248,0.0027509285,0.0014596345,0.0037007933],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011988858,0.0017522944,0.005372749,0.0020699378,0.00036141058,0.0029100329,0.0051417747,0.03477994,0.07528607,0.17686445,0.14115296,0.5531096],"study_design_scores_gemma":[0.00045914995,0.00021814302,0.0022356533,0.00026899562,0.00028409727,0.0013118072,0.0014924809,0.49045575,0.06276429,0.12444653,0.3158662,0.00019694954],"about_ca_topic_score_codex":0.0043310127,"about_ca_topic_score_gemma":0.0035056067,"teacher_disagreement_score":0.009312243,"about_ca_system_score_codex":0.0010176203,"about_ca_system_score_gemma":0.0012415508,"threshold_uncertainty_score":0.031152546},"labels":[],"label_agreement":null},{"id":"W3113327296","doi":"10.37514/pra-b.2019.0230","title":"Coding Streams of Language: Techniques for the Systematic Coding of Text, Talk, and Other Verbal Data","year":2019,"lang":"en","type":"book","venue":"The WAC Clearinghouse; University Press of Colorado eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":125,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Coding (social sciences); Computer science; Natural language processing; Linguistics; Mathematics; Statistics","score_opus":0.028908686537832993,"score_gpt":0.25390181804888856,"score_spread":0.22499313151105557,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3113327296","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0007758597,0.00051571656,0.9908235,0.00030807342,0.0000945875,0.00052437343,0.0012009498,0.0023864205,0.0033705248],"genre_scores_gemma":[0.008103521,0.00058881094,0.983095,0.000076338205,0.000047463567,0.001256381,0.0016930301,0.0010493888,0.0040902053],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9887755,0.005064794,0.0015673655,0.0012252794,0.0031444859,0.00022266718],"domain_scores_gemma":[0.9627913,0.022304874,0.0015573496,0.0056016487,0.007311384,0.00043341075],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01169264,0.0023362392,0.0012766059,0.008905233,0.0025494255,0.0075637815,0.0034909055,0.0014586361,0.013238922],"category_scores_gemma":[0.045189623,0.0021310374,0.0016226635,0.011538266,0.006335039,0.010258372,0.004801935,0.0038119564,0.007853834],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015462599,0.000065203574,0.0006606684,0.001170974,0.000058729944,0.00015620953,0.010968667,0.0016768149,0.004873513,0.20969906,0.048817538,0.72169805],"study_design_scores_gemma":[0.00009459298,0.00008873798,0.0016900799,0.0014192052,0.000125054,0.0006671013,0.0066288807,0.044536263,0.025247112,0.5664886,0.3527804,0.00023397143],"about_ca_topic_score_codex":0.008851165,"about_ca_topic_score_gemma":0.009618827,"teacher_disagreement_score":0.013238922,"about_ca_system_score_codex":0.0024188466,"about_ca_system_score_gemma":0.0073022735,"threshold_uncertainty_score":0.061837375},"labels":[],"label_agreement":null},{"id":"W3113495691","doi":"10.4000/memini.1787","title":"PolimaWiki : un Wiki sémantique pour l’analyse de listes au Moyen Âge, limites et apports","year":2020,"lang":"fr","type":"article","venue":"Memini","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"Agence Nationale de la Recherche","keywords":"Computer science; Philosophy","score_opus":0.038178282305804084,"score_gpt":0.29929806114601204,"score_spread":0.261119778840208,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3113495691","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023208162,0.0025569694,0.88303626,0.0014446181,0.0005946637,0.0004522557,0.025995912,0.044402536,0.018308574],"genre_scores_gemma":[0.104869656,0.0023922212,0.8185444,0.00046158236,0.00020407897,0.0011551467,0.034516122,0.013222274,0.024634566],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9975776,0.00053678174,0.00045412977,0.00058286503,0.0007341893,0.00011438012],"domain_scores_gemma":[0.99423283,0.002871813,0.00059279153,0.0010573477,0.0009920461,0.00025309954],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002638184,0.0013739883,0.0010042178,0.006208134,0.0017182799,0.0068585365,0.0018108237,0.0013194241,0.00850022],"category_scores_gemma":[0.012354763,0.0015822125,0.0010805457,0.00496924,0.0012924408,0.01872981,0.0038617784,0.0032601168,0.0067823376],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005844042,0.00028717698,0.009556126,0.003457003,0.00036430793,0.00096044166,0.011820818,0.00352423,0.03257291,0.12622166,0.1156433,0.6950076],"study_design_scores_gemma":[0.00007051772,0.00009362023,0.0076021235,0.00087055797,0.00019440084,0.0015719198,0.0044884644,0.047408998,0.02998603,0.082701035,0.82471424,0.00029819543],"about_ca_topic_score_codex":0.009947806,"about_ca_topic_score_gemma":0.017551154,"teacher_disagreement_score":0.009947806,"about_ca_system_score_codex":0.0013538471,"about_ca_system_score_gemma":0.0034387377,"threshold_uncertainty_score":0.028436065},"labels":[],"label_agreement":null},{"id":"W3114346701","doi":"10.18653/v1/2020.aacl-main.36","title":"A Systematic Characterization of Sampling Algorithms for Open-ended Language Generation","year":2020,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Institute for Advanced Research","funders":"Samsung Advanced Institute of Technology; Canadian Institute for Advanced Research; Samsung; Nvidia","keywords":"Computer science; Sampling (signal processing); Algorithm; Set (abstract data type); Entropy (arrow of time); Programming language","score_opus":0.06128786778520071,"score_gpt":0.3250131025448521,"score_spread":0.2637252347596514,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3114346701","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0050653378,0.00030771515,0.99176496,0.00014578235,0.00004561367,0.0001724551,0.0000756772,0.0006741878,0.0017482165],"genre_scores_gemma":[0.16100623,0.0004059042,0.8329521,0.00033320874,0.0002168673,0.00081270974,0.0008397443,0.0011113839,0.0023218733],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9710471,0.01434184,0.002708913,0.0031123636,0.0076444447,0.001145269],"domain_scores_gemma":[0.8090525,0.14709084,0.003062702,0.025720645,0.013657793,0.0014155363],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.017895542,0.0013188525,0.0018604499,0.0027380069,0.0020126232,0.0041921795,0.005495729,0.002613689,0.004709756],"category_scores_gemma":[0.12807688,0.001321186,0.002128029,0.002987829,0.0030102348,0.0067539215,0.006662648,0.004402932,0.0019414657],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00067549286,0.0005721041,0.004906033,0.0005550629,0.00015341044,0.00021918102,0.0012175916,0.08404525,0.01091299,0.32361114,0.00741888,0.56571275],"study_design_scores_gemma":[0.00013646726,0.00022865867,0.00067110005,0.00016339537,0.00007514864,0.00036229208,0.00014825365,0.69597715,0.011214544,0.2846945,0.006267612,0.000060892635],"about_ca_topic_score_codex":0.0017544084,"about_ca_topic_score_gemma":0.0028909482,"teacher_disagreement_score":0.017895542,"about_ca_system_score_codex":0.0021550704,"about_ca_system_score_gemma":0.0039096097,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W3115819292","doi":"10.18653/v1/2020.semeval-1.92","title":"Defx at SemEval-2020 Task 6: Joint Extraction of Concepts and Relations for Definition Extraction","year":2020,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Banting and Best Diabetes Centre, University of Toronto; Bundesministerium für Bildung und Forschung","keywords":"SemEval; Computer science; Task (project management); Joint (building); Natural language processing; Relationship extraction; Artificial intelligence; Extraction (chemistry); Information extraction; Chromatography","score_opus":0.04253042089633639,"score_gpt":0.3153415079120207,"score_spread":0.2728110870156843,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3115819292","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.087698355,0.0058491747,0.45560637,0.006651891,0.0037439787,0.004056306,0.20221932,0.16458753,0.06958702],"genre_scores_gemma":[0.09663817,0.00094068976,0.40878895,0.0012653223,0.0003897261,0.0020170382,0.4540311,0.012075166,0.023853812],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.98748,0.0053615565,0.0012426347,0.0025952,0.0027731147,0.00054747285],"domain_scores_gemma":[0.97555465,0.012609572,0.00070663134,0.005388708,0.004981108,0.0007592675],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012435462,0.003486307,0.0021789696,0.005252888,0.002476352,0.004577837,0.0036710296,0.0034531609,0.040826123],"category_scores_gemma":[0.028720804,0.001375139,0.0020704987,0.0030661018,0.001407319,0.009910071,0.009964703,0.004667313,0.02868752],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00085116463,0.0006805065,0.0036263634,0.004141822,0.00023495112,0.00091161794,0.0015778884,0.002492497,0.024762983,0.018665433,0.678087,0.26396775],"study_design_scores_gemma":[0.0006346779,0.00044827457,0.00954075,0.0005566702,0.00012648145,0.0023959347,0.0019782924,0.047970496,0.06197647,0.017159738,0.85699505,0.00021713992],"about_ca_topic_score_codex":0.004912678,"about_ca_topic_score_gemma":0.008170514,"teacher_disagreement_score":0.040826123,"about_ca_system_score_codex":0.002093968,"about_ca_system_score_gemma":0.0038600052,"threshold_uncertainty_score":0.13657695},"labels":[],"label_agreement":null},{"id":"W3116370925","doi":"10.3968/11858","title":"A Study on the E-C Consecutive Interpreting in Ocean Security Conference Guided by Compression Theory","year":2020,"lang":"en","type":"article","venue":"Cross-cultural communication","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Syllabic verse; Compression (physics); Interpretation (philosophy); Computer science; Natural language processing; Linguistics; Data compression; Data compression ratio; Semantic interpretation; Semantic compression; Artificial intelligence; Speech recognition; Image compression","score_opus":0.044209439891439,"score_gpt":0.3667829312377945,"score_spread":0.3225734913463555,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3116370925","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8607845,0.00046935954,0.016493695,0.0013295825,0.000101046906,0.00023256296,0.00012619377,0.00006471061,0.120398335],"genre_scores_gemma":[0.9901114,0.00023635797,0.0031951412,0.00013093409,0.000037683454,0.000069193185,0.000097310585,0.000054783628,0.006067244],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.99578106,0.0028588811,0.00013196886,0.0002496069,0.0007363173,0.0002421538],"domain_scores_gemma":[0.97027844,0.024824398,0.0013735669,0.0009591319,0.0020547265,0.0005098623],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028708868,0.00039974836,0.00021445073,0.0020784896,0.0034322403,0.0035686078,0.00070061133,0.00083107036,0.0055484143],"category_scores_gemma":[0.025671376,0.00024141224,0.00021844546,0.0032242753,0.005274695,0.0038155268,0.0014553033,0.0025618235,0.00037690747],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002810497,0.00022569015,0.025500137,0.00042400733,0.000020108753,0.0021990705,0.74955535,0.0008421004,0.0076978216,0.1238614,0.0039549125,0.085438415],"study_design_scores_gemma":[0.000049412814,0.00030839304,0.10132994,0.00041309776,0.00004247133,0.00297191,0.7578197,0.00806118,0.008442277,0.022505939,0.097939074,0.00011655676],"about_ca_topic_score_codex":0.0070480104,"about_ca_topic_score_gemma":0.0072136694,"teacher_disagreement_score":0.0070480104,"about_ca_system_score_codex":0.0027653924,"about_ca_system_score_gemma":0.0017734389,"threshold_uncertainty_score":0.020064354},"labels":[],"label_agreement":null},{"id":"W3116587780","doi":"10.18653/v1/2020.coling-main.410","title":"Revitalization of Indigenous Languages through Pre-processing and Neural Machine Translation: The case of Inuktitut","year":2020,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Machine translation; Computer science; Artificial intelligence; Natural language processing; Indigenous; Preprocessor; Context (archaeology); Focus (optics); Linguistics; History","score_opus":0.022148490118493902,"score_gpt":0.29940314502512555,"score_spread":0.27725465490663165,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3116587780","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8017094,0.0040223994,0.12822425,0.0024679138,0.00032179273,0.0003873145,0.0015880377,0.0020831984,0.0591957],"genre_scores_gemma":[0.86199474,0.0016622195,0.11613689,0.00035489132,0.000035628967,0.000085742366,0.0022935579,0.00037208718,0.017064165],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9997193,0.00006120731,0.000020030377,0.00006770156,0.000058329562,0.000073505755],"domain_scores_gemma":[0.9995907,0.000113695816,0.000032823154,0.000056210934,0.000183047,0.00002361621],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00035371818,0.00054324954,0.0003124865,0.00073064363,0.0020676681,0.0013738705,0.0006234014,0.0004845273,0.0025280123],"category_scores_gemma":[0.0012017546,0.00012260127,0.00032614273,0.0012959711,0.00084088015,0.0009907958,0.0009561113,0.00091359514,0.00068155164],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00039459654,0.00021300065,0.013028496,0.001725711,0.00011654089,0.0054075043,0.013962279,0.021449966,0.08873742,0.033880617,0.012868821,0.80821514],"study_design_scores_gemma":[0.00012837627,0.00045852768,0.08552148,0.00083717424,0.0005743869,0.0074155275,0.059159543,0.25227824,0.21515156,0.030081548,0.34797674,0.00041684406],"about_ca_topic_score_codex":0.3859372,"about_ca_topic_score_gemma":0.5559869,"teacher_disagreement_score":0.3859372,"about_ca_system_score_codex":0.002637386,"about_ca_system_score_gemma":0.006011603,"threshold_uncertainty_score":0.7673816},"labels":[],"label_agreement":null},{"id":"W3116628494","doi":"10.18653/v1/2020.coling-tutorials.7","title":"Endangered Languages meet Modern NLP","year":2020,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"University of Notre Dame; Carleton University; Carnegie Mellon University","keywords":"Computer science; Natural language processing; Endangered species; Artificial intelligence; Biology; Ecology","score_opus":0.01814678136304938,"score_gpt":0.270956021396855,"score_spread":0.2528092400338056,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3116628494","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008754135,0.05089046,0.56859666,0.1848985,0.0051298756,0.00021755455,0.0015203146,0.0048367293,0.17515586],"genre_scores_gemma":[0.11209372,0.07175855,0.70567167,0.028348897,0.007361669,0.00060668384,0.005044915,0.0037243133,0.06538951],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99253476,0.0033925602,0.00060757046,0.00079592265,0.0022685113,0.00040069164],"domain_scores_gemma":[0.97971696,0.014296434,0.00069583755,0.0024010057,0.0022188115,0.00067094987],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010383535,0.0008550705,0.000912578,0.003003322,0.0035486273,0.013923379,0.0024583277,0.0058970586,0.016581737],"category_scores_gemma":[0.02278717,0.0010466494,0.0009580968,0.0030543632,0.005837586,0.031759262,0.009527443,0.009275192,0.008418476],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000039849518,0.000040074712,0.00079679507,0.002015808,0.000040420004,0.0008456575,0.0063010794,0.003011164,0.0038837784,0.47620147,0.12942713,0.37739682],"study_design_scores_gemma":[0.0000043314617,0.00000896882,0.00030133172,0.00077647285,0.00000794679,0.00072057423,0.0022299858,0.0023385915,0.0007879318,0.21842884,0.7743617,0.00003327243],"about_ca_topic_score_codex":0.0026058855,"about_ca_topic_score_gemma":0.0040822243,"teacher_disagreement_score":0.016581737,"about_ca_system_score_codex":0.0035803297,"about_ca_system_score_gemma":0.0035849,"threshold_uncertainty_score":0.05547148},"labels":[],"label_agreement":null},{"id":"W3117059669","doi":"10.18653/v1/2020.coling-main.527","title":"Data Selection for Bilingual Lexicon Induction from Specialized Comparable Corpora","year":2020,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; Computer Research Institute of Montréal","funders":"Institut de Valorisation des Données; Agence Nationale de la Recherche","keywords":"Computer science; Artificial intelligence; Exploit; Lexicon; Margin (machine learning); Selection (genetic algorithm); Natural language processing; Machine translation; Computation; Entropy (arrow of time); Machine learning; Algorithm","score_opus":0.13890980270002126,"score_gpt":0.3420511324608173,"score_spread":0.20314132976079605,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3117059669","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08159215,0.0006531989,0.89875406,0.00028764512,0.00013982692,0.0005205581,0.0031090465,0.010683079,0.004260379],"genre_scores_gemma":[0.31208584,0.000443524,0.64936,0.00024487264,0.0001252861,0.0016362914,0.031439964,0.0013083903,0.0033558002],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9976336,0.0010804256,0.00024801545,0.0005328348,0.00036821602,0.00013698582],"domain_scores_gemma":[0.99649113,0.0017517045,0.00014158132,0.0007757579,0.0007303479,0.00010947339],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028774997,0.0013943611,0.0013953077,0.0036013215,0.000891828,0.001228229,0.0012210109,0.00074206793,0.0059913746],"category_scores_gemma":[0.009091569,0.00073972595,0.0013535538,0.0039368095,0.0007664684,0.002708669,0.0030489957,0.001356854,0.0050556646],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00070518797,0.00037374522,0.003849199,0.00073730067,0.00015791462,0.00046548265,0.00056445174,0.020285629,0.067452066,0.006449685,0.01830346,0.8806558],"study_design_scores_gemma":[0.00052178436,0.0005892245,0.008290646,0.00014007011,0.0002850297,0.00079829764,0.0010895245,0.74724245,0.14375526,0.030912984,0.06623002,0.00014469377],"about_ca_topic_score_codex":0.0018771404,"about_ca_topic_score_gemma":0.004441638,"teacher_disagreement_score":0.0059913746,"about_ca_system_score_codex":0.00072195585,"about_ca_system_score_gemma":0.0018781448,"threshold_uncertainty_score":0.020043135},"labels":[],"label_agreement":null},{"id":"W3117574438","doi":"10.18653/v1/2020.semeval-1.54","title":"CLaC at SemEval-2020 Task 5: Muli-task Stacked Bi-LSTMs","year":2020,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Argumentative; Computer science; Task (project management); SemEval; Artificial intelligence; Natural language processing; Set (abstract data type); Word (group theory); Linguistics; Programming language","score_opus":0.015030420126732506,"score_gpt":0.25995333788693387,"score_spread":0.24492291776020136,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3117574438","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.43272778,0.0036873855,0.29370636,0.0023593956,0.0036966896,0.0011134435,0.052702297,0.14848338,0.06152332],"genre_scores_gemma":[0.70632887,0.00032561814,0.2025717,0.0011260851,0.00032718264,0.00078853755,0.060254693,0.0030891686,0.025188124],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991823,0.00016176233,0.000036928195,0.000360476,0.00012248996,0.00013612286],"domain_scores_gemma":[0.9983839,0.00050586724,0.000066646,0.00049320346,0.0004039754,0.00014643763],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016172297,0.002641639,0.0011684202,0.00065724045,0.0007426497,0.0018720245,0.0026590659,0.0031497737,0.022753224],"category_scores_gemma":[0.00496788,0.0006211702,0.000785928,0.0005676897,0.0004891857,0.004327804,0.0022101726,0.0028698482,0.011723164],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002479254,0.0011210829,0.0035317293,0.0016169728,0.00052074454,0.0010934304,0.00059993146,0.04317635,0.105860576,0.008879457,0.26588893,0.56523144],"study_design_scores_gemma":[0.000420054,0.00093308627,0.005541534,0.0001518347,0.00016659073,0.0009207154,0.00035977797,0.81590605,0.09865888,0.018651986,0.058101814,0.00018776048],"about_ca_topic_score_codex":0.006459409,"about_ca_topic_score_gemma":0.011177225,"teacher_disagreement_score":0.022753224,"about_ca_system_score_codex":0.0009364479,"about_ca_system_score_gemma":0.00134868,"threshold_uncertainty_score":0.07611704},"labels":[],"label_agreement":null},{"id":"W3117912976","doi":"10.18653/v1/2020.semeval-1.32","title":"UAlberta at SemEval-2020 Task 2: Using Translations to Predict Cross-Lingual Entailment","year":2020,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Machine Intelligence Institute","keywords":"Leverage (statistics); Computer science; Textual entailment; Logical consequence; Natural language processing; SemEval; Artificial intelligence; Task (project management); Word (group theory); Linguistics","score_opus":0.027191833433764728,"score_gpt":0.3227681545403572,"score_spread":0.29557632110659243,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3117912976","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5741653,0.011153367,0.08402567,0.0054989695,0.003571288,0.0010628173,0.20218024,0.031600688,0.08674159],"genre_scores_gemma":[0.48255846,0.0010860489,0.1029343,0.001026782,0.0004118322,0.0006742544,0.36932123,0.0019123842,0.04007482],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9975121,0.0009124159,0.00018210559,0.0006737043,0.00056238595,0.00015737231],"domain_scores_gemma":[0.9960453,0.0012941026,0.00027785453,0.0009789653,0.0011025472,0.0003012219],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00361579,0.001639503,0.0009891273,0.0029496357,0.001628465,0.0024909426,0.0012700222,0.0021311124,0.01893384],"category_scores_gemma":[0.011053766,0.0004730983,0.0009855584,0.0017077049,0.0005732846,0.003003776,0.0023672024,0.0017347703,0.013401868],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004165897,0.0015548726,0.02182702,0.0013987495,0.000521231,0.0013211307,0.0006309507,0.013829557,0.019220965,0.008235448,0.49428934,0.43300495],"study_design_scores_gemma":[0.0019501965,0.0015429509,0.07190728,0.0007489196,0.000524919,0.0027072942,0.0018650094,0.36119324,0.07825108,0.03002797,0.44893423,0.0003469291],"about_ca_topic_score_codex":0.018632457,"about_ca_topic_score_gemma":0.038726404,"teacher_disagreement_score":0.01893384,"about_ca_system_score_codex":0.0013068994,"about_ca_system_score_gemma":0.001668201,"threshold_uncertainty_score":0.06334001},"labels":[],"label_agreement":null},{"id":"W3119425390","doi":"","title":"Challenges in Neural Language Identification: NRC at VarDial 2020","year":2020,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Identification (biology); Task (project management); Computer science; Sample (material); Baseline (sea); Function (biology); Artificial intelligence; Artificial neural network; Natural language processing; Machine learning; Engineering; Political science","score_opus":0.03508458236170361,"score_gpt":0.283119764766989,"score_spread":0.2480351824052854,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3119425390","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.45448062,0.021804463,0.22574525,0.0913265,0.011439082,0.0015099215,0.0327897,0.059710756,0.10119366],"genre_scores_gemma":[0.6937256,0.0019404176,0.19273925,0.010967485,0.001142109,0.0007137729,0.05342429,0.0025108496,0.042836294],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9867488,0.004390523,0.0004931923,0.0022818218,0.0044057947,0.0016797947],"domain_scores_gemma":[0.9824675,0.004075219,0.0003791211,0.002414685,0.0091621885,0.0015013082],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.018431973,0.0010159812,0.0015271411,0.0016477172,0.003119498,0.0048098597,0.0035161918,0.0037513073,0.005065924],"category_scores_gemma":[0.02492836,0.0005394128,0.00060873147,0.00115669,0.0015676304,0.004799383,0.004973439,0.005062166,0.00736856],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019551797,0.0008369565,0.011123357,0.0010276642,0.00019146058,0.0007885411,0.0015114086,0.011462324,0.025583223,0.016954988,0.3862385,0.5423264],"study_design_scores_gemma":[0.00049398374,0.0016185313,0.0263796,0.0007186783,0.00024076886,0.0030740625,0.0077276104,0.28902808,0.13555714,0.03503285,0.49954915,0.0005795392],"about_ca_topic_score_codex":0.07628047,"about_ca_topic_score_gemma":0.12032182,"teacher_disagreement_score":0.07628047,"about_ca_system_score_codex":0.004910636,"about_ca_system_score_gemma":0.007070822,"threshold_uncertainty_score":0.15167296},"labels":[],"label_agreement":null},{"id":"W3119505789","doi":"10.21248/zaspil.60.2018.475","title":"processing cost of Downward Entailingness: the representation and verification of comparative constructions","year":2018,"lang":"en","type":"article","venue":"ZAS Papers in Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Quantifier (linguistics); Computer science; Representation (politics); Monotonic function; Negation; Semantics (computer science); Process (computing); Theoretical computer science; Point (geometry); Algorithm; Artificial intelligence; Arithmetic; Mathematics; Programming language","score_opus":0.02808243805788615,"score_gpt":0.3413883978666717,"score_spread":0.31330595980878556,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3119505789","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4702647,0.00070831034,0.5012019,0.0024443872,0.0001275993,0.00021457407,0.00055125443,0.0008526076,0.023634674],"genre_scores_gemma":[0.9204378,0.00015142807,0.07719828,0.00016160963,0.00004025691,0.00017444746,0.0003046366,0.00020676831,0.0013246745],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9903016,0.0048384573,0.0006455046,0.001714597,0.0021498378,0.00035002053],"domain_scores_gemma":[0.8929604,0.073519856,0.0072630257,0.021963643,0.0037305793,0.0005623806],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013500932,0.0007012873,0.00072491734,0.0015960223,0.0011343823,0.004714615,0.0021004574,0.001701781,0.015302216],"category_scores_gemma":[0.09347602,0.0007246811,0.0009644423,0.0013249929,0.006243025,0.019693805,0.0047185523,0.0026791173,0.0005560459],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011241925,0.0002915351,0.009974579,0.0009860925,0.000120598306,0.000258713,0.006067448,0.007971628,0.047587093,0.81966394,0.0014475537,0.104506634],"study_design_scores_gemma":[0.00011016666,0.00051408226,0.019636177,0.00015215081,0.00013907842,0.0006390019,0.0016846964,0.0647599,0.07356348,0.8299433,0.00869098,0.00016694759],"about_ca_topic_score_codex":0.0009400783,"about_ca_topic_score_gemma":0.0006426255,"teacher_disagreement_score":0.015302216,"about_ca_system_score_codex":0.0024549996,"about_ca_system_score_gemma":0.0010057655,"threshold_uncertainty_score":0.07140058},"labels":[],"label_agreement":null},{"id":"W3119989665","doi":"","title":"ARBERT & MARBERT: Deep Bidirectional Transformers for Arabic.","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":138,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Arabic; Transformer; Natural language processing; Inference; Language model; Task (project management); Artificial intelligence; Benchmark (surveying); Machine learning; Linguistics","score_opus":0.057217742704012906,"score_gpt":0.20698105795943467,"score_spread":0.14976331525542175,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3119989665","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.036935635,0.0038063757,0.80831975,0.0020873095,0.0012049597,0.0003106577,0.010641716,0.11369475,0.02299882],"genre_scores_gemma":[0.42754132,0.0019240502,0.4894152,0.0014304442,0.0001909156,0.00064629293,0.029007979,0.0044147996,0.04542899],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9996809,0.00008169644,0.000020377454,0.0001091901,0.00006654281,0.00004139594],"domain_scores_gemma":[0.99947363,0.00017109145,0.000033421,0.00015334241,0.000115627794,0.000052936546],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009352138,0.0020960884,0.00048211758,0.00094840623,0.00056280324,0.0016089496,0.0019823802,0.0013029949,0.010744474],"category_scores_gemma":[0.0033541147,0.00066940085,0.0011881426,0.00064596656,0.000505908,0.003913511,0.0021722307,0.0028971885,0.011182839],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00064514176,0.0002147596,0.0021028568,0.00049955514,0.00020458059,0.00041397157,0.00039455137,0.08453224,0.021973029,0.02650709,0.16311525,0.69939685],"study_design_scores_gemma":[0.000086015876,0.00012554672,0.00056985964,0.00009113724,0.000060361446,0.00023640195,0.00015634783,0.8874684,0.027007394,0.037740223,0.046408266,0.00004997968],"about_ca_topic_score_codex":0.0059758243,"about_ca_topic_score_gemma":0.010877181,"teacher_disagreement_score":0.010744474,"about_ca_system_score_codex":0.0007751402,"about_ca_system_score_gemma":0.0011821362,"threshold_uncertainty_score":0.035943866},"labels":[],"label_agreement":null},{"id":"W3120179416","doi":"10.18653/v1/2020.wmt-1.99","title":"Extended Study on Using Pretrained Language Models and YiSi-1 for Machine Translation Evaluation","year":2020,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Machine translation; Computer science; Task (project management); Artificial intelligence; Natural language processing; Translation (biology); Evaluation of machine translation; BLEU; Language model; Quality (philosophy); Embedding; Machine learning; Example-based machine translation; Machine translation software usability","score_opus":0.11028485885458243,"score_gpt":0.37524723392901643,"score_spread":0.264962375074434,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3120179416","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6181668,0.0098881535,0.33340287,0.0018428297,0.000989196,0.00068867236,0.0042675165,0.0126555925,0.01809838],"genre_scores_gemma":[0.8136781,0.0013232235,0.16410805,0.0005308018,0.00026342907,0.00043023695,0.012836225,0.0009994254,0.005830479],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9910424,0.004551359,0.0007426015,0.0014877333,0.00176908,0.0004067929],"domain_scores_gemma":[0.9512072,0.024213403,0.001959222,0.010634053,0.011272188,0.0007139447],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.014467006,0.0026666755,0.0012469835,0.0020638471,0.0006555152,0.0020886967,0.0017520714,0.0013257791,0.0035455956],"category_scores_gemma":[0.050176628,0.0006313009,0.0009861684,0.0024246443,0.0008125327,0.004399806,0.0021017331,0.0036152678,0.0018362795],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018640484,0.0017347962,0.048488013,0.001064736,0.0017597398,0.00035329294,0.00059273525,0.19995995,0.035534102,0.004331681,0.020877646,0.68343925],"study_design_scores_gemma":[0.000083773644,0.002247755,0.022851165,0.00016012536,0.0002914939,0.00034771886,0.00019124895,0.9108247,0.0520087,0.004150748,0.0066911275,0.00015147545],"about_ca_topic_score_codex":0.008055374,"about_ca_topic_score_gemma":0.013308826,"teacher_disagreement_score":0.014467006,"about_ca_system_score_codex":0.0014136226,"about_ca_system_score_gemma":0.0019954664,"threshold_uncertainty_score":0.076509714},"labels":[],"label_agreement":null},{"id":"W3120459072","doi":"10.18653/v1/2020.wmt-1.110","title":"Improving Parallel Data Identification using Iteratively Refined Sentence Alignments and Bilingual Mappings of Pre-trained Language Models","year":2020,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Computer science; Machine translation; Natural language processing; Sentence; Artificial intelligence; Context (archaeology); Metric (unit); Task (project management); Language model; Similarity (geometry)","score_opus":0.06009170999017288,"score_gpt":0.31448635780946804,"score_spread":0.25439464781929516,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3120459072","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07487601,0.00060917466,0.8968086,0.00036687354,0.0002719831,0.00018222141,0.00095254905,0.023533419,0.0023991691],"genre_scores_gemma":[0.24750327,0.00024951308,0.73293227,0.00032085014,0.00013588268,0.00039404206,0.009978201,0.002198704,0.0062872544],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99651355,0.0012681527,0.00026706053,0.0010516223,0.00072415144,0.00017554138],"domain_scores_gemma":[0.99348265,0.002428565,0.00034315913,0.0014642985,0.0020966209,0.00018470112],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003957412,0.0020255656,0.0016643172,0.002231308,0.0009280328,0.0018299235,0.0018876907,0.001303042,0.0049924874],"category_scores_gemma":[0.014263733,0.0010047719,0.0015523317,0.0020952588,0.0006131063,0.004155117,0.0025250162,0.0025634628,0.0065219244],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007095042,0.0006992445,0.004460384,0.00040278304,0.0003463902,0.00044537886,0.00061575725,0.08017517,0.05749413,0.0063577867,0.022791995,0.82550144],"study_design_scores_gemma":[0.00013967825,0.00030187628,0.0017041801,0.000026806334,0.000098094744,0.00023336204,0.0003032535,0.94241315,0.036927726,0.007852281,0.009938086,0.000061540755],"about_ca_topic_score_codex":0.007678352,"about_ca_topic_score_gemma":0.015503183,"teacher_disagreement_score":0.007678352,"about_ca_system_score_codex":0.000901178,"about_ca_system_score_gemma":0.0031530692,"threshold_uncertainty_score":0.020929039},"labels":[],"label_agreement":null},{"id":"W3120549770","doi":"10.18653/v1/2020.wmt-1.132","title":"NRC Systems for Low Resource German-Upper Sorbian Machine Translation 2020: Transfer Learning with Lexical Modifications","year":2020,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"National Research Council Canada","funders":"","keywords":"German; Machine translation; Transfer of learning; Computer science; Natural language processing; Artificial intelligence; Transformer; Dropout (neural networks); Linguistics; Machine learning; Engineering","score_opus":0.023031025691231617,"score_gpt":0.2704460351172031,"score_spread":0.2474150094259715,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3120549770","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06327147,0.004737928,0.6795691,0.0029999479,0.0020104935,0.0011606746,0.022567485,0.16488458,0.058798272],"genre_scores_gemma":[0.33327696,0.0014738783,0.48199627,0.0012949173,0.00045416833,0.0011129131,0.10585939,0.009601834,0.06492974],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9975151,0.0007337044,0.0001339625,0.00051963766,0.00080056203,0.00029710087],"domain_scores_gemma":[0.9957819,0.0006656003,0.00014706119,0.0014218505,0.0017517888,0.00023187531],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004444555,0.0019203888,0.001739448,0.0016612791,0.0014313324,0.0025454524,0.0027117017,0.0017993605,0.015082768],"category_scores_gemma":[0.0086229965,0.00085339404,0.00086172175,0.0019181023,0.00064530934,0.0028492052,0.0033398184,0.002757284,0.02097867],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000839346,0.00035307952,0.0015675994,0.00052587275,0.00025347454,0.00030495413,0.0003174531,0.03900858,0.016879776,0.013628003,0.2624987,0.66382325],"study_design_scores_gemma":[0.00031083854,0.00051662634,0.002811632,0.00015279952,0.00017040038,0.00048338823,0.00017464506,0.7389397,0.047464363,0.021470675,0.18732437,0.00018063086],"about_ca_topic_score_codex":0.02554713,"about_ca_topic_score_gemma":0.04764185,"teacher_disagreement_score":0.02554713,"about_ca_system_score_codex":0.0020340057,"about_ca_system_score_gemma":0.0042785397,"threshold_uncertainty_score":0.050796807},"labels":[],"label_agreement":null},{"id":"W3121090600","doi":"10.18653/v1/2020.wmt-1.13","title":"NRC Systems for the 2020 Inuktitut-English News Translation Task","year":2020,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Computer science; Task (project management); Machine translation; Domain (mathematical analysis); Transformer; Translation (biology); Natural language processing; Artificial intelligence; Management; Engineering; Mathematics","score_opus":0.027450400878673408,"score_gpt":0.26247543646179694,"score_spread":0.23502503558312354,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3121090600","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07680879,0.007890196,0.12381844,0.02256053,0.024186367,0.005771492,0.3909003,0.15220606,0.19585793],"genre_scores_gemma":[0.06612135,0.001125253,0.13524184,0.002203479,0.0009855046,0.0015134903,0.70001453,0.007554617,0.08523996],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9903246,0.0022912845,0.00068816263,0.0014027155,0.0040391283,0.001254131],"domain_scores_gemma":[0.973421,0.002083601,0.00044591405,0.003324061,0.018079247,0.002646195],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011990907,0.0034294575,0.002532625,0.003929703,0.0055325953,0.005317597,0.004046536,0.0040185885,0.035945676],"category_scores_gemma":[0.023850642,0.0012365987,0.0015366912,0.0046607316,0.001087303,0.003054227,0.005471054,0.0035739562,0.05746366],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00040574718,0.00021165451,0.00069367146,0.00049548753,0.000104057915,0.00032546924,0.00031567723,0.0021100042,0.005629829,0.0022058461,0.92591125,0.0615912],"study_design_scores_gemma":[0.0005900443,0.000390255,0.0055494015,0.00020720043,0.00016522403,0.00054124836,0.0012178037,0.04939819,0.021990154,0.004262954,0.91541654,0.00027096752],"about_ca_topic_score_codex":0.1661918,"about_ca_topic_score_gemma":0.25948104,"teacher_disagreement_score":0.1661918,"about_ca_system_score_codex":0.0050987005,"about_ca_system_score_gemma":0.015580795,"threshold_uncertainty_score":0.33044893},"labels":[],"label_agreement":null},{"id":"W3121207479","doi":"10.1093/pan/mpn007","title":"Lexical Cohesion Analysis of Political Speech","year":2008,"lang":"en","type":"article","venue":"Political Analysis","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":70,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kellogg's (Canada)","funders":"","keywords":"Cohesion (chemistry); Rhetorical question; Computer science; Linguistics; Annotation; Natural language processing; Lexical density; Representation (politics); Politics; Interpretation (philosophy); Artificial intelligence; Lexical item; Political science","score_opus":0.021939312034272573,"score_gpt":0.3055167210051506,"score_spread":0.283577408970878,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3121207479","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.40916166,0.0014216729,0.5642508,0.00065145193,0.00019583876,0.00039161943,0.0021201256,0.0020494591,0.019757338],"genre_scores_gemma":[0.8546861,0.0002940116,0.1404442,0.000052419186,0.00014065477,0.00028889644,0.001695412,0.00028599644,0.0021122545],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9975199,0.0010566231,0.0001820301,0.00048109828,0.0006519725,0.00010830949],"domain_scores_gemma":[0.9899714,0.0064846887,0.0013023039,0.00064115913,0.0014573452,0.00014314573],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013599945,0.00032369007,0.00046583675,0.007816614,0.0013747966,0.0021022223,0.00059430033,0.0004351572,0.0026675786],"category_scores_gemma":[0.013281947,0.00024215248,0.00034039916,0.0034980506,0.0010316711,0.0024926222,0.0013820246,0.00065111835,0.00068506785],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00064572704,0.00015758154,0.031676102,0.001569755,0.0002234356,0.001058562,0.03187554,0.0060863052,0.14193274,0.08701584,0.009741051,0.6880173],"study_design_scores_gemma":[0.00015523435,0.0004701508,0.2110816,0.00061166176,0.00046260917,0.0022416872,0.02651807,0.27297094,0.14091678,0.18026294,0.16385898,0.0004493843],"about_ca_topic_score_codex":0.0012846568,"about_ca_topic_score_gemma":0.00158943,"teacher_disagreement_score":0.007816614,"about_ca_system_score_codex":0.00056143914,"about_ca_system_score_gemma":0.00062893174,"threshold_uncertainty_score":0.008923948},"labels":[],"label_agreement":null},{"id":"W3121601712","doi":"10.7146/hn.v5i2.142742","title":"Interfacing the Hebrew Bible: past, present and future applications for the BHSA","year":2019,"lang":"en","type":"article","venue":"HIPHIL Novum","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Learning Partnership","funders":"","keywords":"Linguistics; Hebrew; Computer science; Hebrew Bible; Syntax; Biblical languages; Grammar; Semantics (computer science); Interface (matter); Interpretation (philosophy); Artificial intelligence; Biblical studies; Literature; Philosophy; Art; Programming language","score_opus":0.010003490996676682,"score_gpt":0.2611827599101989,"score_spread":0.2511792689135222,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3121601712","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14384283,0.0434743,0.53743947,0.03200256,0.0029102936,0.0006258278,0.0013841623,0.01700134,0.22131924],"genre_scores_gemma":[0.26381192,0.01998244,0.5626865,0.0028265547,0.0012218789,0.00039646396,0.0032019122,0.0029957613,0.14287657],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9987783,0.000539905,0.000088222834,0.00012102262,0.0003810658,0.00009146477],"domain_scores_gemma":[0.997931,0.0010065358,0.000042521933,0.00031231157,0.00036610587,0.0003416821],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003485269,0.0006344365,0.0003607432,0.00095962477,0.00081747887,0.0037954866,0.0011691716,0.0020489127,0.036758598],"category_scores_gemma":[0.00459113,0.00038763517,0.00047905336,0.0011264221,0.0015029502,0.007641089,0.0028665452,0.0016834434,0.00785712],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00052724715,0.00021333655,0.0022248544,0.00098463,0.000029836447,0.0007065156,0.016009917,0.0007130681,0.03329304,0.04881043,0.03948268,0.85700446],"study_design_scores_gemma":[0.000086337604,0.00041139373,0.0044370336,0.0014014611,0.000045208104,0.0017380295,0.00822611,0.009385414,0.011359573,0.038874023,0.9239013,0.00013422339],"about_ca_topic_score_codex":0.0020287645,"about_ca_topic_score_gemma":0.0026628294,"teacher_disagreement_score":0.036758598,"about_ca_system_score_codex":0.0008007475,"about_ca_system_score_gemma":0.000660052,"threshold_uncertainty_score":0.12296975},"labels":[],"label_agreement":null},{"id":"W3121743396","doi":"","title":"O contributo da noção de plano de texto para a formação de tradutores","year":2019,"lang":"pt","type":"article","venue":"Revista de Estudos Anglo-Portugueses/Journal of Anglo-Portuguese Studies","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Linguistic Association","funders":"","keywords":"Philosophy","score_opus":0.04886953919867267,"score_gpt":0.34775663920965727,"score_spread":0.2988871000109846,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3121743396","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.43783456,0.008213014,0.2422711,0.026578883,0.0011461613,0.0010392204,0.0017204056,0.0030599898,0.27813673],"genre_scores_gemma":[0.89492756,0.0014792582,0.088761576,0.00062288716,0.00017783199,0.0002216118,0.00034147178,0.0008931144,0.012574739],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.96477836,0.019059764,0.0024535346,0.0043529696,0.008324712,0.0010307133],"domain_scores_gemma":[0.8412314,0.10744533,0.010875954,0.019189987,0.0187094,0.002547957],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.023862904,0.0007493079,0.00057332806,0.0043470436,0.004246881,0.015485587,0.0027142265,0.0020681394,0.013549806],"category_scores_gemma":[0.12872785,0.0012500493,0.0007451249,0.004610778,0.008502652,0.016979244,0.00487489,0.002922435,0.003049023],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00072314584,0.0002472048,0.052447036,0.0028338924,0.00012836911,0.00067509856,0.13749725,0.0022737354,0.014340748,0.17488272,0.010395123,0.6035557],"study_design_scores_gemma":[0.00023464246,0.00095830625,0.08652451,0.0051199235,0.0009891996,0.0030548226,0.14103428,0.026439063,0.03832293,0.18655649,0.5103435,0.00042239268],"about_ca_topic_score_codex":0.012345702,"about_ca_topic_score_gemma":0.014011002,"teacher_disagreement_score":0.023862904,"about_ca_system_score_codex":0.0050104456,"about_ca_system_score_gemma":0.009606312,"threshold_uncertainty_score":0.12620056},"labels":[],"label_agreement":null},{"id":"W3122399969","doi":"10.1007/s10579-020-09519-z","title":"Exploring the role of lexis and grammar for the stable identification of register in an unrestricted corpus of web documents","year":2021,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University; Brockhouse Institute for Materials Research; Brock University","funders":"National Science Foundation of Sri Lanka; Social Sciences and Humanities Research Council of Canada; Turun Yliopisto; Emil Aaltosen Säätiö; AGE-WELL; National Science Foundation","keywords":"Lexis; Register (sociolinguistics); Computer science; Natural language processing; Variation (astronomy); Artificial intelligence; Corpus linguistics; Identification (biology); Linguistics; Text corpus","score_opus":0.05024759076246877,"score_gpt":0.3165752133910301,"score_spread":0.26632762262856136,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3122399969","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.74908245,0.002278871,0.22822389,0.005804178,0.00013401487,0.00021147594,0.001898811,0.0013280645,0.01103816],"genre_scores_gemma":[0.9544843,0.0004414921,0.040895242,0.00020979805,0.00006928164,0.0001268929,0.0021465234,0.000320802,0.0013057092],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9966619,0.002075292,0.0001680064,0.0007769508,0.0002063381,0.0001114537],"domain_scores_gemma":[0.96832216,0.026222484,0.0014701281,0.0024877246,0.0011075392,0.00038987646],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0093822945,0.0008739304,0.0008480606,0.0035224175,0.001760971,0.005902129,0.0013116822,0.0014121163,0.0024349797],"category_scores_gemma":[0.045744207,0.0007533643,0.0013115305,0.0021675392,0.0032064586,0.009930177,0.002618184,0.0036517715,0.0011714369],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001468755,0.000623674,0.24601637,0.0008695507,0.00096391677,0.0013044752,0.014271765,0.21931721,0.015526873,0.11786512,0.011755605,0.37001666],"study_design_scores_gemma":[0.000041722968,0.000088275396,0.016617645,0.00011382177,0.00012043375,0.00020283823,0.0013933112,0.8837363,0.002129802,0.09122659,0.004254509,0.00007474152],"about_ca_topic_score_codex":0.013598489,"about_ca_topic_score_gemma":0.020875715,"teacher_disagreement_score":0.013598489,"about_ca_system_score_codex":0.0019623954,"about_ca_system_score_gemma":0.002071826,"threshold_uncertainty_score":0.0496189},"labels":[],"label_agreement":null},{"id":"W3123290982","doi":"10.18653/v1/2021.eacl-main.117","title":"Syntactic Nuclei in Dependency Parsing – A Multilingual Exploration","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Atomic Energy of Canada Limited; China Scholarship Council; Vetenskapsrådet","keywords":"Dependency (UML); Dependency grammar; Parsing; Computer science; Natural language processing; Artificial intelligence; Conjunction (astronomy); Nucleus; Psychology; Physics","score_opus":0.03580973746419729,"score_gpt":0.31571945113486016,"score_spread":0.27990971367066286,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3123290982","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03590866,0.00083350076,0.95367116,0.0006939914,0.000036972284,0.000038382008,0.00019394755,0.0015565222,0.0070668375],"genre_scores_gemma":[0.5651404,0.0017773841,0.42621043,0.0002653013,0.000080197264,0.00014372331,0.0007303412,0.0015144898,0.004137726],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9978242,0.0013176509,0.00008831315,0.00045238482,0.00022728427,0.00009016291],"domain_scores_gemma":[0.9960502,0.0027265183,0.00020029824,0.0006456402,0.00027862808,0.000098733115],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035302876,0.0007620542,0.0008499074,0.0014846121,0.0014004576,0.0038019954,0.0011603417,0.0009908868,0.003930033],"category_scores_gemma":[0.0054327655,0.001355909,0.0013087409,0.0018096243,0.0027580617,0.015696354,0.0039447793,0.0018476484,0.0013838368],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00032524654,0.00008665176,0.0058826767,0.00042729048,0.00013920132,0.00087544974,0.005343248,0.04116086,0.010226981,0.71025115,0.0039651967,0.221316],"study_design_scores_gemma":[0.00002398148,0.00006198134,0.0012512561,0.00012712786,0.00012023295,0.00063491595,0.0010592898,0.24001807,0.010820239,0.71147215,0.03431691,0.00009391045],"about_ca_topic_score_codex":0.0041546356,"about_ca_topic_score_gemma":0.005574258,"teacher_disagreement_score":0.0041546356,"about_ca_system_score_codex":0.0014081829,"about_ca_system_score_gemma":0.0015959425,"threshold_uncertainty_score":0.018670201},"labels":[],"label_agreement":null},{"id":"W3123455859","doi":"10.33011/computel.v2i.985","title":"A Digital Corpus of St. Lawrence Island Yupik","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Pipeline (software); Spell; Indigenous language; Indigenous; Linguistics; Computer science; Sociology; Anthropology","score_opus":0.013111800238786138,"score_gpt":0.2611552939209713,"score_spread":0.24804349368218517,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3123455859","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.24427177,0.0021557454,0.0040672026,0.0008704289,0.00020523256,0.0004710815,0.6612255,0.0013310121,0.08540207],"genre_scores_gemma":[0.17468639,0.001117366,0.014395089,0.0001661735,0.000036558522,0.0010295341,0.7792603,0.00084804476,0.028460598],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99942833,0.00007725074,0.00007741107,0.00017190404,0.0001691389,0.00007585128],"domain_scores_gemma":[0.9969778,0.00077302055,0.00023579235,0.000523856,0.0013167644,0.00017284746],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00053269445,0.00033657762,0.00032424938,0.0056248456,0.0019696117,0.0014124803,0.00080581714,0.0005427827,0.013820858],"category_scores_gemma":[0.003233042,0.00028643076,0.00014843256,0.011153628,0.0012136665,0.00079224835,0.0021619166,0.0006742721,0.006079882],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006987442,0.00014687877,0.027066836,0.005776265,0.00009105319,0.0038742619,0.048212234,0.0025501736,0.028645243,0.014984099,0.57084507,0.29710913],"study_design_scores_gemma":[0.00003144409,0.000017315813,0.1287786,0.00044222432,0.0000346048,0.00043976694,0.0102939345,0.0007692116,0.0047821533,0.0009466482,0.85340506,0.000058959362],"about_ca_topic_score_codex":0.2734634,"about_ca_topic_score_gemma":0.53446424,"teacher_disagreement_score":0.2734634,"about_ca_system_score_codex":0.0038763005,"about_ca_system_score_gemma":0.005499076,"threshold_uncertainty_score":0.5437434},"labels":[],"label_agreement":null},{"id":"W3124063403","doi":"10.1080/19331680802149640","title":"Automatic Annotation of Semantic Fields for Political Science Research","year":2008,"lang":"en","type":"article","venue":"Journal of Information Technology & Politics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kellogg's (Canada)","funders":"","keywords":"Annotation; Computer science; Rhetorical question; Cluster analysis; Strengths and weaknesses; Politics; Artificial intelligence; Natural language processing; Semantic field; Information retrieval; Data science; Linguistics; Political science; Psychology","score_opus":0.029909371152053877,"score_gpt":0.36597348591931433,"score_spread":0.33606411476726045,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3124063403","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09019123,0.0011290382,0.8547144,0.0038029237,0.00058698707,0.0011836231,0.007386555,0.009284416,0.03172089],"genre_scores_gemma":[0.18411005,0.00030749556,0.8012519,0.00017311743,0.00015635815,0.0011719001,0.008197454,0.000722952,0.003908773],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9848104,0.009628035,0.0009590905,0.0012605822,0.0030317125,0.00031009788],"domain_scores_gemma":[0.9140298,0.05025488,0.005069146,0.012734847,0.016947608,0.0009636834],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012798474,0.0006470374,0.0006826743,0.014346063,0.0033061504,0.004388974,0.0014178393,0.0012568993,0.006671408],"category_scores_gemma":[0.04833902,0.0005251609,0.00053640775,0.0092545375,0.0018472695,0.0047538774,0.0036535212,0.00193694,0.003031996],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00049129385,0.00042979553,0.016936535,0.0017908611,0.00006058393,0.00030272963,0.008596601,0.005517827,0.066684306,0.09136332,0.049624223,0.7582019],"study_design_scores_gemma":[0.00020001095,0.00015391724,0.036279496,0.0009307882,0.00011056594,0.0008099313,0.008525072,0.20056522,0.12307436,0.19634978,0.43271875,0.00028204208],"about_ca_topic_score_codex":0.0024537514,"about_ca_topic_score_gemma":0.005542227,"teacher_disagreement_score":0.014346063,"about_ca_system_score_codex":0.002571641,"about_ca_system_score_gemma":0.004467515,"threshold_uncertainty_score":0.067685604},"labels":[],"label_agreement":null},{"id":"W3126731857","doi":"10.5121/csit.2021.110109","title":"Megalite: A New Spanish Literature Corpus for NLP Tasks","year":2021,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Consejo Nacional de Ciencia y Tecnología","keywords":"Computer science; Natural language processing; Artificial intelligence; Stylometry; Rhetorical question; Corpus linguistics; Treebank; Natural language generation; Narrative; Text corpus; Linguistics; Annotation; Natural language","score_opus":0.013308060389277213,"score_gpt":0.2768090376117321,"score_spread":0.26350097722245486,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3126731857","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16062891,0.013426018,0.083381854,0.0026282542,0.0025939087,0.0029916924,0.6324442,0.0051805144,0.09672462],"genre_scores_gemma":[0.11358842,0.003604799,0.15074247,0.0005636322,0.00088177546,0.007044157,0.70330685,0.0021551284,0.018112725],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9986512,0.00035490748,0.00023294608,0.00030282937,0.0003833814,0.00007480528],"domain_scores_gemma":[0.9938339,0.0024592262,0.00035368826,0.00076764566,0.0022430045,0.0003426848],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018957854,0.00068834657,0.00072268693,0.01263466,0.0014597832,0.0019774907,0.0013413918,0.00079641526,0.02462739],"category_scores_gemma":[0.010153327,0.0002751883,0.00036135106,0.010946787,0.00072089606,0.0016382806,0.0022253648,0.00084452896,0.0077432506],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006180028,0.0003499382,0.010036574,0.0069112154,0.000104314684,0.0021420966,0.006031134,0.002329381,0.018483493,0.02082869,0.46686962,0.46529555],"study_design_scores_gemma":[0.0001493589,0.000049792547,0.01761475,0.0005239884,0.000043430533,0.000858302,0.001963959,0.0016789183,0.0034251155,0.0031484838,0.97049487,0.000049024933],"about_ca_topic_score_codex":0.005862453,"about_ca_topic_score_gemma":0.007999883,"teacher_disagreement_score":0.02462739,"about_ca_system_score_codex":0.0009631807,"about_ca_system_score_gemma":0.002517715,"threshold_uncertainty_score":0.08238679},"labels":[],"label_agreement":null},{"id":"W3127633255","doi":"10.1017/9781108886741","title":"English Dialect Dictionary Online: A New Departure in English Dialectology","year":2021,"lang":"en","type":"book","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Dialectology; Linguistics; Sociolinguistics; Australian English; Wright; Pragmatics; Lexicography; Historical linguistics; Phonetics; History; Pidgin; Computer science; Creole language","score_opus":0.010091402425225436,"score_gpt":0.24950699271329085,"score_spread":0.23941559028806542,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3127633255","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0029559457,0.026935956,0.051036973,0.020449795,0.010954992,0.00011379791,0.005583298,0.00420135,0.87776786],"genre_scores_gemma":[0.015964406,0.018894184,0.04923217,0.0076289168,0.0019960864,0.00019176515,0.0061764736,0.00686022,0.8930557],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99944836,0.00011137322,0.00006277728,0.00009385741,0.0002485555,0.00003508319],"domain_scores_gemma":[0.9985238,0.0006469584,0.000043253167,0.00018713767,0.0004155027,0.00018335189],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007599656,0.00047592577,0.00061390235,0.0025268148,0.0015615491,0.009826566,0.0009232167,0.0010891597,0.06880132],"category_scores_gemma":[0.002813766,0.00044726406,0.00030298613,0.0040783742,0.0015770737,0.012131467,0.0029328999,0.0029735197,0.046405744],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000013327302,0.000011318893,0.00023097468,0.00020871639,0.0000029727762,0.000099367695,0.002572801,0.000047294747,0.0007109102,0.12065683,0.71200305,0.16344239],"study_design_scores_gemma":[6.719011e-7,0.00000105385,0.00006051607,0.00003765661,4.4589123e-7,0.00006605393,0.0001864117,0.000016075863,0.000043612126,0.0015278433,0.99805754,0.0000019942147],"about_ca_topic_score_codex":0.003986823,"about_ca_topic_score_gemma":0.008969409,"teacher_disagreement_score":0.06880132,"about_ca_system_score_codex":0.0012433206,"about_ca_system_score_gemma":0.0020329093,"threshold_uncertainty_score":0.23016334},"labels":[],"label_agreement":null},{"id":"W3128846734","doi":"10.4000/ijcol.298","title":"An Exploration of Semantic Features in an Unsupervised Thematic Fit Evaluation Framework","year":2015,"lang":"en","type":"article","venue":"Italian Journal of Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Innovation Cluster (Canada)","funders":"","keywords":"Thematic map; Computer science; Natural language processing; Information retrieval; Artificial intelligence; Cartography; Geography","score_opus":0.0878316024480657,"score_gpt":0.38528677686459284,"score_spread":0.29745517441652713,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3128846734","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.120442145,0.00033915517,0.8689204,0.0008191421,0.00003413466,0.00024647795,0.0014187426,0.0030242587,0.0047555426],"genre_scores_gemma":[0.6793705,0.00006717584,0.31560144,0.00015194212,0.000042229596,0.00038983487,0.0026819583,0.00052547536,0.0011694683],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99156284,0.004782369,0.0004715603,0.0013948937,0.0014196335,0.00036877455],"domain_scores_gemma":[0.9854914,0.009749872,0.0007657485,0.0016137942,0.002032817,0.000346361],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.014930266,0.0011059669,0.0012817694,0.0054354873,0.0013452248,0.0043446645,0.0031597188,0.0020772407,0.0025149994],"category_scores_gemma":[0.027303757,0.00046246822,0.0014065211,0.004040462,0.0019459105,0.0076418705,0.0029546546,0.0033758697,0.0006962595],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011048962,0.001214898,0.034941763,0.000634918,0.000560077,0.00034348835,0.002336741,0.20547493,0.014274761,0.11875318,0.012469909,0.6078904],"study_design_scores_gemma":[0.000032014577,0.00014020168,0.0039589074,0.000046514346,0.000038761733,0.000085386615,0.00032997486,0.9398238,0.0036275696,0.049668413,0.0022038263,0.00004461993],"about_ca_topic_score_codex":0.00804622,"about_ca_topic_score_gemma":0.01684902,"teacher_disagreement_score":0.014930266,"about_ca_system_score_codex":0.0030257613,"about_ca_system_score_gemma":0.001904343,"threshold_uncertainty_score":0.0789597},"labels":[],"label_agreement":null},{"id":"W3129448568","doi":"10.1109/icdmw51313.2020.00114","title":"Immigration Document Classification and Automated Response Generation","year":2020,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Becton Dickinson (Canada)","funders":"","keywords":"Computer science; Categorization; Classifier (UML); Artificial intelligence; Machine learning; Contextual image classification; Information retrieval; Natural language processing; Image (mathematics)","score_opus":0.034619421895968706,"score_gpt":0.3160772426573347,"score_spread":0.281457820761366,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3129448568","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.069285996,0.0013778922,0.8942177,0.0024467388,0.00076951477,0.0013179337,0.0031506955,0.020302128,0.007131323],"genre_scores_gemma":[0.22330795,0.000445439,0.7585173,0.0005878973,0.0005841782,0.00062515575,0.0063733454,0.00046106384,0.009097589],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99551684,0.0019361095,0.00034361656,0.0009122128,0.0009307588,0.00036051974],"domain_scores_gemma":[0.9856125,0.008570801,0.0013071352,0.0015936175,0.00257939,0.00033667346],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037542863,0.0014557497,0.0012872986,0.004922842,0.0014791278,0.0021722596,0.0022260118,0.0020123338,0.006591576],"category_scores_gemma":[0.014897675,0.00043174016,0.0010456668,0.003631091,0.00075047556,0.0021791758,0.0011741198,0.0016641896,0.005499415],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00046017318,0.00048344632,0.003453729,0.0004856589,0.000040751725,0.00037999733,0.00042702744,0.024137735,0.013476689,0.006323618,0.028559923,0.9217713],"study_design_scores_gemma":[0.00014760374,0.00027593345,0.0040558456,0.00015527222,0.000089471316,0.0008727132,0.0010943965,0.88571596,0.03903984,0.027577681,0.040864676,0.000110557696],"about_ca_topic_score_codex":0.0049713603,"about_ca_topic_score_gemma":0.0049109417,"teacher_disagreement_score":0.006591576,"about_ca_system_score_codex":0.0012939797,"about_ca_system_score_gemma":0.0017358672,"threshold_uncertainty_score":0.022051036},"labels":[],"label_agreement":null},{"id":"W3131063663","doi":"","title":"On the use of linguistic similarities to improve Neural Machine Translation for African Languages","year":2021,"lang":"en","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Machine translation; Linguistics; Computer science; Translation (biology); Natural language processing; Artificial intelligence; Biology; Philosophy","score_opus":0.029939520325381403,"score_gpt":0.2667732652375862,"score_spread":0.2368337449122048,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3131063663","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.55619615,0.0069852388,0.39751056,0.0034219546,0.0008842684,0.00024747432,0.0007883792,0.003933308,0.030032635],"genre_scores_gemma":[0.8482271,0.0010854416,0.1422597,0.00043474152,0.0003047169,0.00008180182,0.0009137048,0.00038884263,0.0063040587],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990138,0.00047740224,0.00008147166,0.00019406392,0.0001534266,0.000079878104],"domain_scores_gemma":[0.9949456,0.0032827985,0.00026379508,0.0005649534,0.0008467462,0.00009616121],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002241667,0.0006042995,0.00082696497,0.0015060969,0.0009837628,0.0012364644,0.0009134076,0.0011356556,0.0056613227],"category_scores_gemma":[0.012658506,0.00026235436,0.0004940296,0.0018238575,0.00059033727,0.0028704049,0.0017254385,0.0010264941,0.001557321],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010161827,0.00031908246,0.00388861,0.00036023062,0.00020300207,0.0002624758,0.0004203674,0.073454544,0.02124685,0.017365973,0.00783598,0.87362677],"study_design_scores_gemma":[0.00017315059,0.0004246989,0.0037528423,0.00010248878,0.00023314185,0.0002560047,0.00042567833,0.9241466,0.02376878,0.038909614,0.0077630877,0.000043898395],"about_ca_topic_score_codex":0.0030354734,"about_ca_topic_score_gemma":0.0068871486,"teacher_disagreement_score":0.0056613227,"about_ca_system_score_codex":0.00045471083,"about_ca_system_score_gemma":0.00072835194,"threshold_uncertainty_score":0.018938959},"labels":[],"label_agreement":null},{"id":"W3131725513","doi":"","title":"KETG: A Knowledge Enhanced Text Generation Framework","year":2021,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Thinkpath Engineering Services (Canada)","funders":"","keywords":"Rhetorical question; Computer science; Text generation; Natural language processing; Artificial intelligence; Embedding; Knowledge graph; Rhetoric; Linguistics","score_opus":0.01904883074304768,"score_gpt":0.30257649583324636,"score_spread":0.28352766509019867,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3131725513","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0044034207,0.00030545832,0.97911865,0.0003646842,0.00011012914,0.0002305318,0.0008732874,0.011524239,0.003069539],"genre_scores_gemma":[0.1356219,0.0003291779,0.8523188,0.00035960108,0.000105249106,0.000333561,0.0040062326,0.0010396396,0.00588581],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993051,0.00023993671,0.000037454138,0.0001685593,0.00020487439,0.000044093227],"domain_scores_gemma":[0.99870527,0.0007419977,0.00007781862,0.00020312761,0.00020732742,0.000064416396],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014485264,0.0007181369,0.00046504234,0.0013783546,0.0004911555,0.0011046223,0.0017218958,0.00109357,0.006862585],"category_scores_gemma":[0.0038628385,0.0003289742,0.0008290736,0.00078674173,0.00073221093,0.0020912057,0.0016240637,0.0015928298,0.0024684914],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00032379356,0.00026867294,0.0009383387,0.0006166841,0.000113004106,0.00060139067,0.0005110763,0.10718945,0.02246603,0.05624494,0.039351095,0.7713756],"study_design_scores_gemma":[0.00011302324,0.000121784695,0.00039260092,0.00005920862,0.00007294629,0.0003733,0.00009846561,0.85534066,0.018307945,0.07429823,0.05077053,0.000051325347],"about_ca_topic_score_codex":0.0027810812,"about_ca_topic_score_gemma":0.004356831,"teacher_disagreement_score":0.006862585,"about_ca_system_score_codex":0.00067434506,"about_ca_system_score_gemma":0.0012242479,"threshold_uncertainty_score":0.022957623},"labels":[],"label_agreement":null},{"id":"W3132011002","doi":"10.48550/arxiv.2105.05541","title":"Evaluating Gender Bias in Natural Language Inference","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Debiasing; Gender bias; Task (project management); Premise; Inference; Computer science; Natural (archaeology); Natural language processing; Artificial intelligence; Psychology; Social psychology; Cognitive psychology; Linguistics; Geography","score_opus":0.19138479936833486,"score_gpt":0.2933840641489097,"score_spread":0.10199926478057486,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3132011002","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6894932,0.005444693,0.26943758,0.0055626538,0.0008757749,0.0005730543,0.0051506152,0.005559079,0.01790333],"genre_scores_gemma":[0.905234,0.0005091661,0.08352478,0.0010865054,0.00019137154,0.000196788,0.0067255297,0.00038472566,0.0021471241],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9749615,0.015826838,0.001183638,0.003388865,0.0040188176,0.00062032446],"domain_scores_gemma":[0.82972604,0.14740992,0.0046418537,0.010260453,0.006727359,0.0012343888],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04224915,0.0014403139,0.0009785873,0.0022725111,0.0011600074,0.0029616316,0.0025209517,0.0025369073,0.003593647],"category_scores_gemma":[0.13324507,0.0005247531,0.0009027683,0.0013806833,0.0019198541,0.0067012412,0.0033285632,0.0035470848,0.0018610178],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0043152953,0.001398495,0.18556437,0.002032031,0.0012101759,0.000720196,0.004537342,0.12675236,0.022623288,0.02695777,0.03456929,0.58931935],"study_design_scores_gemma":[0.00021853884,0.00095163856,0.034216348,0.0004393512,0.00026541637,0.0006345225,0.0025784047,0.8248323,0.0429447,0.076446585,0.016302587,0.00016955135],"about_ca_topic_score_codex":0.0055213626,"about_ca_topic_score_gemma":0.008333063,"teacher_disagreement_score":0.04224915,"about_ca_system_score_codex":0.0018825406,"about_ca_system_score_gemma":0.0018703262,"threshold_uncertainty_score":0.22343755},"labels":[],"label_agreement":null},{"id":"W3133076675","doi":"","title":"AriEL: Volume Coding for Sentence Generation Comparisons","year":2021,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Sherbrooke","funders":"","keywords":"Computer science; Benchmark (surveying); Language model; Artificial intelligence; Sampling (signal processing); Autoencoder; Coding (social sciences); Natural language processing; Machine learning; Data mining; Deep learning; Mathematics; Computer vision; Statistics","score_opus":0.04400621447422936,"score_gpt":0.3029627731544972,"score_spread":0.25895655868026785,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3133076675","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.054990087,0.0019333387,0.9125796,0.0006695077,0.0007055812,0.00045807293,0.004936838,0.014430535,0.009296454],"genre_scores_gemma":[0.43424618,0.0003988908,0.54005975,0.0005208196,0.00024057341,0.0011487924,0.014668214,0.003274155,0.0054425807],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99598926,0.0018189376,0.00031978454,0.0007606545,0.00087101874,0.00024040713],"domain_scores_gemma":[0.99113256,0.005869704,0.0003194204,0.0014720338,0.0010170249,0.00018924384],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0049708276,0.0012589121,0.00085990335,0.0026564433,0.00079598743,0.0021366917,0.0028884795,0.001951055,0.018859193],"category_scores_gemma":[0.028107438,0.00035255722,0.001092509,0.0015054488,0.0008867338,0.0041830502,0.0028596474,0.0024594644,0.0031235833],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019164025,0.00033305783,0.0027856634,0.0007596192,0.00024029163,0.00026023778,0.00040867424,0.15814672,0.011071313,0.05934055,0.039913025,0.7248244],"study_design_scores_gemma":[0.00025572328,0.00054810714,0.001076358,0.000091087866,0.000040536033,0.00019474766,0.00018452057,0.8843219,0.013086971,0.08786459,0.012267644,0.00006778468],"about_ca_topic_score_codex":0.0022056974,"about_ca_topic_score_gemma":0.003663197,"teacher_disagreement_score":0.018859193,"about_ca_system_score_codex":0.0013292193,"about_ca_system_score_gemma":0.0012530396,"threshold_uncertainty_score":0.063090265},"labels":[],"label_agreement":null},{"id":"W3133717081","doi":"","title":"Approche statistique pour le repérage de mots informatifs dans les textes oraux","year":2004,"lang":"fr","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Humanities; Philosophy; Art","score_opus":0.017785263396566806,"score_gpt":0.2767911436136335,"score_spread":0.25900588021706666,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3133717081","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016206563,0.0011767491,0.9672538,0.00074104074,0.00020273858,0.00031738653,0.0020382006,0.008262714,0.0038008012],"genre_scores_gemma":[0.09719273,0.00083258824,0.888311,0.00023245893,0.00013638307,0.0007534429,0.0050396025,0.0011283099,0.006373337],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9851741,0.003727229,0.001823997,0.003188366,0.0057587884,0.0003274734],"domain_scores_gemma":[0.957707,0.026399517,0.0022014668,0.0054354295,0.0079226205,0.0003339335],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00884144,0.0018097019,0.0022834928,0.009550231,0.0019636222,0.008218414,0.0022125219,0.0029203824,0.0102047],"category_scores_gemma":[0.043759428,0.00137704,0.002592324,0.008718375,0.0019839646,0.007868632,0.002674729,0.003244731,0.00779519],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011980894,0.00023448528,0.016843233,0.001797747,0.00075953815,0.00048555856,0.0023324501,0.028755525,0.060854636,0.024491299,0.009898332,0.85234904],"study_design_scores_gemma":[0.00032040387,0.00058844296,0.021255791,0.00074248377,0.00086769054,0.0018129294,0.002906355,0.63367563,0.116059095,0.088368624,0.13296212,0.0004405028],"about_ca_topic_score_codex":0.008739343,"about_ca_topic_score_gemma":0.01575671,"teacher_disagreement_score":0.0102047,"about_ca_system_score_codex":0.0020049051,"about_ca_system_score_gemma":0.0036631532,"threshold_uncertainty_score":0.046758592},"labels":[],"label_agreement":null},{"id":"W3134519208","doi":"10.33011/computel.v1i.949","title":"Expanding the JHU Bible Corpus for Machine Translation of the Indigenous Languages of North America","year":2021,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Indigenous; Machine translation; Variety (cybernetics); Translation (biology); Computer science; Linguistics; Natural language processing; Artificial intelligence; Resource (disambiguation); Philosophy; Biology","score_opus":0.014575739387418834,"score_gpt":0.28071564855940984,"score_spread":0.266139909171991,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3134519208","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.78909934,0.0042937044,0.0397464,0.002272792,0.0014897063,0.0009140512,0.078122206,0.0048765396,0.07918538],"genre_scores_gemma":[0.66898495,0.001533744,0.09436734,0.00073252665,0.0005352989,0.001494694,0.20501322,0.0017236389,0.0256146],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9987878,0.0003866819,0.00010330774,0.0002589534,0.00035810008,0.000105091734],"domain_scores_gemma":[0.9966537,0.000989624,0.00013519565,0.00071957445,0.0012429645,0.00025891513],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013350263,0.0005898153,0.00067722396,0.0029903925,0.0027327922,0.0012189391,0.0007191508,0.00046769678,0.0132485675],"category_scores_gemma":[0.0060091983,0.00034198668,0.00030504426,0.0036278937,0.0009103362,0.0011750686,0.0025271284,0.0011645494,0.0040563275],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00085568253,0.000659312,0.02637507,0.0021880073,0.0001874239,0.0021382945,0.013134562,0.005462439,0.06975743,0.009577246,0.20042591,0.66923857],"study_design_scores_gemma":[0.000423484,0.0002882507,0.20679972,0.00042376513,0.00022325765,0.0025753728,0.0070553203,0.02539401,0.042097177,0.0063821557,0.7081166,0.00022097229],"about_ca_topic_score_codex":0.03060237,"about_ca_topic_score_gemma":0.06995197,"teacher_disagreement_score":0.03060237,"about_ca_system_score_codex":0.0010762801,"about_ca_system_score_gemma":0.0030635174,"threshold_uncertainty_score":0.060848475},"labels":[],"label_agreement":null},{"id":"W3134981945","doi":"10.5539/ells.v11n1p51","title":"Application of Translation Technologies in the Translation of IMTFE Transcripts","year":2021,"lang":"en","type":"article","venue":"English Language and Literature Studies","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Machine translation; Computer science; Translation (biology); Computer-assisted translation; Artificial intelligence; Quality (philosophy); Example-based machine translation; Dynamic and formal equivalence; Rule-based machine translation; Natural language processing; Machine translation software usability; Cloud computing; Messenger RNA","score_opus":0.011756045257573003,"score_gpt":0.2770292942357446,"score_spread":0.2652732489781716,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3134981945","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.066297635,0.0045311763,0.86109763,0.006269058,0.0019233483,0.0006112309,0.0016051236,0.003078343,0.054586448],"genre_scores_gemma":[0.25095978,0.00497364,0.72611946,0.00090857456,0.00051343325,0.0004791972,0.002603553,0.00092451746,0.01251791],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9945368,0.0032190415,0.00051265425,0.0005506217,0.0010130883,0.00016779918],"domain_scores_gemma":[0.9881576,0.0067395833,0.00080192374,0.0014533849,0.002739922,0.00010752002],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0046522096,0.0008266392,0.00047235392,0.0037362953,0.0012698289,0.0030792244,0.0005859316,0.0010546483,0.0058579966],"category_scores_gemma":[0.018819742,0.00030366745,0.00086088135,0.004160375,0.0014351036,0.0029681637,0.0014455244,0.0016051327,0.004241051],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028533422,0.00016830834,0.0033518511,0.0025361252,0.00011000373,0.0021158878,0.011553849,0.0036812113,0.056749783,0.05708437,0.016977528,0.8453857],"study_design_scores_gemma":[0.00012199946,0.00077770837,0.015466523,0.0024853556,0.00036474867,0.007306915,0.020460678,0.060024668,0.19937204,0.10061077,0.59265435,0.00035425217],"about_ca_topic_score_codex":0.00079992117,"about_ca_topic_score_gemma":0.0007730158,"teacher_disagreement_score":0.0058579966,"about_ca_system_score_codex":0.000878488,"about_ca_system_score_gemma":0.0012495922,"threshold_uncertainty_score":0.024603546},"labels":[],"label_agreement":null},{"id":"W3135020526","doi":"","title":"Recherche contextuelle d'équivalents en banque de terminologie.","year":2010,"lang":"fr","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Humanities; Philosophy","score_opus":0.10506453448566039,"score_gpt":0.3643683572419204,"score_spread":0.25930382275626,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3135020526","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.29502144,0.01158944,0.6165354,0.003228212,0.0016887189,0.000620648,0.013513891,0.007101271,0.050700933],"genre_scores_gemma":[0.5437455,0.0033134145,0.414902,0.00047568028,0.00064665807,0.00045245813,0.020484332,0.00096206233,0.01501781],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.97094786,0.012167366,0.0021418803,0.003866143,0.010327034,0.000549697],"domain_scores_gemma":[0.95899034,0.027263781,0.0024895421,0.004408808,0.0060488638,0.0007987204],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01078444,0.0014830712,0.0019300578,0.00853306,0.0017073262,0.0070288284,0.0016432217,0.0018499881,0.011819963],"category_scores_gemma":[0.06625325,0.0005199215,0.0014704255,0.006268931,0.0018893922,0.0066678924,0.0021939415,0.003545319,0.005981971],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019572792,0.00050708244,0.029735975,0.0019397384,0.00039645465,0.0013047798,0.0038482184,0.008866036,0.024165472,0.123359114,0.025245436,0.7786745],"study_design_scores_gemma":[0.0005484012,0.0010370782,0.053047705,0.000956156,0.00035897162,0.012765715,0.008747253,0.13051146,0.057882626,0.11971777,0.61410916,0.0003177187],"about_ca_topic_score_codex":0.008682583,"about_ca_topic_score_gemma":0.0071767154,"teacher_disagreement_score":0.011819963,"about_ca_system_score_codex":0.0018313148,"about_ca_system_score_gemma":0.002467186,"threshold_uncertainty_score":0.057034194},"labels":[],"label_agreement":null},{"id":"W3135324665","doi":"10.33011/computel.v1i.971","title":"Computational Analysis versus Human Intuition: A Critical Comparison of Vector Semantics with Manual Semantic Classification in the Context of Plains Cree","year":2021,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; USable; Natural language processing; Artificial intelligence; Sophistication; Semantics (computer science); Context (archaeology); Intuition; Support vector machine; Data science; Geography; World Wide Web; Psychology; Cognitive science; Sociology; Archaeology; Programming language","score_opus":0.0455823190219722,"score_gpt":0.3708319930079069,"score_spread":0.32524967398593474,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3135324665","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.38150582,0.0041277665,0.39816216,0.021397362,0.00076979975,0.00059061695,0.0009487496,0.0011661903,0.19133158],"genre_scores_gemma":[0.8979068,0.00065921346,0.0958939,0.0009559629,0.00029032084,0.00032746914,0.0006417508,0.00051787583,0.0028068027],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.97297317,0.01874072,0.00089912076,0.0014973211,0.00531513,0.0005745509],"domain_scores_gemma":[0.7984745,0.17521568,0.0036941417,0.0120691825,0.009181324,0.001365156],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02865048,0.00097176863,0.00095691456,0.0079975445,0.0025179007,0.010030319,0.002748648,0.0021241454,0.010759944],"category_scores_gemma":[0.15845573,0.00050523016,0.00091497233,0.004467746,0.0126424655,0.020817773,0.006384239,0.0027376686,0.0011072772],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001064536,0.00033779803,0.011523676,0.00087870436,0.00015160779,0.00027822584,0.015496119,0.0066858116,0.0018094714,0.8148624,0.010074583,0.13683711],"study_design_scores_gemma":[0.00016681006,0.00042067148,0.015211195,0.0004557132,0.00006936774,0.000464591,0.020596208,0.12352935,0.0033422362,0.8055908,0.02999014,0.00016291266],"about_ca_topic_score_codex":0.0056403438,"about_ca_topic_score_gemma":0.0052770996,"teacher_disagreement_score":0.02865048,"about_ca_system_score_codex":0.003797236,"about_ca_system_score_gemma":0.0024889745,"threshold_uncertainty_score":0.15152001},"labels":[],"label_agreement":null},{"id":"W3135383608","doi":"10.26378/rnlael1429406","title":"Estudio comparativo de métodos de transcripción para corpus orales: el caso del español","year":2020,"lang":"es","type":"article","venue":"LA Referencia (Red Federada de Repositorios Institucionales de Publicaciones Científicas)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Transcription (linguistics); Annotation; Conversation; Natural language processing; Spoken language; Artificial intelligence; Software; Speech recognition; Automation; Corpus linguistics; Linguistics; Engineering","score_opus":0.036489910562743755,"score_gpt":0.28363535798384165,"score_spread":0.2471454474210979,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3135383608","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.719787,0.010590065,0.19522329,0.0022325576,0.0012585742,0.0020112873,0.004675714,0.0053027617,0.058918744],"genre_scores_gemma":[0.8044508,0.0039812922,0.15417062,0.0009053506,0.00025609677,0.0022783868,0.004545701,0.00344436,0.025967317],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9925713,0.003392863,0.00070874294,0.0012352029,0.0018341994,0.00025765796],"domain_scores_gemma":[0.962966,0.02514926,0.0007874845,0.002384781,0.008361499,0.00035094065],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009251119,0.0012017768,0.0007691834,0.003032188,0.001700571,0.0041550756,0.0012977726,0.0016029662,0.007844402],"category_scores_gemma":[0.053238984,0.0005870182,0.0007140573,0.003382928,0.0012546752,0.0037674496,0.002699481,0.001114011,0.0039214096],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002332173,0.0002886265,0.0283975,0.005202387,0.00038481975,0.0021236967,0.07317114,0.004366998,0.0999495,0.0041042506,0.013823118,0.7658557],"study_design_scores_gemma":[0.0005525605,0.0023912254,0.14586107,0.0035883333,0.0015246982,0.0069995937,0.16008578,0.060506627,0.16628082,0.016056404,0.43535435,0.0007985721],"about_ca_topic_score_codex":0.010966758,"about_ca_topic_score_gemma":0.0129797,"teacher_disagreement_score":0.010966758,"about_ca_system_score_codex":0.0014019448,"about_ca_system_score_gemma":0.0022157487,"threshold_uncertainty_score":0.04892516},"labels":[],"label_agreement":null},{"id":"W3135618437","doi":"10.21203/rs.3.rs-246079/v1","title":"Towards AI-powered Language Assessment Tools","year":2021,"lang":"en","type":"preprint","venue":"Research Square","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Natural language processing","score_opus":0.0667866618727071,"score_gpt":0.4603486543558466,"score_spread":0.3935619924831395,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3135618437","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0038517413,0.00030453163,0.9775474,0.0011089328,0.000089558605,0.00015432366,0.00015679936,0.004835093,0.011951515],"genre_scores_gemma":[0.09828513,0.0004986987,0.8884151,0.00052607583,0.00017159664,0.00042749377,0.0007882447,0.0009883041,0.009899395],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98954296,0.0053380914,0.00062879705,0.000866592,0.0033364615,0.00028710038],"domain_scores_gemma":[0.9701081,0.017178962,0.001005293,0.004690252,0.0061704405,0.00084704865],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010641405,0.001190003,0.00089934614,0.0044606784,0.0008755908,0.009379179,0.002937417,0.002263386,0.0118746925],"category_scores_gemma":[0.04548597,0.0007734874,0.00091873924,0.002717211,0.0025610083,0.012399751,0.0055087963,0.004011787,0.0066703185],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015708587,0.0003704228,0.0019434716,0.0006322681,0.00008657163,0.00025202564,0.0013996079,0.009455042,0.014284716,0.428256,0.017759074,0.5254037],"study_design_scores_gemma":[0.00005217043,0.00008211993,0.0006135364,0.00035258915,0.00005005358,0.00022931819,0.000692625,0.17077412,0.017959146,0.7333857,0.07575586,0.00005279824],"about_ca_topic_score_codex":0.0011469082,"about_ca_topic_score_gemma":0.0013267315,"teacher_disagreement_score":0.0118746925,"about_ca_system_score_codex":0.0011644934,"about_ca_system_score_gemma":0.0026574149,"threshold_uncertainty_score":0.05627781},"labels":[],"label_agreement":null},{"id":"W3137010024","doi":"10.1162/tacl_a_00447","title":"Quality at a Glance: An Audit of Web-Crawled Multilingual Datasets","year":2022,"lang":"en","type":"article","venue":"Transactions of the Association for Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":167,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; Google (Canada)","funders":"Agence Nationale de la Recherche","keywords":"Computer science; USable; Audit; Natural language processing; Quality (philosophy); Artificial intelligence; Information retrieval; World Wide Web; Accounting","score_opus":0.021970551721164733,"score_gpt":0.3306852318074935,"score_spread":0.30871468008632874,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3137010024","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.70204556,0.0058487775,0.059483822,0.009382596,0.0017443822,0.0026674161,0.11685233,0.0827925,0.019182699],"genre_scores_gemma":[0.51107514,0.0016432845,0.1286201,0.0025777775,0.0003854511,0.0020085403,0.3251124,0.01962509,0.00895226],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.89964235,0.028618874,0.016830813,0.010182744,0.042489927,0.0022352845],"domain_scores_gemma":[0.49999642,0.12704392,0.030558437,0.1397169,0.19572927,0.006955071],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.06712028,0.0013971047,0.0013271903,0.017156474,0.0034351202,0.007021653,0.0033195333,0.0016078869,0.0013709468],"category_scores_gemma":[0.21080942,0.0016156565,0.0012024564,0.01735238,0.0035907335,0.005904365,0.007036,0.0029385095,0.0030484057],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021886,0.0015372297,0.27039644,0.0046564774,0.0013479162,0.0026935723,0.014175259,0.006669717,0.039679494,0.004891357,0.30200702,0.349757],"study_design_scores_gemma":[0.00040336043,0.0008357639,0.46844193,0.0022615187,0.0007260713,0.0026223608,0.00593911,0.05185007,0.08352879,0.004778695,0.37783757,0.0007747139],"about_ca_topic_score_codex":0.020012474,"about_ca_topic_score_gemma":0.028378762,"teacher_disagreement_score":0.93287975,"about_ca_system_score_codex":0.0023211828,"about_ca_system_score_gemma":0.0056340466,"threshold_uncertainty_score":0.35497022},"labels":[],"label_agreement":null},{"id":"W3139165132","doi":"10.3765/plsa.v6i1.5022","title":"Common names and proper nouns: Morphosyntactic evidence of a complete nominal paradigm","year":2021,"lang":"en","type":"article","venue":"Proceedings of the Linguistic Society of America","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Proper noun; Noun; Linguistics; Conflation; Determiner; Computer science; Mathematics; Natural language processing; Philosophy","score_opus":0.023098726633910577,"score_gpt":0.27645556193719156,"score_spread":0.253356835303281,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3139165132","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.70275027,0.0008240028,0.15429361,0.002652714,0.00010853994,0.000080355756,0.00058383617,0.00032175335,0.13838492],"genre_scores_gemma":[0.9902716,0.00012662492,0.00762285,0.00016064696,0.00003852344,0.000033346667,0.00018179379,0.00009058774,0.0014740508],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.9970619,0.0012115496,0.0002718695,0.0006110216,0.00072108465,0.00012260982],"domain_scores_gemma":[0.98715603,0.0074808714,0.0011380892,0.0025652768,0.0014139295,0.00024588295],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003195659,0.00026347183,0.00050483545,0.0010976023,0.0022002752,0.0028031014,0.0013142751,0.0010047776,0.005465624],"category_scores_gemma":[0.010200948,0.00048656095,0.00034478577,0.0012840678,0.0072554173,0.011824576,0.004082339,0.0014908414,0.0005764686],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00038285207,0.00007793823,0.007100331,0.00036263195,0.000041829157,0.000498299,0.024855327,0.0004062918,0.026017772,0.8888774,0.0011816501,0.05019758],"study_design_scores_gemma":[0.00018793029,0.0001929411,0.027767148,0.00014712113,0.00009700604,0.0018903344,0.016810862,0.007645449,0.012079462,0.8765038,0.05652468,0.00015315227],"about_ca_topic_score_codex":0.0018287906,"about_ca_topic_score_gemma":0.0031246864,"teacher_disagreement_score":0.005465624,"about_ca_system_score_codex":0.0009920738,"about_ca_system_score_gemma":0.0010062946,"threshold_uncertainty_score":0.01828432},"labels":[],"label_agreement":null},{"id":"W3139993938","doi":"10.18280/isi.260113","title":"Kannada to English Machine Translation Using Deep Neural Network","year":2021,"lang":"en","type":"article","venue":"Ingénierie des systèmes d information","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":43,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Kannada; Machine translation; Translation (biology); Computer science; Natural language processing; Artificial intelligence; Artificial neural network; Chemistry","score_opus":0.01545658833218301,"score_gpt":0.2468130811814459,"score_spread":0.2313564928492629,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3139993938","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17257117,0.0051697697,0.77974075,0.0012863702,0.0010746197,0.0001688656,0.0013453792,0.014725689,0.023917325],"genre_scores_gemma":[0.6984467,0.0015554643,0.27912164,0.00036903485,0.0001619143,0.0001725679,0.0035271212,0.00039426587,0.016251294],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996815,0.00010602916,0.000026486288,0.00009394047,0.00005006815,0.00004206197],"domain_scores_gemma":[0.9997528,0.000086957065,0.000019939944,0.000034153632,0.00009363305,0.000012488418],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00039813315,0.0006239943,0.00045904267,0.0005268945,0.00044231443,0.00075248966,0.00038550643,0.00048827654,0.0034563555],"category_scores_gemma":[0.0010511491,0.00017633088,0.00051326543,0.0006991439,0.00019281491,0.0009945916,0.0006178919,0.0007832812,0.0018326585],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00034202557,0.00022697663,0.0018543578,0.00048807508,0.00017461188,0.00076181337,0.00031548337,0.09272896,0.032728508,0.00788052,0.013889214,0.84860945],"study_design_scores_gemma":[0.000036609526,0.000169606,0.0012784568,0.00007586639,0.00008028853,0.0002406554,0.00020292563,0.9435633,0.027962998,0.011257203,0.0150976,0.000034441255],"about_ca_topic_score_codex":0.004894448,"about_ca_topic_score_gemma":0.007669325,"teacher_disagreement_score":0.004894448,"about_ca_system_score_codex":0.00038846495,"about_ca_system_score_gemma":0.00064076355,"threshold_uncertainty_score":0.011562645},"labels":[],"label_agreement":null},{"id":"W3140050815","doi":"10.18653/v1/2021.eacl-main.241","title":"Better Neural Machine Translation by Extracting Linguistic Information from BERT","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Natural Sciences and Engineering Research Council of Canada; Ministère de la Défense Nationale","keywords":"Computer science; Machine translation; Transformer; Embedding; Natural language processing; Artificial intelligence; Rule-based machine translation; Language model; Point (geometry); Syntax; Semantics (computer science); Programming language; Mathematics; Engineering","score_opus":0.013495445619936012,"score_gpt":0.26648486903626584,"score_spread":0.25298942341632985,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3140050815","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04905663,0.0006036874,0.9393232,0.0005593247,0.00018440244,0.00006377769,0.00051776506,0.005431223,0.004259947],"genre_scores_gemma":[0.54972845,0.0005929956,0.43755263,0.00046114347,0.00021460712,0.00017303042,0.0040587075,0.0012218485,0.0059966473],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99930537,0.000242403,0.000045317796,0.0002323669,0.00012707581,0.000047539135],"domain_scores_gemma":[0.99848443,0.0006668748,0.00011097311,0.00040779804,0.00028924466,0.000040748022],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011176888,0.0015447617,0.00086439936,0.0011003347,0.00047198037,0.0011733377,0.0007353433,0.0009927299,0.003648437],"category_scores_gemma":[0.006299948,0.000684518,0.00076972414,0.0013051492,0.00063053216,0.00373558,0.0017095023,0.0020393943,0.0031837167],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037051868,0.0002334726,0.0017067371,0.00035124057,0.00013869489,0.00035134997,0.00029710308,0.28493658,0.05846241,0.022551078,0.011999481,0.6186014],"study_design_scores_gemma":[0.0000282312,0.00008124884,0.00042471595,0.000022808423,0.00003218461,0.00007894323,0.00004025767,0.9530414,0.01072087,0.03095211,0.004552958,0.00002428875],"about_ca_topic_score_codex":0.0018124664,"about_ca_topic_score_gemma":0.0037672548,"teacher_disagreement_score":0.003648437,"about_ca_system_score_codex":0.00050143444,"about_ca_system_score_gemma":0.0007833768,"threshold_uncertainty_score":0.012205243},"labels":[],"label_agreement":null},{"id":"W3140085973","doi":"10.22541/au.160478473.32682470/v1","title":"Plain-Language Summaries: An Essential Component to Promote Knowledge Translation","year":2020,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba","funders":"","keywords":"Publishing; Work (physics); Public relations; Plain language; Component (thermodynamics); Sociology of scientific knowledge; Engineering ethics; Political science; Knowledge management; Sociology; Computer science; Social science; Engineering","score_opus":0.03893204951313923,"score_gpt":0.3280472257253029,"score_spread":0.2891151762121637,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3140085973","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009084691,0.012685426,0.63020223,0.1796301,0.029078027,0.0047998494,0.009560586,0.015619636,0.10933944],"genre_scores_gemma":[0.081798434,0.015751902,0.79912615,0.022387005,0.019140908,0.0048175245,0.012541212,0.009886384,0.034550462],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.8975541,0.06555473,0.013892569,0.003870447,0.018099146,0.0010290197],"domain_scores_gemma":[0.4773684,0.30876088,0.025092661,0.08511766,0.0975295,0.0061308076],"candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.09019391,0.0016524484,0.0018219278,0.011665132,0.003550447,0.02133389,0.0037667914,0.005765413,0.03752049],"category_scores_gemma":[0.4307084,0.0018656931,0.0010233909,0.009645246,0.005044916,0.032856163,0.013348723,0.009787253,0.055147298],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036272279,0.00029929727,0.0007067108,0.0067827897,0.00011039877,0.0005750896,0.025574131,0.0007358016,0.006228065,0.13479628,0.35338965,0.47043908],"study_design_scores_gemma":[0.000070639784,0.00010480334,0.0005752509,0.0028852986,0.000055689197,0.00034417806,0.004100703,0.0008734894,0.0022692045,0.071232684,0.91738784,0.00010023472],"about_ca_topic_score_codex":0.0009577551,"about_ca_topic_score_gemma":0.0012779295,"teacher_disagreement_score":0.9786661,"about_ca_system_score_codex":0.0028389692,"about_ca_system_score_gemma":0.012626714,"threshold_uncertainty_score":0.47699672},"labels":[],"label_agreement":null},{"id":"W3147221178","doi":"10.54590/pop.2020.007","title":"Digitizing Humanities in South Africa: Computational linguistic resources, training, and community building","year":2020,"lang":"en","type":"article","venue":"Pop! Public Open Participatory","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Digital humanities; Scarcity; Computer science; Computational linguistics; Field (mathematics); Data science; Knowledge management; World Wide Web; Artificial intelligence","score_opus":0.253500346954573,"score_gpt":0.3502769815862521,"score_spread":0.09677663463167907,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3147221178","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.33210456,0.013487997,0.12852503,0.17123541,0.0008096446,0.0043822513,0.006434134,0.0024734093,0.34054756],"genre_scores_gemma":[0.73020375,0.011750262,0.20327982,0.0030013714,0.00027842098,0.0029838113,0.00642741,0.0006068198,0.041468147],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9946091,0.003188203,0.0003828612,0.0005375141,0.00071623654,0.000566061],"domain_scores_gemma":[0.977509,0.013533388,0.0016852639,0.003282461,0.0018696354,0.002120321],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0101401415,0.00046010167,0.000499732,0.009658821,0.006085503,0.0072044106,0.0026019895,0.0015595208,0.025505977],"category_scores_gemma":[0.036554843,0.0008627011,0.00041448718,0.01245737,0.004374797,0.016688827,0.01742489,0.0020814629,0.0036274528],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008033811,0.000215361,0.014331901,0.0016056824,0.000023739907,0.0013432531,0.09093042,0.0017119773,0.0032575345,0.09575633,0.045041203,0.7457023],"study_design_scores_gemma":[0.00006227476,0.000064373315,0.02861424,0.0025037935,0.000026267466,0.00089211704,0.11384322,0.0039580627,0.0031327307,0.07512797,0.77167004,0.000104863255],"about_ca_topic_score_codex":0.03143247,"about_ca_topic_score_gemma":0.04648293,"teacher_disagreement_score":0.03143247,"about_ca_system_score_codex":0.005285326,"about_ca_system_score_gemma":0.017996803,"threshold_uncertainty_score":0.085326016},"labels":[],"label_agreement":null},{"id":"W3147742655","doi":"10.48550/arxiv.2103.04225","title":"Translating the Unseen? Yoruba-English MT in Low-Resource, Morphologically-Unmarked Settings","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Yoruba; Linguistics; Resource (disambiguation); Computer science; History; Artificial intelligence; Philosophy","score_opus":0.03069907811360257,"score_gpt":0.1864195074878402,"score_spread":0.15572042937423763,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3147742655","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.72006774,0.0011478562,0.23242195,0.002912594,0.0004700318,0.00013933165,0.0036193242,0.011792154,0.027428878],"genre_scores_gemma":[0.9379498,0.00021291019,0.054101408,0.00023683584,0.00003615383,0.00005792676,0.002758398,0.0006650941,0.0039815437],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9990283,0.00048381652,0.000057904068,0.00027330918,0.00009292148,0.00006375817],"domain_scores_gemma":[0.9976718,0.0013232464,0.00011254917,0.00039680605,0.00043406116,0.00006151384],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012637278,0.0008509467,0.00065151,0.00038794702,0.000678983,0.00179068,0.00072764326,0.0010674628,0.0038711317],"category_scores_gemma":[0.008836833,0.0003510209,0.00030512369,0.00060755975,0.0005830473,0.002814474,0.0011743227,0.0008406194,0.0023477862],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019556517,0.00054862094,0.020440849,0.002176252,0.00033175084,0.004200136,0.0064502447,0.18614903,0.16483511,0.04701292,0.04339914,0.52250034],"study_design_scores_gemma":[0.00014036705,0.00026211416,0.010597225,0.00021632845,0.00014894901,0.0011463267,0.00331292,0.8135267,0.07899545,0.050870694,0.040654592,0.00012831921],"about_ca_topic_score_codex":0.014442672,"about_ca_topic_score_gemma":0.024495486,"teacher_disagreement_score":0.014442672,"about_ca_system_score_codex":0.0009366683,"about_ca_system_score_gemma":0.00092914817,"threshold_uncertainty_score":0.02871722},"labels":[],"label_agreement":null},{"id":"W3148481662","doi":"10.5539/elt.v14n4p55","title":"Characteristics of Pronoun “Who and Its Concordance” in Chinese College Students’ English Narrative Writing from the Perspective of Corpus-Based Method - A Case Study of Series of Compositions of “The Most Unforgettable Person I Ever Know”","year":2021,"lang":"en","type":"article","venue":"English Language Teaching","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"Ministry of Education of the People's Republic of China","keywords":"Pronoun; Linguistics; Psychology; Sentence; Perspective (graphical); Adverbial; Attributive; Narrative; Computer science; Artificial intelligence","score_opus":0.00881204284776729,"score_gpt":0.2992193224155203,"score_spread":0.29040727956775303,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3148481662","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99425006,0.00022134998,0.0006819038,0.00011265757,0.00001094485,0.000024051118,0.00007602828,0.000009507337,0.0046135234],"genre_scores_gemma":[0.9974775,0.00015018744,0.0008962719,0.000027827196,0.000007972336,0.00002016181,0.00013741248,0.000012018322,0.0012707033],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9981583,0.00068721216,0.00026133205,0.0003123163,0.00048367327,0.00009725655],"domain_scores_gemma":[0.9932591,0.0031090877,0.0014064533,0.00047203084,0.0014528756,0.00030061114],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018020287,0.00017142028,0.0001849009,0.002475995,0.0019136681,0.0016411688,0.00039868217,0.0003106579,0.0012298697],"category_scores_gemma":[0.008597572,0.00014754273,0.000118407304,0.0024978812,0.0014913447,0.0012178435,0.0009781311,0.00038939834,0.0001880205],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000120956116,0.000095073374,0.2766158,0.0004404096,0.00002646407,0.006101958,0.62405443,0.00013098422,0.015175928,0.0042082313,0.0014717202,0.07155803],"study_design_scores_gemma":[0.000008809327,0.00011406484,0.5517442,0.00022569326,0.0000626857,0.007987131,0.38468614,0.0018231223,0.012822632,0.0009926555,0.03946215,0.00007059324],"about_ca_topic_score_codex":0.008643688,"about_ca_topic_score_gemma":0.019906802,"teacher_disagreement_score":0.008643688,"about_ca_system_score_codex":0.001161909,"about_ca_system_score_gemma":0.001318894,"threshold_uncertainty_score":0.01718676},"labels":[],"label_agreement":null},{"id":"W3150672550","doi":"10.1007/978-3-030-72113-8_21","title":"An Argument Extraction Decoder in Open Information Extraction","year":2021,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Computer science; Predicate (mathematical logic); Argument (complex analysis); Information extraction; Artificial intelligence; Sentence; Natural language processing; Algorithm; Theoretical computer science; Programming language","score_opus":0.017036013368029224,"score_gpt":0.3120898653853281,"score_spread":0.2950538520172989,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3150672550","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0031171332,0.0003580381,0.9622306,0.00043051524,0.00030956845,0.00013604142,0.0012341747,0.02177822,0.010405755],"genre_scores_gemma":[0.06373523,0.0004495175,0.9012164,0.0005219815,0.00025533276,0.0002257384,0.0042610103,0.005206434,0.024128288],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99871194,0.00031596294,0.00017467295,0.00026343885,0.00044582487,0.000088196204],"domain_scores_gemma":[0.9957736,0.0023267958,0.000112447044,0.0008499065,0.0008549255,0.00008234627],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016923621,0.0010869763,0.0013510855,0.0023545118,0.0011059775,0.0033560153,0.0020473981,0.002204191,0.026605763],"category_scores_gemma":[0.007332091,0.0012039796,0.0011737266,0.002409055,0.001063861,0.005820575,0.004313756,0.0023160782,0.017529197],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003827627,0.00013580517,0.00058316934,0.00055029587,0.00005933313,0.00049921236,0.0004277481,0.0035832266,0.015122989,0.13360254,0.061226998,0.78382593],"study_design_scores_gemma":[0.00015384694,0.000109697656,0.0005006126,0.00035340176,0.00019633291,0.0011342149,0.00030112546,0.26680258,0.12283497,0.36370316,0.24379736,0.00011273451],"about_ca_topic_score_codex":0.00094691146,"about_ca_topic_score_gemma":0.0018892708,"teacher_disagreement_score":0.026605763,"about_ca_system_score_codex":0.0007797517,"about_ca_system_score_gemma":0.0016635901,"threshold_uncertainty_score":0.08900511},"labels":[],"label_agreement":null},{"id":"W3152613642","doi":"10.18653/v1/2021.gwc-1.1","title":"On Universal Colexifications","year":2021,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Machine Intelligence Institute","keywords":"Polysemy; WordNet; Computer science; Natural language processing; Linguistics; Term (time); Word (group theory); Artificial intelligence; Philosophy","score_opus":0.012291239442053457,"score_gpt":0.26592881667953255,"score_spread":0.2536375772374791,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3152613642","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8343172,0.004129543,0.10326147,0.0010045848,0.00016072574,0.00015223217,0.0041123964,0.0009943864,0.051867437],"genre_scores_gemma":[0.97939646,0.0008466318,0.01371116,0.00025058966,0.00006646027,0.000099103556,0.0033125093,0.00015980225,0.0021574085],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99374455,0.0013682991,0.00071826106,0.00201571,0.0013458651,0.0008073612],"domain_scores_gemma":[0.9766287,0.013871824,0.0025447132,0.004796741,0.0015392954,0.0006187001],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004466026,0.0005893256,0.0010428394,0.005737053,0.0027449436,0.0025993737,0.0009621891,0.00080703315,0.007128944],"category_scores_gemma":[0.023569958,0.000502348,0.00087336224,0.0069333734,0.007161293,0.008605599,0.006896979,0.0016279162,0.00084179564],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00092468446,0.000121265955,0.16716605,0.00127421,0.00027571802,0.0028243244,0.013007388,0.0051404983,0.014264084,0.4947727,0.0074415826,0.29278752],"study_design_scores_gemma":[0.0000890177,0.00017865392,0.18988553,0.0014930113,0.00025327486,0.008977464,0.0112124225,0.023949817,0.01871506,0.59953254,0.14543642,0.0002767938],"about_ca_topic_score_codex":0.00749522,"about_ca_topic_score_gemma":0.0054480555,"teacher_disagreement_score":0.00749522,"about_ca_system_score_codex":0.0015799482,"about_ca_system_score_gemma":0.0015690041,"threshold_uncertainty_score":0.023848712},"labels":[],"label_agreement":null},{"id":"W3152619767","doi":"10.18653/v1/2021.gwc-1.4","title":"Homonymy and Polysemy Detection with Multilingual Information","year":2021,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Machine Intelligence Institute","keywords":"Polysemy; Homonym (biology); Leverage (statistics); Natural language processing; Artificial intelligence; Computer science; Translation (biology); Linguistics; Philosophy","score_opus":0.005060451444931718,"score_gpt":0.23393946556380596,"score_spread":0.22887901411887424,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3152619767","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.042697728,0.0007452178,0.9501093,0.00045740962,0.00008863544,0.00010512587,0.0003887542,0.0007703995,0.004637555],"genre_scores_gemma":[0.46191564,0.00047542906,0.5329794,0.00020072119,0.00022080269,0.00018257093,0.0011188615,0.00021311255,0.0026934454],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9946613,0.0011282683,0.00048612506,0.0019274121,0.0015061427,0.00029075646],"domain_scores_gemma":[0.98756194,0.007355851,0.0013731318,0.0020337242,0.0013103314,0.00036500552],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035063233,0.0010516769,0.001164606,0.0061534736,0.0017406266,0.003368885,0.0016590283,0.001807381,0.0046408414],"category_scores_gemma":[0.01799502,0.00071293133,0.0013541244,0.0033593152,0.002879647,0.009115325,0.004645639,0.0019238153,0.0016399838],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00053901825,0.00042329216,0.023292687,0.0013730684,0.000414932,0.0018696449,0.002534377,0.015558726,0.05645816,0.21823315,0.008171127,0.6711318],"study_design_scores_gemma":[0.00005764404,0.00017355525,0.010178825,0.00022641571,0.00016673749,0.003299437,0.0013583311,0.39459947,0.040129967,0.5358688,0.013759324,0.00018149897],"about_ca_topic_score_codex":0.0011555529,"about_ca_topic_score_gemma":0.0022934373,"teacher_disagreement_score":0.0061534736,"about_ca_system_score_codex":0.00084215106,"about_ca_system_score_gemma":0.0015121857,"threshold_uncertainty_score":0.018543422},"labels":[],"label_agreement":null},{"id":"W3154002721","doi":"","title":"Identifying negative language transfer in learner errors using POS information.","year":2021,"lang":"en","type":"article","venue":"Workshop on Innovative Use of NLP for Building Educational Applications","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Mistake; Negative transfer; Language model; Natural language processing; Language transfer; Artificial intelligence; First language; Cache language model; Transfer (computing); Recurrent neural network; Artificial neural network; Speech recognition; n-gram; Natural language; Universal Networking Language; Linguistics; Comprehension approach","score_opus":0.04961372036844343,"score_gpt":0.36238472221477247,"score_spread":0.31277100184632906,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3154002721","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9341479,0.00034224198,0.05631784,0.00041394422,0.000160935,0.00010073273,0.001188087,0.0023511776,0.0049770353],"genre_scores_gemma":[0.9865129,0.000102681704,0.009976228,0.00010146674,0.000016843587,0.0000355713,0.0010944549,0.000121400786,0.0020384681],"study_design_codex":"observational","study_design_gemma":"not_applicable","domain_scores_codex":[0.9953461,0.0015386734,0.0004816427,0.0009508895,0.0014372251,0.0002454591],"domain_scores_gemma":[0.97367847,0.012787651,0.003880508,0.0031938516,0.0058449623,0.00061448984],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004061476,0.0009523025,0.00052568025,0.0015459931,0.0005556867,0.0012612483,0.0007883268,0.0010892407,0.0017087839],"category_scores_gemma":[0.029524338,0.00022219043,0.0003666626,0.00082556513,0.0006259797,0.0024925391,0.0016270711,0.0012227328,0.0019302195],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010271727,0.0006546037,0.49824855,0.0005621655,0.00026515723,0.0033206618,0.0045976155,0.009687923,0.060017176,0.0013645786,0.00820464,0.4120498],"study_design_scores_gemma":[0.000071432354,0.0013847161,0.2845087,0.000372243,0.00038883812,0.009392139,0.007794847,0.42389494,0.24148947,0.010994902,0.019396393,0.0003114558],"about_ca_topic_score_codex":0.002405274,"about_ca_topic_score_gemma":0.0042934315,"teacher_disagreement_score":0.004061476,"about_ca_system_score_codex":0.00044021374,"about_ca_system_score_gemma":0.00083448295,"threshold_uncertainty_score":0.021479368},"labels":[],"label_agreement":null},{"id":"W3155240016","doi":"10.2139/ssrn.3422328","title":"Cascaded Finite-State Chunk Parsing for Hindi Language","year":2019,"lang":"en","type":"article","venue":"SSRN Electronic Journal","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Fredericton; University of New Brunswick","funders":"","keywords":"Hindi; Parsing; Natural language processing; Computer science; Finite state; Artificial intelligence; State (computer science); Bottom-up parsing; Top-down parsing; Linguistics; Speech recognition; Programming language; Machine learning","score_opus":0.005949245411400663,"score_gpt":0.2557798451785226,"score_spread":0.24983059976712194,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3155240016","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06536804,0.00045511857,0.8943121,0.00033200035,0.000295998,0.00019542754,0.003158369,0.030371597,0.005511425],"genre_scores_gemma":[0.5567385,0.000212068,0.42755154,0.00019335546,0.00009839005,0.00020631988,0.0063322713,0.0016386118,0.007028962],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.999236,0.00016149801,0.00006668557,0.00028759046,0.00012725909,0.000121028555],"domain_scores_gemma":[0.9974589,0.0014222742,0.00008162904,0.00061337446,0.0003547032,0.000069286856],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011100888,0.0007260802,0.0010614754,0.0007564728,0.0010856402,0.0012694817,0.0020403322,0.0012582078,0.009225827],"category_scores_gemma":[0.0026202046,0.0008902009,0.0012041664,0.00077758083,0.00070858543,0.0032488864,0.0019571707,0.0016557375,0.003550484],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00176457,0.0005469981,0.003506687,0.00081973616,0.00021818008,0.001341133,0.0011812353,0.119431,0.05964757,0.04350807,0.034279376,0.7337554],"study_design_scores_gemma":[0.0000506533,0.00018330467,0.001216766,0.0000627976,0.00009952453,0.00019412025,0.00014663114,0.8803724,0.046819948,0.06383083,0.0069447192,0.000078326426],"about_ca_topic_score_codex":0.0071903598,"about_ca_topic_score_gemma":0.01770698,"teacher_disagreement_score":0.009225827,"about_ca_system_score_codex":0.0008761295,"about_ca_system_score_gemma":0.0021304626,"threshold_uncertainty_score":0.030863404},"labels":[],"label_agreement":null},{"id":"W3155549525","doi":"10.18653/v1/2021.eacl-main.144","title":"Dependency parsing with structure preserving embeddings","year":2021,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Samsung","keywords":"Parsing; Dependency grammar; Computer science; Dependency (UML); Computational linguistics; Volume (thermodynamics); Natural language processing; Artificial intelligence; Linguistics; Programming language; Philosophy; Physics","score_opus":0.006641823263054757,"score_gpt":0.24329361023428261,"score_spread":0.23665178697122785,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3155549525","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015739847,0.0023086595,0.93990564,0.0013081523,0.00047817332,0.00014296109,0.0035530368,0.018494824,0.018068705],"genre_scores_gemma":[0.24749929,0.0025671758,0.70936817,0.0005565911,0.0003665856,0.00024228271,0.017592603,0.0058349352,0.015972279],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9982817,0.00070133293,0.00015703509,0.0004148489,0.00029911782,0.00014583123],"domain_scores_gemma":[0.99718165,0.0011524179,0.000116323994,0.0010739375,0.00041919693,0.000056443365],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015639286,0.0011655927,0.0011284207,0.002229117,0.0009656404,0.0025742094,0.0013325778,0.0012552417,0.00882676],"category_scores_gemma":[0.0051561245,0.001524729,0.0018789109,0.0029844574,0.0009914194,0.007919702,0.0042650094,0.002233494,0.0065309964],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000489171,0.00026172266,0.0012806703,0.00085551455,0.00030411824,0.00060636434,0.0006257872,0.03207704,0.011038072,0.24073005,0.12415145,0.58757997],"study_design_scores_gemma":[0.00010668641,0.00008366768,0.0007605148,0.00016302851,0.00019116816,0.00045803955,0.0002391428,0.22440583,0.019298417,0.67969805,0.07449697,0.000098531695],"about_ca_topic_score_codex":0.0022890705,"about_ca_topic_score_gemma":0.0044292295,"teacher_disagreement_score":0.00882676,"about_ca_system_score_codex":0.0005897602,"about_ca_system_score_gemma":0.0011605927,"threshold_uncertainty_score":0.029528439},"labels":[],"label_agreement":null},{"id":"W3156228691","doi":"10.3758/s13428-021-01570-0","title":"Calculating semantic relatedness of lists of nouns using WordNet path length","year":2021,"lang":"en","type":"article","venue":"Behavior Research Methods","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Memorial University of Newfoundland","funders":"","keywords":"WordNet; Computer science; Noun; Natural language processing; Path (computing); Artificial intelligence; Semantic similarity; Information retrieval; Programming language","score_opus":0.25866170950767636,"score_gpt":0.5705119817345476,"score_spread":0.3118502722268713,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3156228691","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6859854,0.0022834982,0.26671407,0.00052416924,0.0002518056,0.00071959774,0.021441476,0.003885984,0.018194074],"genre_scores_gemma":[0.7068229,0.0007606354,0.26794037,0.000089624715,0.00007368097,0.00056313956,0.019922763,0.0003894,0.0034374592],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.998114,0.00044351604,0.0002847698,0.0005224368,0.0005195858,0.00011575739],"domain_scores_gemma":[0.99222505,0.0044349325,0.0007015261,0.000473049,0.0018570449,0.00030839653],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001550246,0.0007088211,0.00064582063,0.011693373,0.0014077071,0.0017518264,0.00081804424,0.00090003107,0.006434465],"category_scores_gemma":[0.013288291,0.00040800148,0.0010087341,0.007976297,0.00057568616,0.0037616056,0.0014208381,0.00077814603,0.0023880987],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020100386,0.0007903472,0.23153761,0.0036017804,0.0010455671,0.0012541697,0.0047144084,0.015607428,0.05689689,0.05108492,0.016955087,0.6145017],"study_design_scores_gemma":[0.00041653548,0.0013048111,0.30616245,0.0008638071,0.0019127843,0.003568239,0.007934524,0.33495486,0.04372477,0.22381403,0.074991934,0.00035131178],"about_ca_topic_score_codex":0.0069666454,"about_ca_topic_score_gemma":0.011605832,"teacher_disagreement_score":0.011693373,"about_ca_system_score_codex":0.0011995004,"about_ca_system_score_gemma":0.0019205031,"threshold_uncertainty_score":0.021525443},"labels":[],"label_agreement":null},{"id":"W3156371424","doi":"10.18653/v1/2021.americasnlp-1.30","title":"IndT5: A Text-to-Text Transformer for 10 Indigenous Languages","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Compute Canada","keywords":"Indigenous; Computer science; Transformer; Machine translation; Natural language processing; Economic shortage; Artificial intelligence; Linguistics; Engineering; Electrical engineering","score_opus":0.015616885146672617,"score_gpt":0.305233642109253,"score_spread":0.2896167569625804,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3156371424","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1345656,0.0030099028,0.1288579,0.0015131213,0.0020618727,0.0017498287,0.4953928,0.18476456,0.04808442],"genre_scores_gemma":[0.14920929,0.0005826194,0.10110657,0.0005917712,0.00013681481,0.0012135851,0.72626066,0.0047448007,0.016153922],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99856585,0.00033519094,0.0001630531,0.00049386616,0.00026218206,0.00017982771],"domain_scores_gemma":[0.99820554,0.0005979964,0.00011180233,0.00044559033,0.0004903808,0.00014858047],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015603661,0.0028354768,0.0008022957,0.0029288547,0.0013390081,0.0018826992,0.0018379319,0.0013088742,0.026227936],"category_scores_gemma":[0.0063716895,0.00071173004,0.0019728926,0.0020463618,0.00075659686,0.0044205915,0.0030641172,0.0025219135,0.026188426],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011412072,0.0005830431,0.011165188,0.0024173104,0.0003501635,0.000948062,0.0011766304,0.009418117,0.018215703,0.0063785203,0.6044646,0.34374142],"study_design_scores_gemma":[0.0006426897,0.0009230294,0.020226626,0.00057639374,0.00040783736,0.0020436842,0.0023042692,0.20803607,0.051996343,0.018148407,0.69426817,0.00042659018],"about_ca_topic_score_codex":0.021014953,"about_ca_topic_score_gemma":0.04171905,"teacher_disagreement_score":0.026227936,"about_ca_system_score_codex":0.0014961838,"about_ca_system_score_gemma":0.0029395348,"threshold_uncertainty_score":0.08774114},"labels":[],"label_agreement":null},{"id":"W3157027354","doi":"10.25073/2588-1086/vnucsce.231","title":"Adaptation in Statistical Machine Translation for Low-resource Domains in English-Vietnamese Language","year":2020,"lang":"en","type":"article","venue":"VNU Journal of Science Computer Science and Communication Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Machine translation; Vietnamese; Artificial intelligence; Natural language processing; Computer science; Evaluation of machine translation; Phrase; Domain adaptation; Domain (mathematical analysis); Example-based machine translation; Machine translation software usability; Translation (biology); Baseline (sea); Classifier (UML); Linguistics; Mathematics; Philosophy","score_opus":0.013271792697715633,"score_gpt":0.2582265781423414,"score_spread":0.24495478544462576,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3157027354","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020137656,0.000869154,0.9731545,0.00035069237,0.00016680357,0.000091765636,0.00019109156,0.0023455143,0.0026929309],"genre_scores_gemma":[0.3781291,0.001350949,0.6098571,0.0006086296,0.0003479038,0.00036866378,0.0016122159,0.00063868496,0.0070867846],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9980903,0.0009780412,0.00011928704,0.0004272002,0.00030903635,0.000076039825],"domain_scores_gemma":[0.99744225,0.0012192074,0.00015331122,0.00061211316,0.000520841,0.00005235958],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021074412,0.00063777296,0.00065655995,0.0009737126,0.00052325026,0.00079229445,0.0010314849,0.00052632653,0.0017435688],"category_scores_gemma":[0.0042455485,0.0003757037,0.00080910686,0.0013255809,0.00082480354,0.0014635065,0.001265111,0.0013163509,0.0016666118],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022375205,0.00019735562,0.0051353495,0.0004149912,0.00021639378,0.0005417837,0.0006250119,0.12961532,0.06146599,0.017299913,0.01002868,0.7742354],"study_design_scores_gemma":[0.000034625395,0.00016656463,0.004556332,0.000040355528,0.00006763192,0.0006655073,0.0002775705,0.89867836,0.041951608,0.023649514,0.029817976,0.00009392754],"about_ca_topic_score_codex":0.0037027048,"about_ca_topic_score_gemma":0.0037639202,"teacher_disagreement_score":0.0037027048,"about_ca_system_score_codex":0.00061684544,"about_ca_system_score_gemma":0.00093708374,"threshold_uncertainty_score":0.011145353},"labels":[],"label_agreement":null},{"id":"W3157655844","doi":"","title":"HITS-UKP at TAC KBP 2019: Entity Discovery and Linking Track.","year":2019,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Track (disk drive); Computer science; Operating system","score_opus":0.004329036183774147,"score_gpt":0.23979078066081963,"score_spread":0.2354617444770455,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3157655844","genre_codex":"dataset","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004564441,0.001040241,0.010039156,0.0018856408,0.0015112808,0.00044246047,0.91972846,0.04498371,0.015804531],"genre_scores_gemma":[0.0027986318,0.0001323257,0.00839497,0.00020941392,0.0000739348,0.00019974147,0.983021,0.0011099854,0.0040600384],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9932609,0.0013763113,0.00066209806,0.0010019555,0.002991499,0.0007071052],"domain_scores_gemma":[0.97519195,0.00619807,0.0011690038,0.0049495995,0.009753498,0.0027377855],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007164525,0.0037574193,0.0032665392,0.011553236,0.005346678,0.009344711,0.0059959856,0.0061022947,0.056319136],"category_scores_gemma":[0.03915536,0.0014834385,0.0014163636,0.012605567,0.001279954,0.011207204,0.006678057,0.005242414,0.084242545],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024422738,0.000099219935,0.00038678313,0.0004369822,0.000063108884,0.00013350025,0.000090289795,0.00051499717,0.00074190396,0.0014617827,0.9878131,0.00801407],"study_design_scores_gemma":[0.00085786235,0.00028458418,0.005714642,0.0005018399,0.00018590654,0.0004236236,0.0006586409,0.01650068,0.0075859874,0.011648488,0.95545125,0.00018649182],"about_ca_topic_score_codex":0.078891695,"about_ca_topic_score_gemma":0.09037265,"teacher_disagreement_score":0.078891695,"about_ca_system_score_codex":0.002931265,"about_ca_system_score_gemma":0.007370328,"threshold_uncertainty_score":0.18840629},"labels":[],"label_agreement":null},{"id":"W3157707325","doi":"10.5539/elt.v14n5p77","title":"Summarization in English as a Foreign Language: A Study Comparing Summary Performances to Summarizers’ Vocabulary Size","year":2021,"lang":"en","type":"article","venue":"English Language Teaching","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Vocabulary; Psychology; Language proficiency; Linguistics; Foreign language; Automatic summarization; Mathematics education; Computer science; Natural language processing","score_opus":0.00830816340101185,"score_gpt":0.2700364138330177,"score_spread":0.2617282504320058,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3157707325","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.999199,0.00009228044,0.0001716073,0.000010601835,0.0000026892747,0.00001006674,0.000028850565,0.000006519814,0.0004783277],"genre_scores_gemma":[0.9988306,0.00011224884,0.00034992414,0.000015970727,0.000009440124,0.000014123557,0.00011888426,0.000005249948,0.0005435744],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99881494,0.0003444575,0.00019758982,0.00021258237,0.0003473971,0.000083037135],"domain_scores_gemma":[0.9857927,0.0061712596,0.004669221,0.000548154,0.001974449,0.0008441139],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021209952,0.00044467233,0.00040702723,0.0009127376,0.00026131512,0.0012379669,0.00031389433,0.00037715168,0.0014735346],"category_scores_gemma":[0.01807355,0.00016137998,0.00035424516,0.00052047905,0.00033598454,0.0012829779,0.0006208074,0.00030548376,0.00043791867],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008846497,0.0009987169,0.86655307,0.00051542773,0.00040818978,0.0007719128,0.022662727,0.0004275551,0.024869043,0.00013551749,0.00047652458,0.08129676],"study_design_scores_gemma":[0.000022822738,0.0017413915,0.9873892,0.00004255529,0.00011159852,0.000360527,0.00598568,0.0005631636,0.002603882,0.00010426082,0.0010446489,0.000030179384],"about_ca_topic_score_codex":0.0008909158,"about_ca_topic_score_gemma":0.001036692,"teacher_disagreement_score":0.0021209952,"about_ca_system_score_codex":0.00021208747,"about_ca_system_score_gemma":0.00023040609,"threshold_uncertainty_score":0.011217058},"labels":[],"label_agreement":null},{"id":"W3157747653","doi":"","title":"A Baseline Fine-Grained Entity Extraction System for TAC-KBP2019.","year":2019,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Baseline (sea); Computer science; Extraction (chemistry); Information retrieval; Chemistry; Chromatography; Geology","score_opus":0.006244639929203409,"score_gpt":0.26448438999127327,"score_spread":0.25823975006206984,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3157747653","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.043041445,0.00506419,0.28270805,0.0018849316,0.0013365874,0.002208027,0.37012076,0.2602725,0.033363435],"genre_scores_gemma":[0.056887254,0.00064477226,0.26589954,0.0005030723,0.00011166332,0.0006379989,0.66004443,0.002786625,0.01248466],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99807966,0.00024127451,0.0002877254,0.0006680141,0.0005420022,0.00018139674],"domain_scores_gemma":[0.99575126,0.00074300345,0.0001946825,0.0014883328,0.0016131583,0.00020954676],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018903814,0.0017025911,0.0014899622,0.005411849,0.0017325109,0.002494272,0.0027733028,0.0023409764,0.018680941],"category_scores_gemma":[0.0080932705,0.0008787706,0.0010048317,0.005785867,0.0004868537,0.0058602886,0.0025087714,0.00170786,0.028791372],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009499243,0.0005387408,0.0031595337,0.001990937,0.00038645463,0.0009004473,0.000365132,0.0039081024,0.048630822,0.005145358,0.5850356,0.34898898],"study_design_scores_gemma":[0.0004925534,0.00046884277,0.020219848,0.0005003847,0.00054351246,0.0023909544,0.00096714596,0.149334,0.07065576,0.017036995,0.7371008,0.00028915145],"about_ca_topic_score_codex":0.025707068,"about_ca_topic_score_gemma":0.031467777,"teacher_disagreement_score":0.025707068,"about_ca_system_score_codex":0.0010432473,"about_ca_system_score_gemma":0.003414278,"threshold_uncertainty_score":0.06249392},"labels":[],"label_agreement":null},{"id":"W3158064982","doi":"10.3765/amp.v9i0.4940","title":"Learning French Liaison with Gradient Symbolic Representations: Errors, Predictions, Consequences","year":2021,"lang":"en","type":"article","venue":"Proceedings of the Annual Meetings on Phonology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Collocation (remote sensing); Linguistics; Word (group theory); Grammar; Computer science; Cryptographic nonce; Natural language processing; Psychology; Cognitive psychology; Artificial intelligence; Machine learning; Philosophy","score_opus":0.009114294725875724,"score_gpt":0.2513841410717786,"score_spread":0.24226984634590287,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3158064982","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.93916726,0.00006463455,0.055625454,0.000597832,0.000025446649,0.000027140617,0.000062713385,0.00035903516,0.0040705083],"genre_scores_gemma":[0.98994935,0.00004179135,0.009157577,0.00003472716,0.0000038913367,0.00001854402,0.000044859953,0.000032468484,0.0007167527],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993212,0.00030385875,0.000037847705,0.00012216179,0.000104974126,0.000109989385],"domain_scores_gemma":[0.9957301,0.0027706977,0.0003394341,0.0006151917,0.00034687368,0.00019763362],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014124735,0.00046033727,0.00037149718,0.00036838694,0.00037544433,0.0014806187,0.0007340911,0.0013531948,0.002148798],"category_scores_gemma":[0.01452434,0.00034161584,0.00056630676,0.00018096165,0.0017540826,0.0022535073,0.0011749001,0.0012791259,0.00029377756],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000950236,0.00045728774,0.117059186,0.00046966982,0.00025374827,0.0034752025,0.008349997,0.4837664,0.06920764,0.1511654,0.0022134986,0.16263181],"study_design_scores_gemma":[0.000084609186,0.00044673524,0.014171989,0.0000556893,0.00004807242,0.0005836581,0.0012261348,0.8379911,0.020216351,0.123498134,0.001588061,0.000089313144],"about_ca_topic_score_codex":0.005664161,"about_ca_topic_score_gemma":0.004113432,"teacher_disagreement_score":0.005664161,"about_ca_system_score_codex":0.0007448281,"about_ca_system_score_gemma":0.00070816954,"threshold_uncertainty_score":0.011262417},"labels":[],"label_agreement":null},{"id":"W3158101525","doi":"","title":"Building a Graph Representation of LOINC<sup>®</sup> to Facilitate its Alignment to French Terminologies.","year":2020,"lang":"en","type":"article","venue":"PubMed","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; Centre Hospitalier de l’Université de Montréal","funders":"","keywords":"Computer science; Natural language processing; Punctuation; Graph; Artificial intelligence; Machine translation; Translation (biology); Process (computing); Information retrieval; Programming language; Theoretical computer science","score_opus":0.06937073188817151,"score_gpt":0.2780700772869849,"score_spread":0.2086993453988134,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3158101525","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017888412,0.0015346584,0.87134,0.0014490788,0.00049875304,0.0006661488,0.043365803,0.018498057,0.044759076],"genre_scores_gemma":[0.087847024,0.0015573519,0.7999334,0.00045090815,0.00010298908,0.00081711705,0.09159722,0.0018629369,0.015831003],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9996136,0.000120537065,0.000047564503,0.00010488194,0.000089304725,0.000024185061],"domain_scores_gemma":[0.99907887,0.00035340432,0.000082545725,0.00011907029,0.00033515424,0.000030974046],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004807273,0.0006430991,0.0002824883,0.0032602854,0.0005425846,0.0016440918,0.000611771,0.0006946255,0.015617648],"category_scores_gemma":[0.002586921,0.00024708136,0.00081781007,0.002919025,0.000374891,0.0015479664,0.00066971354,0.000765816,0.005275663],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00033999968,0.00017717804,0.0028284646,0.004020488,0.00017273486,0.0014556119,0.002870621,0.016117541,0.035000205,0.32595327,0.14449081,0.46657303],"study_design_scores_gemma":[0.000054830485,0.00010594121,0.0018625199,0.00049317256,0.00011975118,0.00091732095,0.0006205859,0.04385575,0.012301297,0.06709669,0.87250656,0.00006545493],"about_ca_topic_score_codex":0.010919839,"about_ca_topic_score_gemma":0.016706623,"teacher_disagreement_score":0.015617648,"about_ca_system_score_codex":0.0009966048,"about_ca_system_score_gemma":0.0019371131,"threshold_uncertainty_score":0.052246213},"labels":[],"label_agreement":null},{"id":"W315919923","doi":"","title":"De la Chambre des communes à la chambre d'isolement : adaptabilité d'un système de traduction basé sur les segments de phrases","year":2006,"lang":"fr","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Humanities; Political science; Philosophy","score_opus":0.020691660768530587,"score_gpt":0.26166805099740936,"score_spread":0.2409763902288788,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W315919923","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.36249053,0.0010397907,0.6000407,0.0009489313,0.00023864061,0.00045701626,0.00084734807,0.02533304,0.008603959],"genre_scores_gemma":[0.62904423,0.0005310818,0.34812665,0.00029672935,0.000075750824,0.00040753273,0.0020511418,0.0024147888,0.017052151],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99735934,0.00056953606,0.00019221593,0.0010622207,0.0006505937,0.00016604757],"domain_scores_gemma":[0.9881142,0.0076593217,0.00065498875,0.001847996,0.0014505489,0.00027298316],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00348195,0.0011664358,0.0009083646,0.0013147196,0.0011855238,0.003273494,0.0014573405,0.0017721985,0.007307373],"category_scores_gemma":[0.0172693,0.0007850328,0.000805492,0.0012402612,0.0012860701,0.0044156546,0.0020607882,0.0015145598,0.0035497926],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018737,0.00024025263,0.008605419,0.0007595218,0.0002032678,0.00080708356,0.008452519,0.016799612,0.31420556,0.0041883304,0.0048004286,0.6390643],"study_design_scores_gemma":[0.00022877265,0.0012348429,0.02479682,0.00021484122,0.0005012638,0.0017414525,0.005290982,0.3793815,0.48750904,0.009206035,0.08962351,0.00027093958],"about_ca_topic_score_codex":0.0074146558,"about_ca_topic_score_gemma":0.0061727674,"teacher_disagreement_score":0.0074146558,"about_ca_system_score_codex":0.0006704495,"about_ca_system_score_gemma":0.0011053623,"threshold_uncertainty_score":0.024445593},"labels":[],"label_agreement":null},{"id":"W3159606340","doi":"10.5281/zenodo.3664807","title":"A good TACTIC for lexicographical work: football terms encoded in TEI Lex-0","year":2019,"lang":"en","type":"article","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Linguistic Association","funders":"European Commission","keywords":"Football; Lexicographical order; Work (physics); Computer science; Business; Advertising; History; Mathematics; Engineering; Combinatorics","score_opus":0.022197094333770442,"score_gpt":0.25484141208871103,"score_spread":0.2326443177549406,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3159606340","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06615228,0.0014472927,0.7299243,0.013258823,0.0015529884,0.00023580043,0.002579752,0.0047666198,0.18008205],"genre_scores_gemma":[0.5491368,0.00093535346,0.4099197,0.0026410748,0.00051737606,0.00018542806,0.0030766465,0.002399177,0.031188434],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99924755,0.0003234025,0.000083385996,0.00012559141,0.00014487281,0.00007520276],"domain_scores_gemma":[0.9984091,0.0005741035,0.000105906016,0.000568173,0.0002758094,0.000066938774],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009327769,0.00044563838,0.0004552191,0.0016563777,0.0015913319,0.0037926286,0.00090260536,0.0009290365,0.012589737],"category_scores_gemma":[0.004423514,0.00035199217,0.00038633603,0.0026422963,0.0023668658,0.005195181,0.0018962566,0.0020263174,0.0041501303],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00034824017,0.00005646143,0.002082775,0.00036358746,0.000031536034,0.00061675144,0.003682972,0.0012823428,0.018490389,0.75983846,0.030386789,0.18281977],"study_design_scores_gemma":[0.000055320306,0.00009386745,0.0018498695,0.0006063958,0.000051444444,0.0013951786,0.0051127607,0.016607137,0.025863688,0.51315194,0.43510765,0.00010479304],"about_ca_topic_score_codex":0.0027873619,"about_ca_topic_score_gemma":0.005255482,"teacher_disagreement_score":0.012589737,"about_ca_system_score_codex":0.00087842613,"about_ca_system_score_gemma":0.0008694109,"threshold_uncertainty_score":0.04211682},"labels":[],"label_agreement":null},{"id":"W3161148137","doi":"10.29173/pathfinder38","title":"Indians in the Database","year":2021,"lang":"en","type":"article","venue":"Pathfinder A Canadian Journal for Information Science Students and Early Career Professionals","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Indigenous; Terminology; Subject (documents); Vocabulary; Exploratory research; Feeling; Psychology; Sociology; Linguistics; Library science; Social psychology; Social science; Computer science","score_opus":0.033056637444206255,"score_gpt":0.3417646107089234,"score_spread":0.3087079732647171,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3161148137","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15345581,0.007373026,0.011611086,0.010569329,0.0021204425,0.00086878793,0.4385346,0.011142624,0.3643244],"genre_scores_gemma":[0.35719904,0.005527598,0.03174939,0.00492315,0.0006165599,0.00050596177,0.49332204,0.002197605,0.103958696],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9970414,0.00039664004,0.0004996887,0.00061142293,0.0010268788,0.0004239557],"domain_scores_gemma":[0.9938852,0.0011734903,0.00056829257,0.0018144149,0.0018691518,0.0006893411],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020564732,0.0003932971,0.00077353744,0.007179913,0.0034501115,0.007882798,0.0017915836,0.0006939449,0.047218412],"category_scores_gemma":[0.009446729,0.00035011585,0.0007436937,0.017766738,0.00087238004,0.0041442676,0.00377008,0.0015345308,0.02619601],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015041973,0.0003736196,0.071863316,0.0022623097,0.00016552334,0.0024155132,0.013621368,0.0007123074,0.007508016,0.049162123,0.54655176,0.30385992],"study_design_scores_gemma":[0.000023872764,0.000032605592,0.01798693,0.0001742409,0.00007455279,0.0006479257,0.006716287,0.00049023604,0.0017070186,0.00241755,0.96967655,0.000052163257],"about_ca_topic_score_codex":0.07762051,"about_ca_topic_score_gemma":0.07689862,"teacher_disagreement_score":0.07762051,"about_ca_system_score_codex":0.0026174083,"about_ca_system_score_gemma":0.0048197005,"threshold_uncertainty_score":0.15796131},"labels":[],"label_agreement":null},{"id":"W3161374759","doi":"10.18653/v1/2021.naacl-main.102","title":"Understanding by Understanding Not: Modeling Negation in Language Models","year":2021,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":40,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; Microsoft (Canada); Université de Montréal; Montreal Police Service","funders":"","keywords":"Negation; Computer science; Computational linguistics; Linguistics; Cognitive science; Programming language; Natural language processing; Artificial intelligence; Psychology; Philosophy","score_opus":0.12776608687194346,"score_gpt":0.29684829609161056,"score_spread":0.1690822092196671,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3161374759","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.066240855,0.0017061747,0.90565073,0.0064232126,0.0003763445,0.00017209814,0.0016668643,0.0012351965,0.016528497],"genre_scores_gemma":[0.73453265,0.0015363902,0.252381,0.0008991758,0.000200356,0.00024899075,0.0028284688,0.00027376082,0.007099261],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99939287,0.0003145195,0.00003520787,0.000118690325,0.00009105034,0.00004760709],"domain_scores_gemma":[0.9955949,0.0034099135,0.00020051202,0.00027237457,0.00038070529,0.00014158056],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018130852,0.00080122036,0.0007248234,0.00049548165,0.00076603726,0.0030077414,0.0017514486,0.0015772029,0.005015324],"category_scores_gemma":[0.010676519,0.0006422367,0.001573874,0.00047770902,0.0008687969,0.007828578,0.0015302604,0.0021471055,0.00079466525],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00056151574,0.00024306666,0.0046344474,0.0003991731,0.00023169296,0.0006040989,0.0018993424,0.5639757,0.0025503917,0.32535717,0.0147122955,0.08483113],"study_design_scores_gemma":[0.000034872493,0.000024151528,0.00011996059,0.000030369109,0.000043214117,0.000050202594,0.00011585495,0.84219676,0.0006164995,0.15341169,0.0033429642,0.000013495013],"about_ca_topic_score_codex":0.012664347,"about_ca_topic_score_gemma":0.023146063,"teacher_disagreement_score":0.012664347,"about_ca_system_score_codex":0.0010042427,"about_ca_system_score_gemma":0.0014796527,"threshold_uncertainty_score":0.025181293},"labels":[],"label_agreement":null},{"id":"W3161584748","doi":"10.1145/1597849.1384417","title":"Web-based dynamic learning through lexical chaining","year":2008,"lang":"en","type":"article","venue":"ACM SIGCSE Bulletin","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Chaining; Computer science; Forward chaining; Backward chaining; Artificial intelligence; Natural language processing; Web application; World Wide Web; Expert system; Psychology","score_opus":0.018063030882019274,"score_gpt":0.2688657226676718,"score_spread":0.2508026917856525,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3161584748","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07647415,0.00029570248,0.9020517,0.0003331671,0.00008557001,0.00022536035,0.00033705,0.010484219,0.009713041],"genre_scores_gemma":[0.62940633,0.00041931693,0.3606966,0.00022520349,0.000050222912,0.00028463584,0.0014472436,0.00053767447,0.006932725],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99914193,0.00027053343,0.000071198716,0.00023046706,0.00023545476,0.000050291823],"domain_scores_gemma":[0.99507695,0.003267115,0.00018385168,0.0007255895,0.0005720052,0.00017454084],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012457073,0.0004908295,0.0005664231,0.0012973802,0.00077206903,0.0013746376,0.0013489685,0.0006844276,0.008071149],"category_scores_gemma":[0.006271263,0.00034290276,0.00035168178,0.0017560859,0.0007971598,0.006505847,0.0023946078,0.0012816405,0.0026068636],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00049526105,0.0007843252,0.0031735948,0.0002969774,0.00005958111,0.00047874518,0.0008473107,0.028114552,0.029275972,0.01592837,0.005640688,0.91490465],"study_design_scores_gemma":[0.00013329196,0.00030420563,0.0017189609,0.00009940006,0.000089542715,0.00042036505,0.00078034285,0.7932532,0.06911603,0.10074397,0.033252843,0.00008781599],"about_ca_topic_score_codex":0.001968893,"about_ca_topic_score_gemma":0.0038339293,"teacher_disagreement_score":0.008071149,"about_ca_system_score_codex":0.0003473132,"about_ca_system_score_gemma":0.0007378496,"threshold_uncertainty_score":0.027000725},"labels":[],"label_agreement":null},{"id":"W3161994961","doi":"10.18653/v1/2021.calcs-1.6","title":"Exploring Text-to-Text Transformers for English to Hinglish Machine Translation with Synthetic Code-Mixing","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Compute Canada","keywords":"Computer science; Machine translation; Natural language processing; Transformer; Language model; Artificial intelligence; Exploit; Encoder; Code-switching; Dependency (UML); Code (set theory); Set (abstract data type); Programming language; Linguistics","score_opus":0.055835367287104046,"score_gpt":0.2782522463936166,"score_spread":0.2224168791065126,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3161994961","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.46048358,0.00063224876,0.5242585,0.0007812426,0.00011055155,0.00025883774,0.001329739,0.005057788,0.007087472],"genre_scores_gemma":[0.8695947,0.00018961837,0.12289442,0.00017212406,0.000030140038,0.00027965338,0.0036553715,0.00040908303,0.002774816],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99867886,0.0007831721,0.00006369267,0.000262111,0.00012956712,0.000082588806],"domain_scores_gemma":[0.99488044,0.0036421502,0.00020568367,0.0006814488,0.00042422445,0.000166079],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029201452,0.0010406224,0.00056481845,0.0008144167,0.0004991276,0.0011160685,0.0012271905,0.0008696338,0.0022338235],"category_scores_gemma":[0.012555646,0.00037168985,0.0007226315,0.00093296927,0.0008544535,0.0026702718,0.0015578787,0.001604979,0.0010396196],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00069000095,0.0007945912,0.0066794483,0.000555364,0.00015165088,0.00029194585,0.0009218714,0.71310914,0.02587608,0.024731262,0.006558726,0.21963991],"study_design_scores_gemma":[0.000051574512,0.000164037,0.00043594316,0.000013230053,0.000015523834,0.000052583553,0.00009896825,0.97697794,0.0110983625,0.009204592,0.0018730551,0.000014307309],"about_ca_topic_score_codex":0.004119316,"about_ca_topic_score_gemma":0.0073498753,"teacher_disagreement_score":0.004119316,"about_ca_system_score_codex":0.0011469367,"about_ca_system_score_gemma":0.0011599207,"threshold_uncertainty_score":0.015443444},"labels":[],"label_agreement":null},{"id":"W3163547676","doi":"10.18280/isi.260201","title":"ARALD: Arabic Annotation Using Linked Data","year":2021,"lang":"en","type":"article","venue":"Ingénierie des systèmes d information","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Linked data; Arabic; Semantic Web; Annotation; Natural language processing; Knowledge base; World Wide Web; Information retrieval; Artificial intelligence; Domain (mathematical analysis); Linguistics","score_opus":0.03849275984637792,"score_gpt":0.2909523724449043,"score_spread":0.2524596125985264,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3163547676","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019509519,0.0015882197,0.66466236,0.0023663964,0.0018814415,0.0015540577,0.06964648,0.10770918,0.13108234],"genre_scores_gemma":[0.10469686,0.0015939926,0.70443636,0.0009256872,0.00039375862,0.0015379753,0.116573535,0.010180336,0.05966149],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99746823,0.000684893,0.0003020132,0.00052545,0.00091406197,0.000105324085],"domain_scores_gemma":[0.9954196,0.00093647634,0.00036387984,0.0010363867,0.0020542818,0.00018942381],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020746742,0.001777663,0.0006039643,0.008092731,0.0025129477,0.0043012276,0.001313049,0.0010602226,0.03927557],"category_scores_gemma":[0.009967746,0.0005641419,0.0008230865,0.004801844,0.0008099442,0.0048778383,0.005312981,0.0015468186,0.032938965],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009841091,0.00022740819,0.002937657,0.002301808,0.00013893875,0.0022801606,0.004172062,0.005599248,0.02606949,0.049396772,0.26047406,0.64541835],"study_design_scores_gemma":[0.00005534535,0.000046491947,0.0017195232,0.00042745084,0.000053235206,0.00048652032,0.0011227932,0.022133924,0.027804403,0.021186695,0.924834,0.00012958523],"about_ca_topic_score_codex":0.0066227703,"about_ca_topic_score_gemma":0.0048308494,"teacher_disagreement_score":0.03927557,"about_ca_system_score_codex":0.0014667698,"about_ca_system_score_gemma":0.0020232662,"threshold_uncertainty_score":0.13138986},"labels":[],"label_agreement":null},{"id":"W3164004475","doi":"10.1111/lang.12452","title":"The Nuclear Word Family List: A List of the Most Frequent Family Members, Including Base and Affixed Words","year":2021,"lang":"en","type":"article","venue":"Language Learning","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":39,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Word (group theory); Construct (python library); Word list; Computer science; Linguistics; Natural language processing; Nuclear family; Psychology; Artificial intelligence; Programming language; Sociology","score_opus":0.012102389857315007,"score_gpt":0.25342631774986185,"score_spread":0.24132392789254684,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3164004475","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19293424,0.00348605,0.39566088,0.0010617225,0.0006231606,0.00094817975,0.30267113,0.04491798,0.05769665],"genre_scores_gemma":[0.21028079,0.0017149638,0.5230901,0.00021855342,0.00021291056,0.0010321153,0.23164818,0.0075465753,0.024255872],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9993563,0.000114539595,0.0001378934,0.00011371003,0.00022788804,0.000049692582],"domain_scores_gemma":[0.996021,0.0016563263,0.0003840931,0.00031951972,0.0014633166,0.0001558237],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00067431614,0.00092178665,0.0005400497,0.008309843,0.0016217243,0.001347123,0.00058326655,0.0004025046,0.04719878],"category_scores_gemma":[0.005814097,0.00042955516,0.00045837078,0.0045300466,0.00037608357,0.0018031492,0.00086606643,0.0005520202,0.02227639],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00067568256,0.00012679606,0.02912011,0.0019197806,0.00009266668,0.0012390285,0.0018666339,0.0015502234,0.044624623,0.013591378,0.1691843,0.7360087],"study_design_scores_gemma":[0.000086496315,0.00022291394,0.023909125,0.00066814286,0.00012567377,0.003452468,0.0013888371,0.008634863,0.03486111,0.013610638,0.9128248,0.00021489638],"about_ca_topic_score_codex":0.0024086838,"about_ca_topic_score_gemma":0.0035584946,"teacher_disagreement_score":0.04719878,"about_ca_system_score_codex":0.0005436085,"about_ca_system_score_gemma":0.001776888,"threshold_uncertainty_score":0.15789562},"labels":[],"label_agreement":null},{"id":"W3164620445","doi":"10.3233/978-1-58603-954-7-243","title":"Hybrid Machine Translation for Low- and Middle-Density Languages","year":2009,"lang":"en","type":"book-chapter","venue":"NATO science for peace and security series. D, Information and communication security","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Translation (biology); Computer science; Machine translation; Linguistics; Natural language processing; Philosophy; Chemistry","score_opus":0.013959333297330307,"score_gpt":0.2678080403004393,"score_spread":0.253848707003109,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3164620445","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.041417766,0.004622162,0.8596754,0.0007822585,0.00046532188,0.00021637807,0.0011601035,0.013002925,0.07865763],"genre_scores_gemma":[0.2415983,0.0018413016,0.70705384,0.0003216088,0.0002582204,0.0002849258,0.003888256,0.0021812906,0.04257221],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9993649,0.00016060831,0.00006500435,0.00014772963,0.0002168137,0.000044845754],"domain_scores_gemma":[0.9990067,0.00048567692,0.00004522655,0.00022670886,0.00019817051,0.00003756141],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00076053943,0.00075262156,0.0007768007,0.0014385572,0.0007422988,0.0030010126,0.0012011736,0.00093194935,0.011095021],"category_scores_gemma":[0.0015913895,0.00042158755,0.0005042328,0.0015051565,0.00055506715,0.003110216,0.0019028481,0.0008080624,0.0076808743],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028785027,0.00010178579,0.0009971062,0.0010969806,0.00009939873,0.0009038933,0.0011949794,0.008658474,0.08154086,0.12715918,0.02541033,0.7525492],"study_design_scores_gemma":[0.00013375546,0.00036876585,0.0019833348,0.00031264086,0.00014984173,0.0039013734,0.001003364,0.1867978,0.15600064,0.1532082,0.49598205,0.00015830895],"about_ca_topic_score_codex":0.0005636947,"about_ca_topic_score_gemma":0.0012764923,"teacher_disagreement_score":0.011095021,"about_ca_system_score_codex":0.0006917251,"about_ca_system_score_gemma":0.00057502097,"threshold_uncertainty_score":0.037116528},"labels":[],"label_agreement":null},{"id":"W3165121784","doi":"10.18653/v1/2021.americasnlp-1","title":"Proceedings of the First Workshop on Natural Language Processing for Indigenous Languages of the Americas","year":2021,"lang":"en","type":"paratext","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Social Sciences and Humanities Research Council of Canada; Consejo Nacional de Ciencia, Tecnología e Innovación Tecnológica; Natural Sciences and Engineering Research Council of Canada; European Commission; National Endowment for the Humanities; Compute Canada; Schweizerischer Nationalfonds zur Förderung der Wissenschaftlichen Forschung; Comisión Nacional para el Conocimiento y Uso de la Biodiversidad, Gobierno de México; Vetenskapsrådet; National Science Foundation","keywords":"Indigenous; Computer science; Natural language; Linguistics; Natural language processing; Natural (archaeology); Artificial intelligence; Programming language; History; Archaeology; Philosophy; Ecology","score_opus":0.011138441318437662,"score_gpt":0.2968246777198269,"score_spread":0.28568623640138924,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3165121784","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.061680473,0.027182057,0.40421548,0.05945849,0.032748852,0.0011459242,0.013384678,0.0074689877,0.39271513],"genre_scores_gemma":[0.127834,0.017552825,0.20359592,0.0046631037,0.005885035,0.00067401223,0.02533874,0.0047283797,0.609728],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99886155,0.0005154247,0.000077137454,0.00021064741,0.00022917212,0.00010603336],"domain_scores_gemma":[0.9957996,0.0020954297,0.000060178078,0.000502701,0.001146959,0.00039515138],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029754513,0.00063046964,0.0008659591,0.0016335003,0.0016496496,0.0055097938,0.0013062173,0.00094316044,0.046374004],"category_scores_gemma":[0.0059453086,0.000372767,0.0009488043,0.0018877229,0.0012293245,0.005709531,0.0024107941,0.0030536533,0.009255364],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00038454676,0.0002852662,0.0012556161,0.0007022047,0.00007792882,0.00040158912,0.0032422983,0.0012145673,0.008285656,0.044437982,0.5065474,0.43316498],"study_design_scores_gemma":[0.000029707051,0.000044308315,0.0014345036,0.0001549818,0.00005049553,0.00020293517,0.0010842409,0.0038808186,0.004127798,0.014896106,0.9740718,0.000022351081],"about_ca_topic_score_codex":0.013278423,"about_ca_topic_score_gemma":0.01714945,"teacher_disagreement_score":0.046374004,"about_ca_system_score_codex":0.001424517,"about_ca_system_score_gemma":0.003101091,"threshold_uncertainty_score":0.15513647},"labels":[],"label_agreement":null},{"id":"W3165604148","doi":"10.31234/osf.io/pckmg","title":"Chaining algorithms and historical adjective extension","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada","keywords":"Adjective; Computer science; Noun; Natural language processing; Linguistics; Artificial intelligence; Chaining; Axiom; Categorization; Set (abstract data type); Extension (predicate logic); Mathematics; Psychology; Philosophy","score_opus":0.03135906133053827,"score_gpt":0.28155254951779896,"score_spread":0.25019348818726067,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3165604148","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13224277,0.0013840111,0.8523826,0.00080136984,0.000058957223,0.00009516289,0.00023106782,0.00086466887,0.011939414],"genre_scores_gemma":[0.7462122,0.0010353515,0.24595399,0.00022479218,0.00008947246,0.00018630028,0.00065254443,0.00031441997,0.00533085],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99822396,0.0006497855,0.00011366907,0.0006667031,0.00025532432,0.000090566224],"domain_scores_gemma":[0.98860407,0.00829785,0.0007879609,0.0014851752,0.00063820876,0.00018683664],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034016008,0.00052907993,0.0006361733,0.0026310883,0.0013367669,0.002618219,0.0015696452,0.0010991681,0.004257895],"category_scores_gemma":[0.021171857,0.00064262794,0.0011887765,0.0032829333,0.0036216115,0.0109712025,0.0025265494,0.0017024996,0.00087764475],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015954881,0.00007252997,0.020305082,0.00038486265,0.00012539036,0.00048183018,0.003781936,0.10060821,0.0042318325,0.6244109,0.002118117,0.24331987],"study_design_scores_gemma":[0.0000139352105,0.000035706304,0.0024396211,0.00006883528,0.000031121202,0.00037441892,0.00036049605,0.22637856,0.0016380154,0.7595663,0.00905658,0.000036438403],"about_ca_topic_score_codex":0.0026231855,"about_ca_topic_score_gemma":0.0030866868,"teacher_disagreement_score":0.004257895,"about_ca_system_score_codex":0.0013824735,"about_ca_system_score_gemma":0.0007528189,"threshold_uncertainty_score":0.017989635},"labels":[],"label_agreement":null},{"id":"W3165821503","doi":"10.18653/v1/2021.calcs-1.8","title":"Investigating Code-Mixed Modern Standard Arabic-Egyptian to English Machine Translation","year":2021,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Compute Canada","keywords":"Machine translation; Computer science; Natural language processing; Artificial intelligence; Transformer; Arabic; Task (project management); Scratch; Language model; Code-switching; Context (archaeology); Code (set theory); Speech recognition; Programming language; Linguistics; Engineering","score_opus":0.020517177510521997,"score_gpt":0.27398834520874443,"score_spread":0.2534711676982224,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3165821503","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.93568283,0.0011471765,0.046058904,0.0008510721,0.00023833186,0.0001711716,0.0017532334,0.0032811083,0.0108160935],"genre_scores_gemma":[0.9572093,0.0002455016,0.03193961,0.00028035,0.000048777314,0.000120547375,0.005337397,0.00034921666,0.0044693765],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988998,0.0004778969,0.00008646108,0.00029244888,0.00014295359,0.0001003766],"domain_scores_gemma":[0.9956637,0.0024810452,0.00018810375,0.0005198114,0.0009505418,0.00019678373],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022298906,0.0011012104,0.00057786424,0.0010575026,0.00082179846,0.0014634646,0.0007255303,0.0010475393,0.0029116883],"category_scores_gemma":[0.008025512,0.00035660417,0.0005962469,0.0012078466,0.0007025653,0.0021424806,0.0013633611,0.0014176435,0.0016691891],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0028692523,0.0015823907,0.056089714,0.001760823,0.000724378,0.0028910174,0.0038066632,0.3718408,0.053111404,0.017722635,0.030417798,0.45718318],"study_design_scores_gemma":[0.00016805086,0.0006236884,0.010605231,0.000103139326,0.00014139268,0.0006113417,0.0016166974,0.91938275,0.043024607,0.008891849,0.014743379,0.000087879656],"about_ca_topic_score_codex":0.020546267,"about_ca_topic_score_gemma":0.02670184,"teacher_disagreement_score":0.020546267,"about_ca_system_score_codex":0.0011788808,"about_ca_system_score_gemma":0.0012590985,"threshold_uncertainty_score":0.04085332},"labels":[],"label_agreement":null},{"id":"W3165916482","doi":"10.18653/v1/2021.americasnlp-1.12","title":"Leveraging English Word Embeddings for Semi-Automatic Semantic Classification in Nêhiyawêwin (Plains Cree)","year":2021,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Natural language processing; Context (archaeology); Word (group theory); Point (geometry); Artificial intelligence; Ontology; Cluster analysis; Noun; Linguistics; Geography; Archaeology; Mathematics","score_opus":0.022613477112900505,"score_gpt":0.28587441667621816,"score_spread":0.26326093956331764,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3165916482","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.46363103,0.00074021955,0.48497984,0.000966248,0.00037237618,0.0007333728,0.012561967,0.009555488,0.02645942],"genre_scores_gemma":[0.5552923,0.00025931987,0.4043608,0.00016303505,0.000035714147,0.00043339367,0.022985695,0.0014095192,0.015060266],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991529,0.00019572649,0.00010185036,0.00035012918,0.00012020745,0.00007921697],"domain_scores_gemma":[0.99902177,0.00037325744,0.000106999316,0.000117046664,0.00032617876,0.00005481584],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006376089,0.0007997684,0.00031158156,0.0026231422,0.000927719,0.0017878644,0.0005896265,0.00050638634,0.0048021786],"category_scores_gemma":[0.0024288325,0.00042309612,0.0005561125,0.0020326907,0.0006273064,0.0033109873,0.0015931157,0.001090991,0.003066669],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005119593,0.00016386909,0.020672712,0.0011558392,0.000094376126,0.0013802919,0.02250172,0.005053706,0.07294789,0.022632921,0.025625918,0.8272588],"study_design_scores_gemma":[0.000100146804,0.0003465782,0.089851156,0.0006760177,0.00018049587,0.002330317,0.066002965,0.26623103,0.082965285,0.04952841,0.4413764,0.00041118494],"about_ca_topic_score_codex":0.015794016,"about_ca_topic_score_gemma":0.031357225,"teacher_disagreement_score":0.015794016,"about_ca_system_score_codex":0.0010363108,"about_ca_system_score_gemma":0.0012461303,"threshold_uncertainty_score":0.031404138},"labels":[],"label_agreement":null},{"id":"W3166956191","doi":"","title":"The Impact of Sentence Alignment Errors on Phrase-Based Machine Translation Performance","year":2012,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":30,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Machine translation; Sentence; Phrase; Robustness (evolution); Natural language processing; Artificial intelligence; Speech recognition; Translation (biology)","score_opus":0.019552186657782765,"score_gpt":0.2901371945109736,"score_spread":0.27058500785319084,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3166956191","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9321532,0.0054934653,0.04495256,0.0016797094,0.00053057325,0.00010289595,0.0019272652,0.0038386409,0.009321651],"genre_scores_gemma":[0.96754974,0.0010800854,0.02553585,0.00021872167,0.00015485923,0.00006612149,0.0030654017,0.0005460874,0.0017831967],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9906086,0.0035714502,0.0011119227,0.0017964495,0.0024084311,0.0005032005],"domain_scores_gemma":[0.9317009,0.054896872,0.0033699807,0.0039054118,0.00564484,0.00048188868],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0072046835,0.0011637034,0.0010170214,0.0010384826,0.0009525011,0.0022517478,0.0009941771,0.0016789929,0.0035115317],"category_scores_gemma":[0.063087486,0.00054894073,0.00037522067,0.0026108527,0.00086491014,0.004267673,0.0014636472,0.0014920961,0.0032206543],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0054480033,0.0010042789,0.06964412,0.0021444762,0.0006947546,0.001221433,0.0018307836,0.17121242,0.16386099,0.0033418515,0.0217815,0.5578155],"study_design_scores_gemma":[0.00018741659,0.0045606513,0.12515944,0.00022108314,0.00051986985,0.0016725347,0.0012003833,0.5360387,0.31298086,0.0083968695,0.008703429,0.00035878408],"about_ca_topic_score_codex":0.0036411053,"about_ca_topic_score_gemma":0.0038184172,"teacher_disagreement_score":0.0072046835,"about_ca_system_score_codex":0.00081334414,"about_ca_system_score_gemma":0.0007460489,"threshold_uncertainty_score":0.038102448},"labels":[],"label_agreement":null},{"id":"W3167304372","doi":"","title":"Thematic relatedness production norms for 100 object concepts","year":2015,"lang":"en","type":"article","venue":"DSpace@MIT (Massachusetts Institute of Technology)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Thematic map; Attributive; Locative case; Computer science; Object (grammar); Similarity (geometry); Cognition; Cognitive science; Linguistics; Psychology; Artificial intelligence","score_opus":0.02781863432352398,"score_gpt":0.30306367178566695,"score_spread":0.27524503746214296,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3167304372","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.90376896,0.0005274193,0.06267694,0.00015264026,0.000111355286,0.00059525657,0.0030107775,0.0003991056,0.028757425],"genre_scores_gemma":[0.8878709,0.00023904683,0.1023617,0.0000540777,0.00003883958,0.0015464979,0.005539228,0.000231986,0.0021176527],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.98643535,0.004779898,0.0016566727,0.0009969723,0.005928715,0.00020231631],"domain_scores_gemma":[0.90506494,0.067801125,0.0063674552,0.0073488215,0.012818946,0.0005986202],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013308927,0.0003245398,0.000342607,0.0036977103,0.0006512077,0.0011216786,0.00072874734,0.0004824941,0.0030104194],"category_scores_gemma":[0.08613867,0.00018110735,0.00037894794,0.002610526,0.0012955909,0.001381943,0.0017749699,0.00063770334,0.00047820222],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.003080173,0.0008024191,0.17547563,0.0016038935,0.00024361689,0.00095303904,0.030615881,0.00605474,0.11386723,0.08050089,0.008644414,0.578158],"study_design_scores_gemma":[0.000245713,0.0023808791,0.59673715,0.00047700075,0.00018638565,0.0049578524,0.014380684,0.029650958,0.14788283,0.12081321,0.08192342,0.0003639328],"about_ca_topic_score_codex":0.0007944022,"about_ca_topic_score_gemma":0.0010902766,"teacher_disagreement_score":0.013308927,"about_ca_system_score_codex":0.0008583467,"about_ca_system_score_gemma":0.00044882315,"threshold_uncertainty_score":0.07038516},"labels":[],"label_agreement":null},{"id":"W3167335398","doi":"10.18653/v1/2021.sigtyp-1.11","title":"SIGTYP 2021 Shared Task: Robust Spoken Language Identification","year":2021,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Typology; Computer science; Identification (biology); Task (project management); Natural language processing; Artificial intelligence; Linguistics; History; Engineering; Philosophy; Biology; Archaeology","score_opus":0.012749892621240653,"score_gpt":0.2627833683118665,"score_spread":0.25003347569062584,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3167335398","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.089144416,0.0060628853,0.21690297,0.0055432897,0.009055163,0.0031535637,0.39084432,0.22823817,0.051055208],"genre_scores_gemma":[0.12024317,0.0005918354,0.12502773,0.0012860866,0.00077323656,0.003729747,0.698807,0.0089154905,0.04062569],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99162537,0.0039738617,0.00047749153,0.0017071848,0.0014834225,0.00073267263],"domain_scores_gemma":[0.9864291,0.0048242067,0.00034254728,0.004102099,0.003188092,0.001113891],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00806047,0.004347791,0.0034170276,0.0026694618,0.003293266,0.004705799,0.004976033,0.0064209066,0.03478792],"category_scores_gemma":[0.016677054,0.0013716272,0.0015966911,0.0018172009,0.0013413797,0.0055892994,0.009635868,0.0042392677,0.063116156],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014670839,0.00047501593,0.0013862598,0.0010854305,0.00030581132,0.0006564936,0.00044041759,0.0026295541,0.0133113945,0.001263477,0.80225646,0.17472261],"study_design_scores_gemma":[0.004311866,0.0018172911,0.023326704,0.00064605894,0.00063589594,0.0026047423,0.005624251,0.24457103,0.07541971,0.02140697,0.6186255,0.0010099475],"about_ca_topic_score_codex":0.021371325,"about_ca_topic_score_gemma":0.028492266,"teacher_disagreement_score":0.03478792,"about_ca_system_score_codex":0.0014699062,"about_ca_system_score_gemma":0.0048307464,"threshold_uncertainty_score":0.116377175},"labels":[],"label_agreement":null},{"id":"W3167933713","doi":"10.1121/10.0005365","title":"A place to share teaching resources: Speech and language resource bank","year":2021,"lang":"en","type":"article","venue":"The Journal of the Acoustical Society of America","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Resource (disambiguation); Computer science; Shared resource; Speech community; Linguistics; Computer security","score_opus":0.008244658172995127,"score_gpt":0.26397772000338104,"score_spread":0.2557330618303859,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3167933713","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012217879,0.002179108,0.2303125,0.030183727,0.007953549,0.003998331,0.10826796,0.17930606,0.4255808],"genre_scores_gemma":[0.08471102,0.0025520194,0.3085726,0.009379822,0.003488904,0.0039782682,0.15455452,0.044872012,0.38789088],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99626213,0.0009062659,0.00055788766,0.000488725,0.0012643433,0.0005206556],"domain_scores_gemma":[0.9615493,0.004582843,0.0015150557,0.0075480416,0.011413986,0.013390827],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0064558634,0.0007959301,0.0014067561,0.005664472,0.0037178863,0.00962603,0.003237657,0.0024964584,0.24026461],"category_scores_gemma":[0.023373172,0.0011432905,0.00052303035,0.0070201005,0.0010815654,0.013730364,0.010122654,0.0038441815,0.26615578],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00026409345,0.00020507605,0.00082968484,0.00017698633,0.00001553628,0.00023923631,0.0005400948,0.00023685466,0.0025994887,0.0076131267,0.859127,0.12815283],"study_design_scores_gemma":[0.00008657481,0.00004144388,0.001213247,0.00016126325,0.000016631178,0.00020965685,0.0006541672,0.0013693068,0.0027612278,0.00577044,0.98762137,0.000094651805],"about_ca_topic_score_codex":0.0043173376,"about_ca_topic_score_gemma":0.004038632,"teacher_disagreement_score":0.24026461,"about_ca_system_score_codex":0.0020070127,"about_ca_system_score_gemma":0.011154107,"threshold_uncertainty_score":0.8037652},"labels":[],"label_agreement":null},{"id":"W3169524818","doi":"10.18653/v1/2021.americasnlp-1.11","title":"Representation of Yine [Arawak] Morphology by Finite State Transducer Formalism","year":2021,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"Consejo Nacional de Ciencia, Tecnología e Innovación Tecnológica","keywords":"Computer science; Linguistics; Representation (politics); Natural language processing; Algorithm; Artificial intelligence","score_opus":0.01363047109877225,"score_gpt":0.28462177071057887,"score_spread":0.2709912996118066,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3169524818","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0206803,0.00013282556,0.96499044,0.0003171788,0.000047145364,0.00007711967,0.0004664128,0.0017982963,0.0114901345],"genre_scores_gemma":[0.41142705,0.00032187818,0.5807664,0.00016491167,0.000038144033,0.00019408435,0.00094254594,0.00041966344,0.0057252925],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995974,0.00009550037,0.0000391068,0.000097932694,0.00011963863,0.00005045019],"domain_scores_gemma":[0.99939823,0.00024994375,0.00007183821,0.00012021266,0.00013814967,0.000021528782],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005082728,0.000318207,0.00040970903,0.0008388778,0.00061671203,0.0018743834,0.001061357,0.00072525383,0.0042974534],"category_scores_gemma":[0.001232351,0.00040710543,0.0011339627,0.00068284606,0.001674676,0.0023137135,0.00095941685,0.0011342787,0.0010854878],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000049533664,0.000033978566,0.0012364283,0.0001910264,0.00003507562,0.0010589915,0.0011498847,0.045227583,0.023436962,0.89470917,0.0019119745,0.030959385],"study_design_scores_gemma":[0.000022436228,0.00005447745,0.00058336876,0.0000657933,0.00004557034,0.0005701388,0.00022398413,0.4891303,0.010377961,0.46201292,0.03685159,0.00006143949],"about_ca_topic_score_codex":0.003200753,"about_ca_topic_score_gemma":0.0036531512,"teacher_disagreement_score":0.0042974534,"about_ca_system_score_codex":0.0007593673,"about_ca_system_score_gemma":0.0011888864,"threshold_uncertainty_score":0.0143764615},"labels":[],"label_agreement":null},{"id":"W3169893328","doi":"10.18653/v1/2023.ijcnlp-main.30","title":"One Sense per Translation","year":2023,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Machine Intelligence Institute","keywords":"Computer science; Computational linguistics; Machine translation; Linguistics; Natural language processing; Artificial intelligence; Philosophy","score_opus":0.06339380363245987,"score_gpt":0.311922666201929,"score_spread":0.2485288625694691,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3169893328","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029123032,0.013387622,0.30219603,0.028128874,0.023558334,0.00079180155,0.043157466,0.0154952295,0.54416156],"genre_scores_gemma":[0.34022462,0.0100287795,0.35970625,0.0057568448,0.003936444,0.0006167972,0.049189616,0.009301286,0.22123943],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.996924,0.0007958405,0.00040304783,0.00075270236,0.000892261,0.00023222694],"domain_scores_gemma":[0.994494,0.0010316685,0.0002418364,0.0022090843,0.0017730477,0.0002502204],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016296245,0.0016735316,0.0011491254,0.0036511007,0.0021674794,0.005902294,0.0012914458,0.0021457155,0.07858707],"category_scores_gemma":[0.009398297,0.0008207846,0.0011135372,0.0041944887,0.0015785904,0.008718396,0.0046479017,0.0018493588,0.07241773],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00045160856,0.00010896665,0.0019785448,0.0017952125,0.00016685757,0.0016370105,0.002639781,0.00094716786,0.013872932,0.20573047,0.3234019,0.44726953],"study_design_scores_gemma":[0.000055666947,0.00009228826,0.0015596029,0.0004970419,0.00012786833,0.003089005,0.0023575558,0.004196588,0.009821899,0.19438837,0.7837082,0.00010578339],"about_ca_topic_score_codex":0.0010636436,"about_ca_topic_score_gemma":0.0017985223,"teacher_disagreement_score":0.07858707,"about_ca_system_score_codex":0.00083083165,"about_ca_system_score_gemma":0.0016724677,"threshold_uncertainty_score":0.26289994},"labels":[],"label_agreement":null},{"id":"W3169915179","doi":"","title":"History of Text Mining over a Quarter of a Century and its Future Potential","year":2021,"lang":"en","type":"article","venue":"IEICE Technical Report; IEICE Tech. Rep.","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Quarter (Canadian coin); Computer science; History; Data science; Archaeology","score_opus":0.008353451844343487,"score_gpt":0.25705365556889365,"score_spread":0.24870020372455015,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3169915179","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007935836,0.7615752,0.054056115,0.13064028,0.008465453,0.00009251461,0.001255245,0.00071129756,0.035268053],"genre_scores_gemma":[0.0895927,0.7182981,0.09380553,0.031984136,0.023908641,0.00026709074,0.0029939502,0.00066081766,0.038489204],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9960517,0.0013032624,0.00042847634,0.00072288385,0.0013470557,0.00014657676],"domain_scores_gemma":[0.966556,0.02162161,0.0014692642,0.002093833,0.006837697,0.0014215494],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0090960935,0.00061447866,0.0009366768,0.006619258,0.0016106303,0.007310823,0.0017266473,0.002541817,0.005066919],"category_scores_gemma":[0.024389867,0.00053237926,0.00057582994,0.009761109,0.004394752,0.0134940045,0.0022015036,0.003988469,0.003507306],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023600462,0.00009826987,0.0027209911,0.002029912,0.00009217331,0.00019759944,0.0011243238,0.0009786552,0.0012083488,0.062650196,0.077153385,0.85151005],"study_design_scores_gemma":[0.000014871075,0.000070595976,0.0028160606,0.0021200902,0.000039500275,0.0005144757,0.0006882712,0.0020991918,0.0013668615,0.08683974,0.90336376,0.00006653849],"about_ca_topic_score_codex":0.002666626,"about_ca_topic_score_gemma":0.003043882,"teacher_disagreement_score":0.0090960935,"about_ca_system_score_codex":0.0027884303,"about_ca_system_score_gemma":0.0036252476,"threshold_uncertainty_score":0.04810536},"labels":[],"label_agreement":null},{"id":"W3169969327","doi":"10.1109/aero50100.2021.9438369","title":"Attended-over Distributed Specificity for Information Extraction in Cybersecurity","year":2021,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University; Royal Bank of Canada","funders":"","keywords":"Computer science; Cyberspace; Context (archaeology); Computer security; Terminology; Domain (mathematical analysis); Benchmark (surveying); Information extraction; Data science; The Internet; Data mining; World Wide Web; Information retrieval","score_opus":0.012501241183125657,"score_gpt":0.28431572062163096,"score_spread":0.2718144794385053,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3169969327","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06084528,0.0054652286,0.9063202,0.001017576,0.00019511183,0.0008673329,0.010405801,0.0094361175,0.0054472815],"genre_scores_gemma":[0.3370547,0.0019261937,0.6298394,0.00057835813,0.00020659335,0.0007505415,0.02563205,0.00044280407,0.0035693394],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99778533,0.00045313532,0.00032819115,0.0008553881,0.00040567567,0.00017218525],"domain_scores_gemma":[0.9969278,0.0015829488,0.0003100048,0.0005716423,0.0005143593,0.00009314728],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014439189,0.0015293631,0.0010434337,0.009439058,0.00103983,0.0019220299,0.0011792228,0.0018345314,0.0034903467],"category_scores_gemma":[0.0074157473,0.0004628168,0.0021141164,0.0076014753,0.0008272025,0.0051092287,0.002701531,0.0016234461,0.0027359258],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00052155793,0.0002569632,0.009722613,0.0019605805,0.0002706801,0.0008788873,0.00103553,0.013627423,0.049391095,0.017754,0.018104918,0.88647574],"study_design_scores_gemma":[0.0001494884,0.0006313795,0.02454241,0.0007680649,0.00073805253,0.0038735613,0.0020353985,0.6007285,0.07461987,0.1647091,0.12697114,0.0002330677],"about_ca_topic_score_codex":0.003477831,"about_ca_topic_score_gemma":0.0064340015,"teacher_disagreement_score":0.009439058,"about_ca_system_score_codex":0.0010422419,"about_ca_system_score_gemma":0.0019966909,"threshold_uncertainty_score":0.011676431},"labels":[],"label_agreement":null},{"id":"W3170656445","doi":"","title":"Automatic Translation of Court Judgments","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Natural language processing; Machine translation; Translation (biology); Artificial intelligence; Linguistics","score_opus":0.025685518519427773,"score_gpt":0.27251442566625256,"score_spread":0.2468289071468248,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3170656445","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5120185,0.0018776346,0.21906216,0.0032298667,0.0022705623,0.004233757,0.025909213,0.063826695,0.16757163],"genre_scores_gemma":[0.7282142,0.0006715942,0.21329056,0.0004285584,0.0001751255,0.00044165255,0.024561018,0.0022006012,0.030016817],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99598724,0.0013394274,0.00023141563,0.00079816085,0.001414554,0.00022913655],"domain_scores_gemma":[0.99150497,0.0028253829,0.00024463708,0.0010290092,0.004204233,0.00019175469],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017283494,0.001267402,0.00087712676,0.0020763397,0.0023514738,0.0026548933,0.00094990514,0.0010033231,0.020738436],"category_scores_gemma":[0.015846767,0.00050310727,0.00073537376,0.002037467,0.0007527798,0.0014115665,0.0013258348,0.0015075132,0.0074112345],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010374578,0.00054994976,0.0032328016,0.0014384602,0.00013683237,0.002273391,0.0053916005,0.01736793,0.084889494,0.01209299,0.08478382,0.7868054],"study_design_scores_gemma":[0.0010180773,0.0012560999,0.031258665,0.0004910483,0.00032700237,0.0031317663,0.009786533,0.38196802,0.21465021,0.013977308,0.3414397,0.0006955211],"about_ca_topic_score_codex":0.11100018,"about_ca_topic_score_gemma":0.09573897,"teacher_disagreement_score":0.11100018,"about_ca_system_score_codex":0.0038068264,"about_ca_system_score_gemma":0.0052785515,"threshold_uncertainty_score":0.22070819},"labels":[],"label_agreement":null},{"id":"W3171517119","doi":"","title":"LTL2Action: Generalizing LTL Instructions for Multi-Task RL","year":2021,"lang":"en","type":"article","venue":"International Conference on Machine Learning","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute; University of Toronto","funders":"","keywords":"Computer science; Task (project management); Reinforcement learning; Syntax; Semantics (computer science); Overhead (engineering); Linear temporal logic; Artificial intelligence; Scheme (mathematics); Sample complexity; Programming language; Theoretical computer science; Natural language processing; Mathematics","score_opus":0.07083520501108485,"score_gpt":0.3596141055672882,"score_spread":0.2887789005562033,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3171517119","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0065829586,0.00005223094,0.98814154,0.00019641587,0.0000468987,0.00008360773,0.000084391824,0.0027262676,0.0020856732],"genre_scores_gemma":[0.4981858,0.0001709485,0.4943732,0.00062603725,0.00006155716,0.0005845864,0.00043091874,0.0007823563,0.004784563],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994055,0.00017247748,0.0000460401,0.00013527846,0.00017118914,0.00006958527],"domain_scores_gemma":[0.9986639,0.0005949424,0.00012185309,0.00034250665,0.00017442304,0.000102281636],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013981794,0.0009527731,0.0005501271,0.00030686706,0.000319811,0.00095388206,0.0022512504,0.0010934033,0.0043117306],"category_scores_gemma":[0.0047989483,0.00039407192,0.000671536,0.00028292174,0.0013586442,0.0017182408,0.0018968742,0.0025999586,0.0011041389],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021926186,0.00020339145,0.001058137,0.00020892324,0.000060591403,0.00021896849,0.00029093222,0.7392447,0.011395651,0.06233893,0.003573049,0.18118747],"study_design_scores_gemma":[0.000017734958,0.00004237684,0.000042516458,0.00000977856,0.0000056150707,0.000015132712,0.0000070846563,0.9717156,0.0023017258,0.02431438,0.001520864,0.0000072355797],"about_ca_topic_score_codex":0.0031192652,"about_ca_topic_score_gemma":0.005143838,"teacher_disagreement_score":0.0043117306,"about_ca_system_score_codex":0.0007491369,"about_ca_system_score_gemma":0.0016555152,"threshold_uncertainty_score":0.014424205},"labels":[],"label_agreement":null},{"id":"W3172041751","doi":"","title":"Evaluation of Domain Adaptation Techniques for TRANSLI in a Real-World Environment","year":2012,"lang":"en","type":"article","venue":"Conference of the Association for Machine Translation in the Americas","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec","funders":"","keywords":"Computer science; Machine translation; Domain (mathematical analysis); Adaptation (eye); Domain adaptation; BLEU; Word (group theory); Artificial intelligence; Machine learning; Quality (philosophy); Translation (biology); Natural language processing; Data mining","score_opus":0.07590646895781045,"score_gpt":0.3457488423422705,"score_spread":0.26984237338446004,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3172041751","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.84516734,0.0020724793,0.12879196,0.0005580918,0.00028554196,0.0011557521,0.0015504148,0.014431905,0.0059865066],"genre_scores_gemma":[0.7292855,0.0010658164,0.25767952,0.00022922337,0.000106843436,0.00094135135,0.0057353815,0.0010398973,0.0039164033],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9917291,0.0051998124,0.0006267937,0.0010961615,0.0010840069,0.00026416083],"domain_scores_gemma":[0.97756726,0.014270603,0.00094949635,0.0030275488,0.0037379533,0.00044716435],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008001961,0.0017822109,0.0009969894,0.0013673251,0.00076303846,0.0012311051,0.0016200476,0.001732153,0.0016562128],"category_scores_gemma":[0.026836796,0.00046676866,0.0007262767,0.0024293652,0.000745841,0.001866934,0.0013934572,0.0016289443,0.0017061472],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.005131621,0.003656771,0.0101913875,0.0025267948,0.0008821227,0.0012586717,0.002296878,0.17391357,0.08627187,0.0012620108,0.013200957,0.6994073],"study_design_scores_gemma":[0.0013813553,0.005366148,0.028086107,0.0001373181,0.00049492274,0.0018035232,0.0017506188,0.7972194,0.1450329,0.0014501096,0.016952457,0.00032511266],"about_ca_topic_score_codex":0.003198481,"about_ca_topic_score_gemma":0.0032370724,"teacher_disagreement_score":0.008001961,"about_ca_system_score_codex":0.0007256572,"about_ca_system_score_gemma":0.00086221297,"threshold_uncertainty_score":0.04231894},"labels":[],"label_agreement":null},{"id":"W3172225000","doi":"10.48550/arxiv.2106.09325","title":"Central Kurdish machine translation: First large scale parallel corpus\\n and experiments","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Machine translation; Translation (biology); Computer science; Parallel corpora; Natural language processing; Scale (ratio); Artificial intelligence; Geography; Cartography; Chemistry","score_opus":0.04133679020691154,"score_gpt":0.20670707092365662,"score_spread":0.1653702807167451,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3172225000","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.71753955,0.012839806,0.05047813,0.0049312823,0.0031588466,0.0037961763,0.08468634,0.03929823,0.08327167],"genre_scores_gemma":[0.49592042,0.0022456083,0.124208346,0.0015831086,0.00057981675,0.0027273928,0.34880504,0.0039255265,0.020004725],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9953674,0.0018778587,0.0004848225,0.0010041033,0.00091158284,0.000354223],"domain_scores_gemma":[0.99065685,0.0034133582,0.00025597846,0.0029257238,0.0022416092,0.00050643244],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004191843,0.0017881452,0.0016292289,0.0026397177,0.0036385963,0.0024092463,0.0020499045,0.0018397981,0.014308327],"category_scores_gemma":[0.013915597,0.0007701497,0.0009877905,0.004445743,0.0023634105,0.0045543755,0.0050356286,0.0027160095,0.013075738],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0030659195,0.0043102102,0.0075740176,0.0048054894,0.0006093754,0.003088568,0.0038076378,0.013567413,0.035170693,0.008650361,0.45275882,0.46259147],"study_design_scores_gemma":[0.005007495,0.0036964582,0.06157495,0.0010920975,0.0008518674,0.006844803,0.010048087,0.13031255,0.1251833,0.022361517,0.6322759,0.0007509606],"about_ca_topic_score_codex":0.012454221,"about_ca_topic_score_gemma":0.023362942,"teacher_disagreement_score":0.014308327,"about_ca_system_score_codex":0.0015312567,"about_ca_system_score_gemma":0.002798986,"threshold_uncertainty_score":0.047866166},"labels":[],"label_agreement":null},{"id":"W3172252280","doi":"","title":"WeBiText: Multilingual Concordancer Built from Public High Quality Web Content","year":2010,"lang":"en","type":"article","venue":"Conference of the Association for Machine Translation in the Americas","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Machine translation; Quality (philosophy); Web page; Word (group theory); Order (exchange); Translation (biology); Envelope (radar); World Wide Web; Information retrieval; Natural language processing; Artificial intelligence; Telecommunications; Mathematics","score_opus":0.08095440750312109,"score_gpt":0.3455252792081267,"score_spread":0.2645708717050056,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3172252280","genre_codex":"software","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":"software","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09760901,0.0011299049,0.2807955,0.00060123875,0.0005482016,0.0017308586,0.034714136,0.54870147,0.034169678],"genre_scores_gemma":[0.2978279,0.0006368231,0.4535072,0.0006495521,0.00029887332,0.0012397255,0.15826696,0.054352988,0.03321998],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9933335,0.0013056075,0.00074111175,0.0012542428,0.003032389,0.00033309724],"domain_scores_gemma":[0.9836701,0.006161117,0.00092475605,0.004136521,0.004455701,0.0006517626],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0042379396,0.0017551215,0.0018203381,0.006222053,0.0016119367,0.0028560152,0.0021583543,0.0015781807,0.021580318],"category_scores_gemma":[0.022849733,0.0011545621,0.0010946447,0.0054613412,0.00076475565,0.00577141,0.0034870992,0.0015949497,0.018055446],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0036761675,0.0008521362,0.017145354,0.0028198522,0.0004612999,0.004075143,0.0036900544,0.004730786,0.06420143,0.008471955,0.24340962,0.64646626],"study_design_scores_gemma":[0.000836063,0.0011015596,0.036936253,0.0006061557,0.00039147367,0.0061758584,0.002368617,0.13181885,0.26055065,0.012665435,0.54553455,0.0010145062],"about_ca_topic_score_codex":0.0076834,"about_ca_topic_score_gemma":0.007869518,"teacher_disagreement_score":0.021580318,"about_ca_system_score_codex":0.0011759547,"about_ca_system_score_gemma":0.0023072427,"threshold_uncertainty_score":0.072193325},"labels":[],"label_agreement":null},{"id":"W3172907523","doi":"10.18653/v1/2021.naacl-main.251","title":"Negative language transfer in learner English: A new dataset","year":2021,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Computational linguistics; Natural language processing; Artificial intelligence; Linguistics; English language; Cognitive science; Psychology; Philosophy","score_opus":0.013141323483006876,"score_gpt":0.27643176638619393,"score_spread":0.26329044290318704,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3172907523","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.34510165,0.0027498738,0.005710467,0.00323814,0.0010305293,0.00043582058,0.61926717,0.0032403779,0.019225918],"genre_scores_gemma":[0.13293873,0.00046605265,0.0052209836,0.00076357764,0.00026789532,0.000581288,0.8465649,0.0004767989,0.012719773],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99835616,0.00044870976,0.00018500366,0.00040436076,0.0003963416,0.00020951669],"domain_scores_gemma":[0.99601483,0.00096263085,0.00022623606,0.0010198752,0.0011530378,0.0006234324],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017342914,0.00091258157,0.00093064806,0.00220391,0.0013212111,0.0017492349,0.0018890244,0.0020728558,0.006397553],"category_scores_gemma":[0.0065405536,0.00024601977,0.000733805,0.0014712933,0.000721184,0.0021710156,0.003451484,0.001593183,0.011439156],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021692656,0.0022286228,0.07540006,0.0018058578,0.00045355785,0.0016929128,0.002209216,0.0017652444,0.008866836,0.0023541609,0.77871096,0.12234317],"study_design_scores_gemma":[0.00094984105,0.0011419123,0.20988445,0.0007311397,0.0003965505,0.004607376,0.007539479,0.0097255,0.015497162,0.0054597137,0.7436766,0.0003903368],"about_ca_topic_score_codex":0.007590585,"about_ca_topic_score_gemma":0.015161961,"teacher_disagreement_score":0.007590585,"about_ca_system_score_codex":0.0005645805,"about_ca_system_score_gemma":0.0010969626,"threshold_uncertainty_score":0.021401942},"labels":[],"label_agreement":null},{"id":"W3173053558","doi":"10.48550/arxiv.2105.13573","title":"Investigating Code-Mixed Modern Standard Arabic-Egyptian to English Machine Translation","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Machine translation; Computer science; Natural language processing; Arabic; Artificial intelligence; Scratch; Task (project management); Transformer; Language model; Code-switching; Code (set theory); Context (archaeology); Modern Standard Arabic; Speech recognition; Programming language; Linguistics; Engineering","score_opus":0.057965637028369245,"score_gpt":0.2112129708741395,"score_spread":0.15324733384577027,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3173053558","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9153649,0.0014271054,0.06343307,0.0010518088,0.00028055796,0.00017827975,0.00211813,0.0039528944,0.012193293],"genre_scores_gemma":[0.9483985,0.00027446426,0.03926381,0.00032307827,0.00005900257,0.00013213359,0.006174089,0.0004168452,0.0049581393],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989145,0.00046802615,0.0000818634,0.00029987138,0.00013795999,0.00009781969],"domain_scores_gemma":[0.99571013,0.0024912034,0.00017905115,0.0005550961,0.00087717286,0.00018736858],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002179348,0.0011081034,0.00059486536,0.0010483459,0.000837518,0.0015175765,0.00074632163,0.0011300478,0.0030213315],"category_scores_gemma":[0.0079861665,0.00036653766,0.00061025814,0.0012462034,0.0007449144,0.0021429881,0.0014242628,0.0014477154,0.0017891242],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0028849624,0.0014417159,0.046296068,0.0017331223,0.0007037662,0.0028550467,0.003791121,0.37195218,0.052633706,0.022260778,0.03436445,0.4590831],"study_design_scores_gemma":[0.00016630054,0.0005256853,0.00891556,0.00009833099,0.00012951117,0.00059571443,0.0015362907,0.91851115,0.041687742,0.011851281,0.015899215,0.000083192266],"about_ca_topic_score_codex":0.018742062,"about_ca_topic_score_gemma":0.024607655,"teacher_disagreement_score":0.018742062,"about_ca_system_score_codex":0.0011699069,"about_ca_system_score_gemma":0.0012393992,"threshold_uncertainty_score":0.037265956},"labels":[],"label_agreement":null},{"id":"W3173145038","doi":"","title":"Spotify at TREC 2020: Genre-Aware Abstractive Podcast Summarization.","year":2020,"lang":"en","type":"article","venue":"Text REtrieval Conference","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Automatic summarization; Computer science; Information retrieval; Baseline (sea); Natural language processing; Task (project management); Key (lock); Granularity; Artificial intelligence; Aggregate (composite); World Wide Web","score_opus":0.029913480157860085,"score_gpt":0.26937939221752977,"score_spread":0.2394659120596697,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3173145038","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09852849,0.009335737,0.52466875,0.010455195,0.0128307855,0.008292456,0.17299177,0.11295652,0.049940277],"genre_scores_gemma":[0.17414032,0.001882046,0.36970568,0.001489445,0.002365349,0.0029453547,0.372997,0.004139656,0.07033512],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99624133,0.0017472742,0.00026464902,0.00056148606,0.00097092404,0.00021443196],"domain_scores_gemma":[0.99038494,0.0029381865,0.0004960323,0.0011597665,0.0040930277,0.0009280529],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0062857396,0.001841992,0.0010972511,0.002453057,0.0011401817,0.0025258474,0.0017745441,0.0018847253,0.011799369],"category_scores_gemma":[0.016034435,0.00038536533,0.0008702049,0.0012450265,0.00046911224,0.0027201483,0.001973068,0.0022864302,0.008495258],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00079004234,0.0004504893,0.0011548286,0.0016918685,0.00019041491,0.00024475562,0.0009326867,0.0068576573,0.044119842,0.002203325,0.6724333,0.26893085],"study_design_scores_gemma":[0.0009184164,0.0024617133,0.013926578,0.00037915978,0.00032820084,0.0006366537,0.0019303355,0.15759172,0.08545105,0.010471565,0.72557765,0.00032701564],"about_ca_topic_score_codex":0.008469019,"about_ca_topic_score_gemma":0.017800663,"teacher_disagreement_score":0.011799369,"about_ca_system_score_codex":0.0015506978,"about_ca_system_score_gemma":0.0019582552,"threshold_uncertainty_score":0.03947282},"labels":[],"label_agreement":null},{"id":"W3173162544","doi":"10.1145/3567592","title":"Neural Machine Translation for Low-resource Languages: A Survey","year":2022,"lang":"en","type":"review","venue":"ACM Computing Surveys","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":278,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Machine translation; Unavailability; Resource (disambiguation); Artificial intelligence; Natural language processing","score_opus":0.09391014352774846,"score_gpt":0.3746552444027698,"score_spread":0.28074510087502136,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3173162544","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.002742281,0.9512467,0.027755775,0.0022002384,0.00055868,0.00006708728,0.00032661905,0.0004610536,0.014641527],"genre_scores_gemma":[0.01670905,0.95295507,0.022739166,0.00068504934,0.00092243485,0.00008187735,0.0010354428,0.00012456915,0.0047473414],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9990159,0.00023398918,0.00011079144,0.00020441624,0.00038044143,0.00005444073],"domain_scores_gemma":[0.99611807,0.002607197,0.00016903218,0.00020867115,0.00083521684,0.00006182407],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001907713,0.0008300945,0.0010490912,0.0042637885,0.00045106138,0.0015264904,0.0012006052,0.0010398017,0.0068157963],"category_scores_gemma":[0.006060451,0.00057020836,0.00070544356,0.0071245516,0.0005747785,0.0036286921,0.0008576522,0.0011773277,0.005881336],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000035832723,0.00005713567,0.0005923346,0.006415778,0.00005756007,0.000062041065,0.00008196132,0.0015808934,0.0011823886,0.007909557,0.023815809,0.9582087],"study_design_scores_gemma":[0.00002291455,0.00024121442,0.0029924011,0.0039743814,0.00016693433,0.0012843569,0.00031966087,0.014571961,0.0037528304,0.01966535,0.9529429,0.00006499683],"about_ca_topic_score_codex":0.0018259273,"about_ca_topic_score_gemma":0.0027066693,"teacher_disagreement_score":0.0068157963,"about_ca_system_score_codex":0.00074783555,"about_ca_system_score_gemma":0.0017316994,"threshold_uncertainty_score":0.022801101},"labels":[],"label_agreement":null},{"id":"W3173187978","doi":"","title":"The CLaC System at the TREC 2020 News Track.","year":2020,"lang":"en","type":"article","venue":"Text REtrieval Conference","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Track (disk drive); Computer science; Operating system","score_opus":0.02671213014052321,"score_gpt":0.2636659143570772,"score_spread":0.236953784216554,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3173187978","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014510235,0.009369059,0.05614618,0.009746712,0.0092373965,0.0029052265,0.5695041,0.17696846,0.15161268],"genre_scores_gemma":[0.020112308,0.0011465113,0.05871778,0.0020796505,0.0010616939,0.00092872285,0.8103909,0.0048173266,0.10074518],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.997074,0.0010091739,0.00017699657,0.00042744706,0.0009783035,0.0003340032],"domain_scores_gemma":[0.99513346,0.00057109003,0.00013166272,0.0008228168,0.00260257,0.00073829235],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004110232,0.0023868175,0.0018109625,0.005277916,0.0021047646,0.0040321676,0.0033944373,0.0029307003,0.07925594],"category_scores_gemma":[0.0080679,0.00064958923,0.0008869894,0.0035606257,0.0005143775,0.0051593664,0.002965828,0.0020164386,0.107140936],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014096285,0.00006762381,0.00016709001,0.00014804809,0.00001882504,0.000023026852,0.000015296593,0.00014336305,0.0011334799,0.00054325967,0.9760424,0.02155667],"study_design_scores_gemma":[0.0008040276,0.0003930037,0.004171336,0.0002820597,0.00019015456,0.0003588063,0.00022009673,0.021141862,0.013841458,0.0063356035,0.9520629,0.00019863404],"about_ca_topic_score_codex":0.042861998,"about_ca_topic_score_gemma":0.071131915,"teacher_disagreement_score":0.07925594,"about_ca_system_score_codex":0.002128896,"about_ca_system_score_gemma":0.0051517403,"threshold_uncertainty_score":0.26513755},"labels":[],"label_agreement":null},{"id":"W3173421696","doi":"10.31234/osf.io/h4jxe","title":"Are translation equivalents special? Evidence from simulations and empirical data from bilingual infants","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"Natural Sciences and Engineering Research Council of Canada; Fonds de Recherche du Québec-Société et Culture; National Institutes of Health; Concordia University","keywords":"Vocabulary; Translation (biology); Referent; Linguistics; Computer science; Language acquisition; Natural language processing; Psychology; Artificial intelligence; Mathematics education","score_opus":0.24858077875921916,"score_gpt":0.4244519078794545,"score_spread":0.17587112912023536,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3173421696","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99087465,0.00013047637,0.0053157806,0.00017592797,0.0000036614365,0.000010862257,0.00016290176,0.000042070897,0.0032836765],"genre_scores_gemma":[0.9969415,0.00009451957,0.0023681985,0.000023748362,0.0000031921534,0.000022688093,0.00030012772,0.000025374056,0.00022057255],"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99901783,0.00050631154,0.00008387634,0.00018346484,0.00014513513,0.00006341845],"domain_scores_gemma":[0.9625133,0.032019857,0.0017786833,0.0022468802,0.0010017391,0.00043957302],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031838361,0.00036344348,0.000595171,0.000533147,0.0003211777,0.0009379605,0.00079371815,0.0008915268,0.0038243488],"category_scores_gemma":[0.041949242,0.00048367566,0.0004433577,0.00050807995,0.0013188029,0.0023259935,0.000831388,0.0005784941,0.00053109234],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0031980632,0.0006971652,0.4520786,0.0015965265,0.0006099124,0.0025820392,0.015028433,0.31572378,0.02909363,0.08568897,0.004875771,0.08882715],"study_design_scores_gemma":[0.0004039623,0.00049743045,0.11398034,0.00012374249,0.00015208065,0.001422846,0.0022897944,0.75186884,0.013038614,0.111554645,0.0045439457,0.00012374308],"about_ca_topic_score_codex":0.004425442,"about_ca_topic_score_gemma":0.0041809534,"teacher_disagreement_score":0.004425442,"about_ca_system_score_codex":0.00066384824,"about_ca_system_score_gemma":0.0005171941,"threshold_uncertainty_score":0.016837955},"labels":[],"label_agreement":null},{"id":"W3174204476","doi":"10.18721/jhss.12209","title":"Searching for multicomponent terms in comparable scientific corpora","year":2021,"lang":"en","type":"article","venue":"St. Petersburg State Polytechnical University Journal","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Education and Early Childhood Development","funders":"","keywords":"Computer science; Linguistics; Epistemology; Natural language processing; History; Philosophy","score_opus":0.021461793423221066,"score_gpt":0.26435196458784016,"score_spread":0.24289017116461908,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3174204476","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.56476384,0.008389934,0.33053476,0.001803788,0.000802969,0.0019834274,0.02632617,0.0033767126,0.06201837],"genre_scores_gemma":[0.5813482,0.0029111276,0.34628797,0.0003282306,0.00043276657,0.0014309033,0.059716374,0.0010181146,0.0065263812],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99505484,0.0014494476,0.0008442652,0.0014267323,0.0010291305,0.0001955467],"domain_scores_gemma":[0.981611,0.010262951,0.0013473338,0.0034133738,0.0030281234,0.00033721433],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0050548613,0.0006563087,0.0009949801,0.011647185,0.0018683233,0.002792646,0.0014174788,0.001001719,0.014093538],"category_scores_gemma":[0.026152035,0.0006481082,0.0008459972,0.013619132,0.0009776419,0.0034498463,0.0028378936,0.0013105033,0.0035991815],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018952596,0.0009612213,0.025256144,0.008873814,0.0007225221,0.013014523,0.014674959,0.006098987,0.17214851,0.053410344,0.02790066,0.67504305],"study_design_scores_gemma":[0.0008684813,0.0012870874,0.12943192,0.0018637003,0.0021986275,0.018681472,0.015859054,0.056892823,0.16144955,0.08743726,0.52362347,0.00040651322],"about_ca_topic_score_codex":0.0014303058,"about_ca_topic_score_gemma":0.0020034772,"teacher_disagreement_score":0.014093538,"about_ca_system_score_codex":0.001028691,"about_ca_system_score_gemma":0.0020897873,"threshold_uncertainty_score":0.04714763},"labels":[],"label_agreement":null},{"id":"W3174951863","doi":"10.18653/v1/2021.findings-acl.257","title":"Hypernym Discovery via a Recurrent Mapping Model","year":2021,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Fundamental Research Funds for the Central Universities; State Key Laboratory of Software Development Environment; National Natural Science Foundation of China","keywords":"Computer science; Data modeling; Artificial intelligence; Information retrieval; Database","score_opus":0.02181117360610446,"score_gpt":0.2640063263932957,"score_spread":0.24219515278719123,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3174951863","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.051514897,0.00087624416,0.9403155,0.0011343898,0.00010965149,0.00017039775,0.00070958724,0.002104005,0.0030653775],"genre_scores_gemma":[0.6679111,0.00075335155,0.31517434,0.00069950725,0.00030396643,0.0004751768,0.0021981238,0.0003731641,0.012111279],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99746335,0.00082272355,0.00016572632,0.00090050447,0.0004952857,0.00015235467],"domain_scores_gemma":[0.99622333,0.002081202,0.0003839107,0.0006847336,0.00046600643,0.00016078433],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027501686,0.0013134718,0.0014590771,0.0029589434,0.00078704435,0.0018805446,0.003454277,0.00233186,0.0035281247],"category_scores_gemma":[0.009446654,0.00076093595,0.0024954586,0.0026825378,0.001121451,0.0062283776,0.0025187607,0.0026130788,0.0014994686],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008244775,0.00078018924,0.009936111,0.00061490096,0.00074361276,0.0014710501,0.0017108803,0.26208714,0.013351558,0.1186092,0.017556025,0.57231486],"study_design_scores_gemma":[0.00002654297,0.00004915731,0.0003462386,0.000013123243,0.000046068122,0.00017413871,0.000036044774,0.9330074,0.0008755548,0.06402659,0.0013773649,0.000021820122],"about_ca_topic_score_codex":0.0032083248,"about_ca_topic_score_gemma":0.0045519094,"teacher_disagreement_score":0.0035281247,"about_ca_system_score_codex":0.0010105405,"about_ca_system_score_gemma":0.0011670826,"threshold_uncertainty_score":0.014544487},"labels":[],"label_agreement":null},{"id":"W3175487467","doi":"10.48550/arxiv.2012.09446","title":"Unsupervised Learning of Discourse Structures using a Tree Autoencoder","year":2020,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Autoencoder; Parsing; Artificial intelligence; Natural language processing; Tree (set theory); Tree structure; Set (abstract data type); Task (project management); Parse tree; Process (computing); Treebank; Annotation; Machine learning; Deep learning; Binary tree","score_opus":0.06558740250914832,"score_gpt":0.21526330955781678,"score_spread":0.14967590704866846,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3175487467","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.027948463,0.00027475724,0.9683875,0.0002478809,0.000050447732,0.00004772412,0.00020258865,0.0016026059,0.0012381256],"genre_scores_gemma":[0.44452894,0.00045548502,0.5462439,0.000281935,0.00010717693,0.00024145785,0.0018484327,0.0002645429,0.0060280445],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99956423,0.00015131866,0.00001943473,0.00016897921,0.000055954115,0.000040044055],"domain_scores_gemma":[0.9988422,0.0008294811,0.00006781048,0.00008264943,0.00014778238,0.000030257734],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00090310216,0.0006712791,0.0005826601,0.00082101574,0.00037267056,0.00063124707,0.0008143325,0.0008611611,0.0013380387],"category_scores_gemma":[0.002291304,0.00048450465,0.00084306515,0.00061481434,0.000544515,0.0012444883,0.0008589415,0.0019284742,0.00078529393],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020771727,0.00021675786,0.0018575909,0.00019330117,0.00016374791,0.00023065008,0.00047805894,0.3373192,0.032883637,0.017950624,0.0050387885,0.60345995],"study_design_scores_gemma":[0.000005646121,0.000018570156,0.00023116653,0.000011631827,0.000011718832,0.000013589547,0.000017602153,0.99140406,0.0031032322,0.0044699484,0.0007075672,0.000005221878],"about_ca_topic_score_codex":0.0032088507,"about_ca_topic_score_gemma":0.008072374,"teacher_disagreement_score":0.0032088507,"about_ca_system_score_codex":0.00064297684,"about_ca_system_score_gemma":0.0009180016,"threshold_uncertainty_score":0.0063803196},"labels":[],"label_agreement":null},{"id":"W3175679307","doi":"10.21428/594757db.acf8fde8","title":"Encoding Dependency Information inside Tree Transformer","year":2021,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Sentence; Computer science; Transformer; Natural language processing; Artificial intelligence; Tree structure; Architecture; Encoding (memory); Data structure; Programming language","score_opus":0.010953681958941783,"score_gpt":0.24902581832451012,"score_spread":0.23807213636556834,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3175679307","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018919488,0.00013164178,0.9750422,0.00016127474,0.00006137847,0.0000524561,0.00038056413,0.0024644672,0.0027864873],"genre_scores_gemma":[0.62211215,0.00039456625,0.37022457,0.00029218473,0.000057333513,0.00016921146,0.0013696225,0.0005795612,0.0048008403],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997663,0.000061016694,0.000021459786,0.000064353124,0.000057738383,0.000029173925],"domain_scores_gemma":[0.9993352,0.0002555496,0.00003955303,0.00017230524,0.00016162886,0.00003568571],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00041751118,0.00050418096,0.0003724871,0.0004089834,0.00022485352,0.0007207659,0.00076717464,0.0005514911,0.0043794666],"category_scores_gemma":[0.002313818,0.00027741666,0.00050493603,0.00061417423,0.00044621326,0.0027178766,0.0010251304,0.0008381475,0.0014580451],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00055327907,0.000121952624,0.0015687689,0.00044968704,0.000081122984,0.0005035329,0.00065570127,0.10779019,0.098281,0.13895065,0.016515585,0.6345286],"study_design_scores_gemma":[0.000025972617,0.0001344794,0.00044811424,0.000030181913,0.00006740319,0.0002912175,0.00006867232,0.83454245,0.0445721,0.10662313,0.013169472,0.00002677943],"about_ca_topic_score_codex":0.0014820345,"about_ca_topic_score_gemma":0.0030737533,"teacher_disagreement_score":0.0043794666,"about_ca_system_score_codex":0.00048596648,"about_ca_system_score_gemma":0.0009247481,"threshold_uncertainty_score":0.014650822},"labels":[],"label_agreement":null},{"id":"W3176031132","doi":"","title":"WaterlooClarke at the Trec 2020 Conversational Assistant Track.","year":2020,"lang":"en","type":"article","venue":"Text REtrieval Conference","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; Toronto Metropolitan University","funders":"","keywords":"Computer science; Track (disk drive); Natural language processing; Information retrieval; Artificial intelligence; World Wide Web; Operating system","score_opus":0.028489320889820827,"score_gpt":0.25937814739525294,"score_spread":0.23088882650543213,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3176031132","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009586497,0.014989157,0.049352422,0.051440634,0.036844406,0.0025930055,0.1813039,0.026069231,0.6278208],"genre_scores_gemma":[0.010336866,0.0024153448,0.0136814015,0.0035945608,0.0023620748,0.00045453265,0.064809516,0.0014924844,0.9008533],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985025,0.00032466266,0.00004972158,0.00030778686,0.0005746954,0.00024060383],"domain_scores_gemma":[0.99679834,0.00038134074,0.000062637344,0.00018895639,0.0014511629,0.0011176522],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0040170853,0.001444672,0.0017685362,0.0017395017,0.0031710607,0.005168783,0.0015665762,0.0018300671,0.32473767],"category_scores_gemma":[0.004308164,0.0005669957,0.00050925236,0.001915611,0.0006377894,0.0044029737,0.0022043316,0.0022221177,0.16095291],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006128644,0.000045177203,0.000059023474,0.000053491392,0.000005339952,0.000016058042,0.000026315718,0.00003722656,0.00073348446,0.00051236514,0.98523116,0.01321903],"study_design_scores_gemma":[0.00008190572,0.00006477211,0.0010873207,0.00006058541,0.00001641769,0.000032087864,0.00018255664,0.0010747551,0.0014577488,0.0016981247,0.9942146,0.000029199231],"about_ca_topic_score_codex":0.1185358,"about_ca_topic_score_gemma":0.36352664,"teacher_disagreement_score":0.32473767,"about_ca_system_score_codex":0.003471375,"about_ca_system_score_gemma":0.0061473493,"threshold_uncertainty_score":0.96317977},"labels":[],"label_agreement":null},{"id":"W3176798136","doi":"10.1007/978-3-030-79942-7_16","title":"The Application of Text Entailment Techniques in COLIEE 2020","year":2021,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Textual entailment; Computer science; Logical consequence; Natural language processing; Fragment (logic); Statute; Artificial intelligence; Context (archaeology); Information retrieval; Programming language; Law; Political science","score_opus":0.008027562739232033,"score_gpt":0.2624456943354297,"score_spread":0.25441813159619764,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3176798136","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010824029,0.001259187,0.9392362,0.0017955097,0.00043690176,0.00037753262,0.0017994215,0.0062738126,0.03799735],"genre_scores_gemma":[0.08316032,0.0008320399,0.87287945,0.0005052611,0.00034140804,0.0002357426,0.005494992,0.0023325835,0.03421828],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99592906,0.0015801118,0.00028484126,0.0006413869,0.0013861728,0.00017833021],"domain_scores_gemma":[0.992539,0.0042714505,0.00021681272,0.0012337894,0.0015850269,0.00015401855],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004175087,0.00079347385,0.0007114795,0.003481566,0.0019675405,0.00407674,0.0022362,0.0013493302,0.024607655],"category_scores_gemma":[0.017395834,0.00085659785,0.0014217326,0.0032274874,0.0012913051,0.006935596,0.0036230553,0.0025614968,0.007841841],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020591677,0.00020613169,0.00063413946,0.0005282566,0.00006478266,0.00029285284,0.0008217848,0.0049795425,0.00904285,0.16184504,0.03718901,0.7841897],"study_design_scores_gemma":[0.00007346949,0.00014711896,0.0015479653,0.0003000634,0.00015363538,0.0009635516,0.0006253155,0.22447217,0.0798593,0.33272806,0.35899353,0.00013585987],"about_ca_topic_score_codex":0.009477661,"about_ca_topic_score_gemma":0.016467936,"teacher_disagreement_score":0.024607655,"about_ca_system_score_codex":0.0018760489,"about_ca_system_score_gemma":0.0028742196,"threshold_uncertainty_score":0.08232081},"labels":[],"label_agreement":null},{"id":"W3181113307","doi":"10.18653/v1/2022.acl-long.87","title":"LexSubCon: Integrating Knowledge from Lexical Resources into Contextual Embeddings for Lexical Substitution","year":2022,"lang":"en","type":"article","venue":"Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Substitution (logic); Computer science; Natural language processing; Artificial intelligence; Embedding; Context (archaeology); Sentence; Word (group theory); Task (project management); Similarity (geometry); Benchmark (surveying); Lexical item; Linguistics; Programming language","score_opus":0.009318653360598592,"score_gpt":0.26338255783588965,"score_spread":0.25406390447529104,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3181113307","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05192531,0.0018677191,0.82373303,0.0009770328,0.00056508475,0.00067456963,0.012491119,0.086707905,0.021058334],"genre_scores_gemma":[0.22342125,0.0010487203,0.7192078,0.0007895234,0.00019454124,0.00077601423,0.03759424,0.005340398,0.01162739],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9990947,0.00022973333,0.000083754814,0.00030345455,0.00020243254,0.000085936816],"domain_scores_gemma":[0.999064,0.00033870377,0.000045096487,0.00031354887,0.0001898557,0.000048743448],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007595406,0.0019109714,0.0010910778,0.0022848032,0.0007400781,0.0021595818,0.0018408869,0.0011243658,0.01672475],"category_scores_gemma":[0.0036631725,0.00073463586,0.0012697085,0.001998828,0.0007002108,0.0059261066,0.0041908496,0.0013696834,0.007814658],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00061121007,0.0005180213,0.0043224227,0.0008011962,0.00028297844,0.0007899771,0.0004770539,0.018232895,0.011157023,0.024876013,0.100901775,0.83702946],"study_design_scores_gemma":[0.00021504718,0.00029780815,0.0024432929,0.00021295167,0.00024503152,0.00075451634,0.0007324844,0.7508488,0.022455491,0.1259713,0.09568084,0.00014242926],"about_ca_topic_score_codex":0.005463846,"about_ca_topic_score_gemma":0.017536378,"teacher_disagreement_score":0.01672475,"about_ca_system_score_codex":0.0007518679,"about_ca_system_score_gemma":0.001563656,"threshold_uncertainty_score":0.055949807},"labels":[],"label_agreement":null},{"id":"W3183679143","doi":"10.18653/v1/2021.mwe-1.1","title":"A Long Hard Look at MWEs in the Age of Language Models","year":2021,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Artificial intelligence; Natural language processing; Meaning (existential); Context (archaeology); Representation (politics); Layer (electronics); Language model; History; Psychology","score_opus":0.020028787561323026,"score_gpt":0.27398885144725704,"score_spread":0.253960063885934,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3183679143","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008837139,0.07384731,0.632966,0.24666312,0.00631293,0.000061912455,0.001505489,0.0027640024,0.02704209],"genre_scores_gemma":[0.27278098,0.111595616,0.44182548,0.072266705,0.031674445,0.00037899063,0.0041208747,0.0054607447,0.059896216],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9957059,0.0017367065,0.0002368061,0.0010204455,0.0011526482,0.0001474443],"domain_scores_gemma":[0.977419,0.01568348,0.0006659451,0.0030771224,0.002472252,0.00068223086],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007431066,0.0016403251,0.0020452898,0.0026818502,0.0016614335,0.008547797,0.002918804,0.0042217835,0.011357114],"category_scores_gemma":[0.044541333,0.0013230881,0.0012728379,0.002433561,0.007492975,0.04398726,0.005323106,0.016573636,0.008717666],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022552478,0.00010462782,0.0022563506,0.00078462187,0.00026597563,0.00022219674,0.00085889036,0.01069072,0.0012763148,0.6774487,0.07946989,0.2263962],"study_design_scores_gemma":[0.000014685832,0.000046235422,0.00037381062,0.000335175,0.00003057313,0.00022032963,0.00035750135,0.025325175,0.0006599732,0.8401357,0.13242874,0.00007206154],"about_ca_topic_score_codex":0.0026958454,"about_ca_topic_score_gemma":0.0027130067,"teacher_disagreement_score":0.011357114,"about_ca_system_score_codex":0.0019388009,"about_ca_system_score_gemma":0.0013387277,"threshold_uncertainty_score":0.039299667},"labels":[],"label_agreement":null},{"id":"W3184106707","doi":"10.18653/v1/2021.semeval-1.101","title":"UAlberta at SemEval-2021 Task 2: Determining Sense Synonymy via Translations","year":2021,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Machine Intelligence Institute","keywords":"SemEval; Computer science; Task (project management); Word-sense disambiguation; Word (group theory); Natural language processing; Focus (optics); Context (archaeology); Artificial intelligence; Linguistics; WordNet","score_opus":0.010810668765250468,"score_gpt":0.2602428169958579,"score_spread":0.24943214823060741,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3184106707","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.049188077,0.012288053,0.23884313,0.009695209,0.004013184,0.0013444156,0.14548892,0.23594342,0.3031957],"genre_scores_gemma":[0.18077014,0.0024566145,0.36858484,0.0026743414,0.0004425323,0.000856366,0.27359983,0.014284368,0.15633097],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99639183,0.0007354611,0.0001548546,0.0012954285,0.0010725304,0.00034974283],"domain_scores_gemma":[0.9962155,0.00078054325,0.00014120723,0.0010276568,0.0013754263,0.00045955295],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036335173,0.0026448998,0.002352962,0.0041619474,0.0041373693,0.007741733,0.0027063077,0.0025977632,0.059166197],"category_scores_gemma":[0.009522701,0.0012758,0.001209763,0.003120124,0.0013432845,0.0050199754,0.0052309255,0.0029370722,0.06564806],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010501936,0.00024847168,0.0024030556,0.00085686275,0.00013160995,0.0004944681,0.00078281475,0.0017917885,0.008560276,0.01729013,0.6527008,0.31368956],"study_design_scores_gemma":[0.00047965877,0.00015547333,0.008783881,0.00047937845,0.00014324939,0.0007239303,0.0010279572,0.040420454,0.019309375,0.041051112,0.88720065,0.00022486398],"about_ca_topic_score_codex":0.1850394,"about_ca_topic_score_gemma":0.2532759,"teacher_disagreement_score":0.1850394,"about_ca_system_score_codex":0.0042132544,"about_ca_system_score_gemma":0.0064743934,"threshold_uncertainty_score":0.3679247},"labels":[],"label_agreement":null},{"id":"W3184137896","doi":"10.1016/j.pragma.2021.06.019","title":"Logical Form – Not logical enough for logic, not linguistic enough for linguistics","year":2021,"lang":"en","type":"article","venue":"Journal of Pragmatics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Linguistics; Sentence; Generative grammar; Utterance; Logical form; Computer science; Atomic sentence; Logical consequence; Representation (politics); Philosophy","score_opus":0.04775607042432032,"score_gpt":0.33888393546512746,"score_spread":0.2911278650408071,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3184137896","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.032976776,0.0120248785,0.49503654,0.24238884,0.009196655,0.00015056557,0.0009919509,0.0012618607,0.20597199],"genre_scores_gemma":[0.7706782,0.0037861185,0.16416436,0.026761424,0.0062792366,0.00031458316,0.0008604709,0.0013601336,0.025795462],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99087924,0.004098753,0.0010168732,0.001693194,0.0018758775,0.00043604168],"domain_scores_gemma":[0.97965425,0.010008675,0.0011654078,0.00446673,0.0037921665,0.00091272965],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009250669,0.0008377084,0.0016933993,0.0018133009,0.0038247358,0.008215865,0.0019869846,0.0048988573,0.009513738],"category_scores_gemma":[0.027593356,0.001354045,0.0013024532,0.0011125538,0.023399329,0.038329177,0.004878884,0.008458427,0.004226802],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000031255746,0.000017159105,0.0002419666,0.0001221839,0.000015935579,0.000056805842,0.0005320068,0.000092738075,0.0003868196,0.9849406,0.0066469503,0.0069156215],"study_design_scores_gemma":[0.0000112019525,0.0000087209455,0.00006563198,0.000039804545,0.000009571821,0.00012291264,0.0002057693,0.000320116,0.0002247817,0.9849128,0.014068513,0.000010193486],"about_ca_topic_score_codex":0.0010463798,"about_ca_topic_score_gemma":0.00072701735,"teacher_disagreement_score":0.009513738,"about_ca_system_score_codex":0.0018684337,"about_ca_system_score_gemma":0.002486939,"threshold_uncertainty_score":0.048922837},"labels":[],"label_agreement":null},{"id":"W3184142975","doi":"10.1145/3462757.3466148","title":"Plum2Text","year":2021,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Université Laval","funders":"","keywords":"Computer science; Annotation; Leverage (statistics); Natural language processing; Table (database); Utterance; Natural language; Artificial intelligence; Paraphrase; Task (project management); Information retrieval; Domain (mathematical analysis); Data mining","score_opus":0.010261531382105987,"score_gpt":0.2613674779868926,"score_spread":0.2511059466047866,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3184142975","genre_codex":"dataset","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0040913206,0.00048129377,0.009265244,0.0008972743,0.00032691334,0.00036354183,0.9101689,0.0504851,0.023920476],"genre_scores_gemma":[0.0051159356,0.00012259885,0.007822229,0.00027943912,0.00002549886,0.00026313515,0.97908735,0.0022076308,0.005076089],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9979348,0.00035731588,0.00016234885,0.0005558256,0.00072600425,0.00026362902],"domain_scores_gemma":[0.9963852,0.00088000676,0.00018398176,0.0013827771,0.0008860769,0.0002819164],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013948407,0.0019262433,0.0007629443,0.0046947766,0.0018233843,0.002571246,0.0032258676,0.0018275279,0.06607636],"category_scores_gemma":[0.008882187,0.0006219918,0.0013265673,0.0048479326,0.00081476255,0.0037318247,0.003848619,0.0021671464,0.060645822],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024843635,0.0000683323,0.0014497023,0.0006549128,0.00003167282,0.00030162052,0.00019343037,0.001393763,0.0010446161,0.005108484,0.96174246,0.027762515],"study_design_scores_gemma":[0.00007276657,0.000032167638,0.0016978839,0.000080769394,0.000011817687,0.0002384368,0.00015583384,0.0038412928,0.0021465966,0.002909772,0.98877174,0.000040837345],"about_ca_topic_score_codex":0.07872315,"about_ca_topic_score_gemma":0.11771991,"teacher_disagreement_score":0.07872315,"about_ca_system_score_codex":0.0024607514,"about_ca_system_score_gemma":0.0037263439,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W3185207032","doi":"10.1186/s41039-021-00163-x","title":"Dr. Mosaik: a holistic framework for understanding the English tense–aspect system based on ontology engineering","year":2021,"lang":"en","type":"article","venue":"Research and Practice in Technology Enhanced Learning","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Sherbrooke; Bishop's University","funders":"Bishop's University","keywords":"Computer science; Ontology; Natural language processing; Mathematics education; Artificial intelligence; Linguistics; Mathematics; Philosophy; Epistemology","score_opus":0.06725506210677408,"score_gpt":0.3890996059478961,"score_spread":0.32184454384112204,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3185207032","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004724477,0.00075798994,0.9682427,0.0031068488,0.00014321861,0.00012006863,0.00075262,0.0029616512,0.019190319],"genre_scores_gemma":[0.10296108,0.0013409546,0.8702834,0.00078046735,0.000092705595,0.00014687872,0.0016079029,0.0009132759,0.021873323],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99948657,0.0001207712,0.000057353103,0.00012563956,0.0001483099,0.000061304454],"domain_scores_gemma":[0.9996469,0.000100765326,0.000031737138,0.000066682784,0.00010267369,0.000051175415],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012514764,0.00073140685,0.00044538398,0.002251411,0.00087240874,0.0039661657,0.0011425758,0.0012784725,0.0083814],"category_scores_gemma":[0.0018720921,0.00044414387,0.0009272692,0.0013171266,0.0020150815,0.009345708,0.0024106347,0.002573891,0.003261865],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000077184035,0.00004402157,0.0016611262,0.0004521912,0.00004140242,0.00067687035,0.004844724,0.002491814,0.0100428965,0.74462783,0.022696644,0.21234329],"study_design_scores_gemma":[0.000024817731,0.000037712067,0.0015126207,0.0003456961,0.000086833636,0.0018388074,0.003231818,0.040133994,0.009928807,0.4168704,0.52591234,0.00007613725],"about_ca_topic_score_codex":0.0052533685,"about_ca_topic_score_gemma":0.006970245,"teacher_disagreement_score":0.0083814,"about_ca_system_score_codex":0.0009791586,"about_ca_system_score_gemma":0.001668644,"threshold_uncertainty_score":0.028038561},"labels":[],"label_agreement":null},{"id":"W3185386956","doi":"","title":"Coarse \"split and lump\" bilingual language models for richer source information in SMT.","year":2014,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Cluster analysis; Machine translation; Natural language processing; Language model; Word (group theory); Artificial intelligence; Phrase; Sentence; Speech recognition; Linguistics","score_opus":0.009044135863205661,"score_gpt":0.2524401928076209,"score_spread":0.24339605694441527,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3185386956","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.097837284,0.0011950325,0.8754168,0.00079144986,0.00019996126,0.00020431547,0.0020998581,0.015949754,0.006305492],"genre_scores_gemma":[0.6659897,0.00027588368,0.31986004,0.00063086295,0.000112186295,0.00038855185,0.0060780193,0.0014886833,0.005176059],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988776,0.0005994936,0.000050103947,0.00029344697,0.0001148318,0.000064569045],"domain_scores_gemma":[0.99800843,0.0010538892,0.00008391426,0.0005853706,0.00016964052,0.00009880362],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019479878,0.0015747892,0.0008261667,0.0010886177,0.00057884626,0.001175083,0.0012100056,0.001083584,0.0063839518],"category_scores_gemma":[0.005834064,0.00059155416,0.0010765786,0.0009479541,0.00060154806,0.0030028638,0.002043454,0.0023150612,0.004733335],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015473476,0.00051489886,0.005927668,0.0005451418,0.00053963304,0.00029790227,0.0006182258,0.44390967,0.031374987,0.022879483,0.023943711,0.46790138],"study_design_scores_gemma":[0.000073608346,0.00012479357,0.00084710727,0.000027347342,0.000051879506,0.000085531756,0.00007289022,0.96541417,0.006200868,0.022617254,0.004452356,0.00003217698],"about_ca_topic_score_codex":0.004568937,"about_ca_topic_score_gemma":0.016414938,"teacher_disagreement_score":0.0063839518,"about_ca_system_score_codex":0.00080817105,"about_ca_system_score_gemma":0.0011912249,"threshold_uncertainty_score":0.021356463},"labels":[],"label_agreement":null},{"id":"W3185437894","doi":"10.18653/v1/2021.mwe-1.4","title":"Contextualized Embeddings Encode Monolingual and Cross-lingual Knowledge of Idiomaticity","year":2021,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of New Brunswick","funders":"Natural Sciences and Engineering Research Council of Canada; New Brunswick Innovation Foundation","keywords":"Literal (mathematical logic); Computer science; Natural language processing; ENCODE; Artificial intelligence; Meaning (existential); Linguistics; Interpretation (philosophy); Psychology","score_opus":0.01638520130137001,"score_gpt":0.33816256705671216,"score_spread":0.32177736575534216,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3185437894","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.47781384,0.0016537608,0.42094067,0.0021823982,0.0010428735,0.0001487302,0.018066479,0.007086321,0.07106493],"genre_scores_gemma":[0.9237479,0.0004219679,0.06482623,0.00016891396,0.00008517489,0.000057570687,0.0056728977,0.0012182642,0.003801068],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9992047,0.00021893175,0.00007079382,0.0003178534,0.00009442152,0.00009329772],"domain_scores_gemma":[0.99691474,0.0012023073,0.00023543097,0.00085600076,0.00069335004,0.00009817115],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00064142555,0.0008753798,0.00038121702,0.0010204031,0.00060377497,0.002796171,0.0007200876,0.00082426256,0.010663709],"category_scores_gemma":[0.00583313,0.0006322119,0.0004588893,0.0013446797,0.00079137913,0.0077880863,0.0029575373,0.0018034952,0.0031833474],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016921428,0.0004662922,0.02238772,0.002153102,0.00030806148,0.0022576929,0.015183523,0.024542294,0.07201145,0.270816,0.03913182,0.54905],"study_design_scores_gemma":[0.00017303461,0.0003956885,0.018566305,0.00083722844,0.0006193961,0.0020305219,0.011505126,0.26254737,0.055301327,0.45826167,0.1893294,0.00043310545],"about_ca_topic_score_codex":0.0019743002,"about_ca_topic_score_gemma":0.005584284,"teacher_disagreement_score":0.010663709,"about_ca_system_score_codex":0.00076951913,"about_ca_system_score_gemma":0.0007005075,"threshold_uncertainty_score":0.035673678},"labels":[],"label_agreement":null},{"id":"W3185884227","doi":"10.17533/udea.mut.v14n2a10","title":"Cadlaws – An English–French Parallel Corpus of Legally Equivalent Documents","year":2021,"lang":"en","type":"article","venue":"Mutatis Mutandis Revista Latinoamericana de Traducción","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Natural language processing; Machine translation; Parallel corpora; Corpus linguistics; Artificial intelligence; Meaning (existential); Linguistics; Translation (biology); Baseline (sea); Psychology; Political science; Law","score_opus":0.01316222300862668,"score_gpt":0.28249666696990344,"score_spread":0.26933444396127676,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3185884227","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.31828138,0.0071573188,0.032763597,0.0029672696,0.0013210214,0.003341808,0.5226185,0.0056562005,0.105892874],"genre_scores_gemma":[0.3095344,0.0018731426,0.05867965,0.0005246812,0.00021467493,0.002071767,0.59670055,0.0011351143,0.029265992],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99758387,0.0005165172,0.00020119791,0.00052860216,0.0009035309,0.0002661974],"domain_scores_gemma":[0.9939658,0.0016831008,0.0002666329,0.0005542261,0.0032161917,0.0003140721],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019018803,0.00081415713,0.00057208916,0.009271092,0.0043819183,0.0020954462,0.0012312387,0.0008089407,0.016214825],"category_scores_gemma":[0.009237751,0.0003532742,0.000463956,0.0076057613,0.0019103731,0.0010544083,0.0012807071,0.0010934612,0.0032416603],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010459971,0.00063040346,0.011553203,0.0037161207,0.0001692532,0.0037900815,0.010869456,0.0054056244,0.020186855,0.04273551,0.511974,0.38792357],"study_design_scores_gemma":[0.00019451269,0.00007279021,0.031591084,0.0003996616,0.00006753658,0.0011292213,0.003548669,0.0034595237,0.0076990016,0.0022428096,0.94948125,0.00011402425],"about_ca_topic_score_codex":0.6728865,"about_ca_topic_score_gemma":0.68485916,"teacher_disagreement_score":0.6728865,"about_ca_system_score_codex":0.008959057,"about_ca_system_score_gemma":0.017179985,"threshold_uncertainty_score":0.65807986},"labels":[],"label_agreement":null},{"id":"W3185893030","doi":"","title":"Expressive hierarchical rule extraction for left-to-right translation","year":2014,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Synchronous context-free grammar; Computer science; Machine translation; Rule-based machine translation; Decoding methods; Grammar; Natural language processing; Transfer-based machine translation; Artificial intelligence; Sentence; Translation (biology); Phrase; Grammar induction; Example-based machine translation; Speech recognition; Theoretical computer science; Algorithm; Linguistics","score_opus":0.013481965533986772,"score_gpt":0.2965913420232593,"score_spread":0.28310937648927254,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3185893030","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01921714,0.000091041875,0.97568125,0.00007966953,0.000012360346,0.000067986315,0.00026919745,0.0025612405,0.0020201788],"genre_scores_gemma":[0.17250106,0.0000794287,0.8239329,0.000072409675,0.0000112039925,0.00009597116,0.00090879196,0.00044336196,0.0019548659],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99914455,0.00023778054,0.000080567515,0.00017379067,0.0002846726,0.000078726924],"domain_scores_gemma":[0.9986702,0.0007483571,0.000066521825,0.00027327504,0.00021343069,0.000028098539],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00075346377,0.000585502,0.0008099731,0.0008554254,0.00054105825,0.0009603739,0.000938415,0.00084805087,0.004609244],"category_scores_gemma":[0.0027004262,0.00045866537,0.0008865682,0.0010660655,0.00072783226,0.0012287631,0.00095597556,0.0011638508,0.0019209902],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018369095,0.00015011877,0.0012289591,0.00034963325,0.00005421377,0.00040004973,0.00036832647,0.20880178,0.08004644,0.047406215,0.0046708463,0.65633976],"study_design_scores_gemma":[0.000030226836,0.00005219398,0.00035805802,0.000014255124,0.000021569043,0.00018163086,0.00005592296,0.9256976,0.034722272,0.03407069,0.0047719767,0.00002359368],"about_ca_topic_score_codex":0.0032765886,"about_ca_topic_score_gemma":0.0061791386,"teacher_disagreement_score":0.004609244,"about_ca_system_score_codex":0.0006994635,"about_ca_system_score_gemma":0.0011900845,"threshold_uncertainty_score":0.015419483},"labels":[],"label_agreement":null},{"id":"W3186655327","doi":"10.18653/v1/2021.gem-1.10","title":"The GEM Benchmark: Natural Language Generation, its Evaluation and Metrics","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":148,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Benchmark (surveying); Natural language generation; Natural language; Natural language processing; Geology","score_opus":0.02864547784444409,"score_gpt":0.3241813393370074,"score_spread":0.2955358614925633,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3186655327","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3233838,0.049206674,0.112384684,0.0056016184,0.006489156,0.0055697584,0.20987695,0.20769693,0.07979043],"genre_scores_gemma":[0.22484341,0.006147384,0.17927594,0.00156742,0.0005987667,0.00289335,0.5564252,0.00953288,0.018715663],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9887658,0.0046043964,0.0012983396,0.0018331571,0.0028828448,0.00061542954],"domain_scores_gemma":[0.9848706,0.007933636,0.0005924458,0.0028384647,0.002835921,0.0009289351],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.008712871,0.0040729223,0.0023989128,0.007818285,0.001578446,0.0046361987,0.0047950097,0.003105307,0.00944643],"category_scores_gemma":[0.028350497,0.00087156455,0.002397244,0.005908105,0.0011890301,0.005483366,0.0035525465,0.002529169,0.0066967257],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0024531337,0.0023503124,0.006666478,0.0046621608,0.0009118736,0.00065370713,0.00050900085,0.08288432,0.0056647146,0.005652303,0.42802528,0.45956674],"study_design_scores_gemma":[0.0036121772,0.004047517,0.014810385,0.0012618791,0.0010562285,0.0026369868,0.0014921015,0.6626534,0.030660268,0.030453604,0.2468436,0.00047183837],"about_ca_topic_score_codex":0.024110375,"about_ca_topic_score_gemma":0.021415763,"teacher_disagreement_score":0.9912871,"about_ca_system_score_codex":0.0033809766,"about_ca_system_score_gemma":0.003861679,"threshold_uncertainty_score":0.047940075},"labels":[],"label_agreement":null},{"id":"W3186734340","doi":"","title":"Bayesian iterative-cascade framework for hierarchical phrase-based translation.","year":2014,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Pipeline (software); Translation (biology); Machine translation; Artificial intelligence; Phrase; Synchronous context-free grammar; Cascade; Natural language processing; Machine learning; Iterative method; Scope (computer science); Bayesian probability; Transfer-based machine translation; Example-based machine translation; Algorithm; Programming language; Engineering","score_opus":0.01851834961854546,"score_gpt":0.29659750797848394,"score_spread":0.27807915835993846,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3186734340","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0016593221,0.00018916136,0.99485886,0.000099033576,0.000025169516,0.00006688031,0.00008583083,0.0015074752,0.001508238],"genre_scores_gemma":[0.15980507,0.0003943512,0.8322125,0.0002671318,0.0000962426,0.00043969188,0.00079340884,0.00048863096,0.005502988],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9983358,0.0007529845,0.00007217329,0.00032930484,0.00039467844,0.00011521122],"domain_scores_gemma":[0.99844307,0.0007797267,0.000118937336,0.00026780227,0.0003089251,0.00008159598],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026813287,0.0014730316,0.0010213314,0.00081288,0.00075881387,0.0008795602,0.0029609124,0.001687345,0.008142301],"category_scores_gemma":[0.0055717477,0.0010067294,0.001007618,0.0011323651,0.001082107,0.0023813138,0.0017029438,0.0023146812,0.005335342],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00048975344,0.00031749846,0.0011265791,0.00054736575,0.00022977378,0.00033595308,0.00055669755,0.3695454,0.026881142,0.10041031,0.018285008,0.48127446],"study_design_scores_gemma":[0.000020385403,0.000084627485,0.00018826206,0.000016480712,0.000020893065,0.00007403901,0.000014858486,0.96512246,0.004216182,0.02634239,0.0038763618,0.000022946848],"about_ca_topic_score_codex":0.0065953806,"about_ca_topic_score_gemma":0.015291421,"teacher_disagreement_score":0.008142301,"about_ca_system_score_codex":0.0011945041,"about_ca_system_score_gemma":0.0023585823,"threshold_uncertainty_score":0.027238667},"labels":[],"label_agreement":null},{"id":"W3186788573","doi":"10.35111/srvm-p675","title":"LORELEI Akan Representative Language Pack","year":2021,"lang":"en","type":"dataset","venue":"Linguistic Data Consortium Catalog","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Annotation; Natural language processing; XML; Artificial intelligence; Context (archaeology); Resource (disambiguation); Sentence; Lemmatisation; Information retrieval; World Wide Web","score_opus":0.029608894861936852,"score_gpt":0.33350167127389563,"score_spread":0.3038927764119588,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3186788573","genre_codex":"other","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013835522,0.001617411,0.0464054,0.00488804,0.0019954778,0.0016820387,0.38727912,0.083443016,0.45885405],"genre_scores_gemma":[0.032997478,0.0011992311,0.06673688,0.0015048214,0.00062933244,0.002793233,0.5173878,0.019367803,0.35738343],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99720037,0.00061595556,0.00029638378,0.0006403611,0.0010290426,0.00021795461],"domain_scores_gemma":[0.98885274,0.0023932767,0.00057698047,0.0025970028,0.00503947,0.0005403791],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0030450919,0.0013736639,0.00093771704,0.005037313,0.001521139,0.0043932605,0.0020052572,0.0007898976,0.30193996],"category_scores_gemma":[0.009225233,0.0009789636,0.00051587704,0.0051115826,0.00072462065,0.005877991,0.004186717,0.0016894428,0.2748518],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023474619,0.000044986104,0.0007070818,0.00044021208,0.000009297602,0.00018775825,0.0006234452,0.00022127843,0.003292623,0.0032436457,0.89637405,0.09462098],"study_design_scores_gemma":[0.0000143151165,0.000015808198,0.0009552281,0.00007740709,0.0000038153316,0.00013577816,0.00029022578,0.0004418933,0.0020560524,0.0007523025,0.995233,0.000024197123],"about_ca_topic_score_codex":0.0132698165,"about_ca_topic_score_gemma":0.012768554,"teacher_disagreement_score":0.30193996,"about_ca_system_score_codex":0.0018648144,"about_ca_system_score_gemma":0.0027931957,"threshold_uncertainty_score":0.9956979},"labels":[],"label_agreement":null},{"id":"W3186976180","doi":"","title":"Pivot-based triangulation for low-resource languages","year":2014,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Triangulation; Computer science; Phrase; Translation (biology); Resource (disambiguation); Natural language processing; Quality (philosophy); Machine translation; Domain (mathematical analysis); Artificial intelligence; Adaptation (eye); Domain adaptation; Mathematics","score_opus":0.00818984215679565,"score_gpt":0.2715831359743267,"score_spread":0.26339329381753107,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3186976180","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03954224,0.00024959058,0.9542367,0.00011725953,0.000045652178,0.00017609406,0.00036326694,0.0024205465,0.0028486256],"genre_scores_gemma":[0.18926238,0.0002053897,0.80596864,0.00008127361,0.000023844954,0.00026358655,0.001413223,0.0008041423,0.0019774893],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9939427,0.0032192469,0.00049025286,0.0010549611,0.0010519706,0.00024081813],"domain_scores_gemma":[0.98876476,0.005839932,0.0007421741,0.0028528038,0.0016116804,0.00018870334],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003962046,0.0009204881,0.0010586083,0.0017584288,0.0014310585,0.0017922454,0.0016309089,0.0008931073,0.005404411],"category_scores_gemma":[0.01895671,0.00093644205,0.0011421737,0.0030592764,0.0014609284,0.0038229742,0.00457203,0.001512188,0.0035078528],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012298751,0.00017265916,0.008667537,0.0015856877,0.00032346515,0.0007068178,0.0077182916,0.05009541,0.15081473,0.03244853,0.006637061,0.7395999],"study_design_scores_gemma":[0.00032123388,0.0010553903,0.008251515,0.00039643914,0.00026122708,0.0019395446,0.00792854,0.60414225,0.22234376,0.06382443,0.089077644,0.00045797662],"about_ca_topic_score_codex":0.0031646758,"about_ca_topic_score_gemma":0.006684804,"teacher_disagreement_score":0.005404411,"about_ca_system_score_codex":0.0006740254,"about_ca_system_score_gemma":0.0015411407,"threshold_uncertainty_score":0.020953536},"labels":[],"label_agreement":null},{"id":"W3188280450","doi":"10.48550/arxiv.2108.03533","title":"Improving Similar Language Translation With Transfer Learning","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Catalan; Machine translation; Portuguese; Computer science; Natural language processing; Task (project management); Artificial intelligence; Transfer of learning; Rank (graph theory); Translation (biology); Linguistics; Mathematics; Philosophy; Engineering","score_opus":0.032462875005548765,"score_gpt":0.18454350822925286,"score_spread":0.1520806332237041,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3188280450","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.114906184,0.0039233053,0.82307684,0.0020315517,0.0013156586,0.0004189658,0.0013175796,0.026742224,0.026267797],"genre_scores_gemma":[0.69814926,0.0014146902,0.26588133,0.0014424272,0.0007001123,0.00042956727,0.0073222783,0.0022879327,0.022372413],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9977894,0.0008446,0.00011280164,0.0006843687,0.00036213067,0.00020669251],"domain_scores_gemma":[0.9960646,0.0015409284,0.00017309903,0.0013789188,0.0007150704,0.00012735945],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030983384,0.0019764362,0.001432744,0.0014979844,0.00094153214,0.00202502,0.0024034441,0.0018494028,0.010018294],"category_scores_gemma":[0.011374614,0.00045249113,0.0014508844,0.0019505454,0.00096167414,0.0048884554,0.0028515037,0.0028701923,0.009273098],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005557011,0.0011128504,0.0028457525,0.0005217461,0.00038848113,0.0003648826,0.00035130634,0.24446437,0.0147032775,0.015358519,0.032027252,0.68730587],"study_design_scores_gemma":[0.00010429034,0.00030784585,0.0004924033,0.000044020722,0.000083425984,0.00015315952,0.00010589733,0.94571924,0.012411756,0.032637134,0.007902328,0.000038451988],"about_ca_topic_score_codex":0.0047019524,"about_ca_topic_score_gemma":0.004809899,"teacher_disagreement_score":0.010018294,"about_ca_system_score_codex":0.0010927723,"about_ca_system_score_gemma":0.001528836,"threshold_uncertainty_score":0.0335145},"labels":[],"label_agreement":null},{"id":"W3189820663","doi":"10.18653/v1/2022.naacl-main.178","title":"WiC = TSV = WSD: On the Equivalence of Three Semantic Tasks","year":2022,"lang":"en","type":"article","venue":"Proceedings of the 2022 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Machine Intelligence Institute","keywords":"Computer science; SemEval; Equivalence (formal languages); Natural language processing; Pairwise comparison; Task (project management); Semantic equivalence; Artificial intelligence; Word (group theory); Context (archaeology); Popularity; Semantic similarity; Word-sense disambiguation; WordNet; Linguistics; Semantic Web; Semantic computing; Psychology","score_opus":0.021952110168749,"score_gpt":0.2601831011040238,"score_spread":0.2382309909352748,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3189820663","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06741209,0.0011170263,0.8991251,0.0063344957,0.0005945058,0.00071985816,0.0020871405,0.0027936033,0.01981627],"genre_scores_gemma":[0.45235327,0.00075856346,0.5268336,0.001607852,0.0004760148,0.001123689,0.00956057,0.0009860203,0.0063004214],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9816098,0.007332444,0.0017663843,0.0051547945,0.0031130747,0.0010234412],"domain_scores_gemma":[0.95516056,0.024979006,0.0025105018,0.011846207,0.0041111507,0.0013925688],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.016153827,0.0013400071,0.0021192075,0.00314033,0.002929813,0.008039965,0.00335785,0.0041210456,0.0073062773],"category_scores_gemma":[0.06776414,0.00065766997,0.0025493877,0.003253804,0.0060655517,0.020799331,0.013475545,0.007106788,0.0023359514],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011307608,0.00046705906,0.008509446,0.00080836104,0.00019885926,0.00021397037,0.0014973326,0.014891495,0.004202111,0.5940169,0.024742156,0.34932148],"study_design_scores_gemma":[0.0001041398,0.00021390202,0.001979668,0.00009575088,0.0000706722,0.0002561315,0.00067127065,0.13826372,0.0039135343,0.84397435,0.010390311,0.00006651175],"about_ca_topic_score_codex":0.0062089222,"about_ca_topic_score_gemma":0.0039770515,"teacher_disagreement_score":0.016153827,"about_ca_system_score_codex":0.0023169045,"about_ca_system_score_gemma":0.0056241443,"threshold_uncertainty_score":0.08543062},"labels":[],"label_agreement":null},{"id":"W3191193672","doi":"10.3968/12146","title":"Etude sur la définition selon le modèle schéma-extension","year":2021,"lang":"fr","type":"article","venue":"Canadian social science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Political science; Philosophy","score_opus":0.03036445504631277,"score_gpt":0.26892130080445037,"score_spread":0.2385568457581376,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3191193672","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009755667,0.003753808,0.9521485,0.0055635176,0.00048898335,0.000271403,0.00062109367,0.0007364519,0.026660625],"genre_scores_gemma":[0.12333579,0.0047758003,0.84161466,0.0019424354,0.00038950483,0.0005302107,0.002428211,0.0008317261,0.02415171],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9922429,0.0028808494,0.0008577335,0.0012570013,0.00247746,0.00028406616],"domain_scores_gemma":[0.9878611,0.0053144577,0.0007089519,0.0027795476,0.0030238994,0.00031211815],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010354735,0.0010792937,0.001036372,0.0023004124,0.002123746,0.008786423,0.0028936355,0.002876854,0.008199885],"category_scores_gemma":[0.015400211,0.0012776144,0.002569081,0.0035244816,0.006276382,0.015593577,0.004382726,0.0063657723,0.002713056],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006277916,0.000061880644,0.0012231896,0.00043318613,0.00006784169,0.00029035463,0.004431942,0.0025101406,0.0036040968,0.9361084,0.0039592166,0.047246877],"study_design_scores_gemma":[0.000044735727,0.00013029229,0.0012236111,0.0013225193,0.00013968755,0.0021858893,0.0026948957,0.027612187,0.009179806,0.27068043,0.684666,0.000119974575],"about_ca_topic_score_codex":0.010996542,"about_ca_topic_score_gemma":0.0090082595,"teacher_disagreement_score":0.010996542,"about_ca_system_score_codex":0.0034335828,"about_ca_system_score_gemma":0.0055334065,"threshold_uncertainty_score":0.054761767},"labels":[],"label_agreement":null},{"id":"W3194817747","doi":"10.82308/39461","title":"Verb-stranding VP ellipsis : a cross-linguistic study","year":2005,"lang":"en","type":"article","venue":"eScholarship@McGill (McGill)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":226,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"McGill University","keywords":"Linguistics; Ellipsis (linguistics); Verb; Verb phrase ellipsis; Computer science; Artificial intelligence; Natural language processing; Modal verb; Philosophy","score_opus":0.020049308654681346,"score_gpt":0.28699706202285924,"score_spread":0.2669477533681779,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3194817747","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.92181534,0.0020034607,0.0061135646,0.00049543544,0.000073456446,0.000047349266,0.00011989302,0.000039734063,0.069291756],"genre_scores_gemma":[0.9912117,0.0009851726,0.0020789765,0.00016458853,0.000072280716,0.000034291632,0.00013341186,0.000050282044,0.005269275],"study_design_codex":"qualitative","study_design_gemma":"observational","domain_scores_codex":[0.99861157,0.00065683376,0.00010123464,0.00028682232,0.00021749527,0.00012611935],"domain_scores_gemma":[0.9962134,0.0026670767,0.0003584034,0.00036657392,0.00031418484,0.00008027395],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002534123,0.0004273853,0.00032675275,0.0020116637,0.0024461627,0.003115189,0.00068213645,0.0010189746,0.0052727438],"category_scores_gemma":[0.0058333785,0.00035649928,0.00027555326,0.0034326334,0.0025033108,0.0039894124,0.002746244,0.0011613773,0.0006221549],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013684778,0.00045366294,0.072643414,0.00057784026,0.000095354735,0.0075008427,0.72672874,0.00026448528,0.014688651,0.09231065,0.0019663908,0.082633145],"study_design_scores_gemma":[0.000051092906,0.0004254205,0.31510627,0.0007045407,0.00020291467,0.019993918,0.42409173,0.0028914306,0.009580446,0.028347922,0.19841224,0.00019215551],"about_ca_topic_score_codex":0.0065923193,"about_ca_topic_score_gemma":0.009218864,"teacher_disagreement_score":0.0065923193,"about_ca_system_score_codex":0.0009025317,"about_ca_system_score_gemma":0.00045153915,"threshold_uncertainty_score":0.0176391},"labels":[],"label_agreement":null},{"id":"W3196115631","doi":"10.5430/elr.v10n3p66","title":"Should We Use It in Our Classrooms: An Analysis of Data-Driven Learning Research","year":2021,"lang":"en","type":"article","venue":"English Linguistics Research","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Variety (cybernetics); Set (abstract data type); Computer science; Applied linguistics; Field (mathematics); Perception; Corpus linguistics; Linguistics; Mathematics education; Psychology; Natural language processing; Artificial intelligence","score_opus":0.3211733851380461,"score_gpt":0.504369009391842,"score_spread":0.18319562425379593,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3196115631","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9401713,0.011496373,0.012444244,0.012160774,0.000115107505,0.0009541364,0.0005859265,0.000067034955,0.02200515],"genre_scores_gemma":[0.9863046,0.0031958595,0.007787988,0.0010396069,0.000028888737,0.00032164142,0.0002593436,0.000058481426,0.0010036639],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9679487,0.01827001,0.0018365869,0.002462164,0.0084153265,0.0010672406],"domain_scores_gemma":[0.65190786,0.29255447,0.013687586,0.006002933,0.031321622,0.004525598],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.055721264,0.0003605476,0.00090955116,0.009054802,0.0022096173,0.008585489,0.002563986,0.001599411,0.001592553],"category_scores_gemma":[0.14593622,0.0005235501,0.00077268435,0.007266767,0.003420527,0.0075719464,0.002851162,0.0022429738,0.00045659332],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00039990517,0.001563231,0.27995047,0.004715541,0.0003898719,0.00074360607,0.28399396,0.0007921404,0.0015057228,0.04563756,0.0049207476,0.3753872],"study_design_scores_gemma":[0.00013233314,0.00095778465,0.28866923,0.010193509,0.00053106004,0.0007519459,0.53788304,0.0112869395,0.005486652,0.029629884,0.11420947,0.00026819587],"about_ca_topic_score_codex":0.0069773514,"about_ca_topic_score_gemma":0.008636704,"teacher_disagreement_score":0.055721264,"about_ca_system_score_codex":0.007906428,"about_ca_system_score_gemma":0.009233475,"threshold_uncertainty_score":0.29468578},"labels":[],"label_agreement":null},{"id":"W3196848038","doi":"","title":"Open Problem: Are all VC-classes CPAC learnable?","year":2021,"lang":"en","type":"article","venue":"Conference on Learning Theory","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University; University of Waterloo","funders":"","keywords":"Computer science","score_opus":0.042294675341726054,"score_gpt":0.3155514981952951,"score_spread":0.27325682285356906,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3196848038","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.44881222,0.0049186316,0.37941355,0.042925604,0.0018126754,0.00036740134,0.0143228425,0.0037425293,0.10368457],"genre_scores_gemma":[0.9430348,0.0011586003,0.032793738,0.0024528808,0.0009351522,0.0002851876,0.0058232252,0.00044727852,0.0130691],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9956664,0.0005621511,0.0001806147,0.0022710802,0.00062350027,0.0006961694],"domain_scores_gemma":[0.9696998,0.020457242,0.00149813,0.003978292,0.0027301519,0.0016363502],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023759583,0.0010188736,0.0017306316,0.0011803135,0.0030853825,0.008209666,0.0037583727,0.005286409,0.02491465],"category_scores_gemma":[0.030739926,0.0008364489,0.0015035162,0.002179039,0.0042169234,0.018273905,0.0034168588,0.009587871,0.002675443],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00074711046,0.00060325995,0.013076667,0.0011520179,0.00026279158,0.00016588513,0.00080346153,0.017118819,0.0021377981,0.5724533,0.08667717,0.3048017],"study_design_scores_gemma":[0.00008549846,0.00010650577,0.0029637127,0.00016367878,0.00005472371,0.00020395579,0.0005863958,0.04003898,0.002019198,0.93897057,0.014762993,0.000043733526],"about_ca_topic_score_codex":0.003838206,"about_ca_topic_score_gemma":0.003478912,"teacher_disagreement_score":0.02491465,"about_ca_system_score_codex":0.0024798834,"about_ca_system_score_gemma":0.003382986,"threshold_uncertainty_score":0.0833478},"labels":[],"label_agreement":null},{"id":"W3196887708","doi":"10.1007/978-3-030-86159-9_37","title":"Graph Representation Learning in Document Wikification","year":2021,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Knowledge base; Paragraph; Representation (politics); Entity linking; Graph; Task (project management); Annotation; Word (group theory); Information retrieval; Context (archaeology); Natural language understanding; Natural language; Theoretical computer science; World Wide Web; Linguistics","score_opus":0.015304455452964782,"score_gpt":0.28531269871287535,"score_spread":0.2700082432599106,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3196887708","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.030439926,0.0033717153,0.9503955,0.0006219043,0.00033149097,0.00013594661,0.0008772448,0.0055842153,0.008242042],"genre_scores_gemma":[0.2163337,0.0023220698,0.7553703,0.00018182273,0.00015463136,0.00019715374,0.005340837,0.00079710165,0.019302303],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993643,0.0002550603,0.0000477813,0.00015039994,0.00014024993,0.00004219887],"domain_scores_gemma":[0.9976683,0.0013909249,0.00009978489,0.00049093267,0.00026603136,0.000083998624],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010375491,0.0005118372,0.0008065558,0.0018619195,0.0006176829,0.001253655,0.0016520937,0.0008460121,0.003820438],"category_scores_gemma":[0.00507896,0.00035208097,0.0007302726,0.002958852,0.00047412087,0.0033750962,0.0013354841,0.0015926856,0.0013803113],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008045003,0.00017401602,0.0009945852,0.00034074357,0.00007352635,0.000072617855,0.00019509492,0.035118807,0.0023818891,0.030567564,0.022486841,0.90751386],"study_design_scores_gemma":[0.000026866994,0.00008863231,0.0011204302,0.00015178224,0.000080221966,0.00021229874,0.00022419514,0.8073606,0.0077264905,0.1499133,0.03305591,0.00003934944],"about_ca_topic_score_codex":0.0044695763,"about_ca_topic_score_gemma":0.00867007,"teacher_disagreement_score":0.0044695763,"about_ca_system_score_codex":0.0008477491,"about_ca_system_score_gemma":0.0009247485,"threshold_uncertainty_score":0.012780666},"labels":[],"label_agreement":null},{"id":"W3196963091","doi":"","title":"In French, DLD is TDL !","year":2021,"lang":"en","type":"article","venue":"ORBi (University of Liège)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science","score_opus":0.009235066172209845,"score_gpt":0.21669828557035162,"score_spread":0.20746321939814177,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3196963091","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.027534636,0.060203418,0.012289772,0.1663388,0.04337846,0.00027259573,0.034582693,0.005951986,0.6494476],"genre_scores_gemma":[0.24562702,0.03270516,0.015932016,0.06198907,0.0057886504,0.0005129029,0.026546307,0.0027036362,0.60819525],"study_design_codex":"not_applicable","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9980161,0.0005343092,0.00019713264,0.0003370304,0.00059640734,0.0003189919],"domain_scores_gemma":[0.99815035,0.00049451593,0.00018930112,0.00013764163,0.000883839,0.00014438151],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017522183,0.0008810763,0.00054200256,0.0015935428,0.002558149,0.0029394259,0.0005807553,0.001763338,0.07409785],"category_scores_gemma":[0.0034966348,0.00018712267,0.0007850231,0.00135807,0.0020373662,0.0015315119,0.0014853403,0.0024256387,0.028404161],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013112796,0.000020483207,0.0024627105,0.00064047764,0.000015109795,0.0009061892,0.0028381967,0.0000734285,0.0010603573,0.05531612,0.7963371,0.14019872],"study_design_scores_gemma":[0.0000052817545,0.000011400453,0.0012290096,0.0001062129,0.0000020022287,0.0006082478,0.00044857955,0.000017113456,0.00011845787,0.00047647377,0.9969658,0.000011421282],"about_ca_topic_score_codex":0.11110916,"about_ca_topic_score_gemma":0.06637483,"teacher_disagreement_score":0.11110916,"about_ca_system_score_codex":0.005734587,"about_ca_system_score_gemma":0.0045833,"threshold_uncertainty_score":0.24788195},"labels":[],"label_agreement":null},{"id":"W3197059915","doi":"10.26615/978-954-452-071-7_004","title":"Machine translation use outside the language industries: a comparison of five delivery formats for machine translation literacy instruction","year":2021,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Social Sciences and Humanities Research Council of Canada; Concordia University","keywords":"Machine translation; Computer science; Residence; Literacy; Translation (biology); Mathematics education; Natural language processing; Artificial intelligence; Library science; Sociology; Pedagogy; Psychology","score_opus":0.033093338073364946,"score_gpt":0.31648440325720556,"score_spread":0.2833910651838406,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3197059915","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9397157,0.0040453086,0.0060687265,0.0029682787,0.00019166611,0.00061408745,0.0010784711,0.00038498914,0.04493285],"genre_scores_gemma":[0.97975314,0.0033119482,0.007643116,0.00069473335,0.00008222109,0.0005690088,0.0011238934,0.0003036246,0.006518261],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.98541623,0.007043972,0.0017195544,0.00065115315,0.004382088,0.0007870081],"domain_scores_gemma":[0.94001013,0.04312061,0.0049325787,0.0020538745,0.007302947,0.0025799368],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012235136,0.00039962298,0.00047369243,0.0040982105,0.0015726882,0.0066195712,0.0012592648,0.0013822042,0.011092312],"category_scores_gemma":[0.065611824,0.0003560625,0.0007044617,0.0039985343,0.0018352738,0.007398331,0.0040750215,0.0015204439,0.0035684127],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.005398295,0.0016805409,0.15637933,0.00417477,0.00018283068,0.001156702,0.15979238,0.0007605859,0.005165635,0.014918759,0.010739353,0.6396509],"study_design_scores_gemma":[0.0009932754,0.009711062,0.47328332,0.006115291,0.00059611246,0.0028855202,0.31892493,0.004769293,0.009900014,0.0068667475,0.16563837,0.00031596675],"about_ca_topic_score_codex":0.0033511382,"about_ca_topic_score_gemma":0.004161975,"teacher_disagreement_score":0.012235136,"about_ca_system_score_codex":0.0030441931,"about_ca_system_score_gemma":0.0026425815,"threshold_uncertainty_score":0.064706326},"labels":[],"label_agreement":null},{"id":"W3197559989","doi":"","title":"An n-gram Approach to Exploiting a Monolingual Corpus for Machine Translation","year":2005,"lang":"en","type":"article","venue":"O2 - Repositori Institucional (Universitat Oberta de Catalunya)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Machine translation; n-gram; Computer science; Natural language processing; Translation (biology); Artificial intelligence; Linguistics; Language model","score_opus":0.018868904411325948,"score_gpt":0.2686632120982062,"score_spread":0.24979430768688027,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3197559989","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0063637835,0.0006644302,0.9843651,0.00031960377,0.0001581256,0.00012663216,0.00046428692,0.0028129574,0.004725001],"genre_scores_gemma":[0.09819503,0.0010609645,0.8881561,0.0003827669,0.00027993816,0.00048553658,0.0023160826,0.0010688296,0.008054792],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9981681,0.00096412655,0.00013520622,0.000363109,0.00030020255,0.000069279144],"domain_scores_gemma":[0.99743724,0.0011913818,0.00019587006,0.0006062095,0.0005177106,0.000051612788],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001819368,0.0014288871,0.0009902603,0.0022136096,0.0016252788,0.0017916254,0.00088334794,0.00087964605,0.0052433563],"category_scores_gemma":[0.0066095483,0.0005764002,0.00086682674,0.0030856451,0.00080489,0.004534315,0.0022210144,0.0015795932,0.00570827],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006787689,0.0002018722,0.0020452177,0.0011014492,0.00030898832,0.00085335167,0.0017884152,0.016374445,0.08245716,0.09152866,0.015942274,0.7867194],"study_design_scores_gemma":[0.0000976744,0.00043266744,0.0026582135,0.00034740486,0.00035935643,0.0016580967,0.0011041071,0.4463812,0.07337281,0.2770282,0.19629934,0.00026096756],"about_ca_topic_score_codex":0.0025863955,"about_ca_topic_score_gemma":0.0072157225,"teacher_disagreement_score":0.0052433563,"about_ca_system_score_codex":0.00066325534,"about_ca_system_score_gemma":0.0018980091,"threshold_uncertainty_score":0.017540812},"labels":[],"label_agreement":null},{"id":"W3197766881","doi":"10.25189/2675-4916.2021.v2.n3.id399","title":"Quantifying the Differences Between Lexical Categories","year":2021,"lang":"en","type":"article","venue":"Cadernos de Linguística","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Humber Polytechnic","funders":"","keywords":"Linguistics; Categorization; Determinative; Computer science; Natural language processing; Grammar; Pronoun; Psychology; Artificial intelligence; Philosophy","score_opus":0.060885525250815246,"score_gpt":0.3226802791093785,"score_spread":0.26179475385856327,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3197766881","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.78685236,0.0011220896,0.16633339,0.0006776954,0.00015488743,0.00016629693,0.0018114428,0.00026631713,0.042615406],"genre_scores_gemma":[0.9692206,0.000121169775,0.028392749,0.00009689607,0.000033122054,0.00012214656,0.00095388247,0.00010894464,0.0009504979],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.991926,0.0030388916,0.0007705859,0.0018065483,0.0020654246,0.00039256836],"domain_scores_gemma":[0.97052705,0.021761158,0.001475672,0.0032499526,0.0025178571,0.00046828092],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005720358,0.00040248004,0.00056594145,0.004387066,0.0008876417,0.0046370546,0.0009347955,0.0009688105,0.0046408237],"category_scores_gemma":[0.054872267,0.00028906716,0.00049828197,0.0038825485,0.0029801442,0.0070060845,0.0038586482,0.0013618045,0.0010299849],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013653592,0.00025110054,0.16374187,0.00069966074,0.0005014411,0.0002901968,0.017862827,0.0047090454,0.030539427,0.33985883,0.0047543454,0.43542585],"study_design_scores_gemma":[0.000061081446,0.0004774294,0.22135252,0.00019343804,0.00016526598,0.0007510697,0.015864372,0.02465451,0.009301377,0.703847,0.023137957,0.00019399394],"about_ca_topic_score_codex":0.0017538229,"about_ca_topic_score_gemma":0.0018269967,"teacher_disagreement_score":0.005720358,"about_ca_system_score_codex":0.0010282936,"about_ca_system_score_gemma":0.00066481787,"threshold_uncertainty_score":0.030252516},"labels":[],"label_agreement":null},{"id":"W3198097706","doi":"","title":"Like Chalk and Cheese? On the Effects of Translationese in MT Training","year":2021,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Translation (biology); Training (meteorology); Computer science; Training set; Machine translation; Quality (philosophy); Artificial intelligence; Matching (statistics); Natural language processing; Machine learning; Mathematics; Statistics; Geography","score_opus":0.014306988910404403,"score_gpt":0.24971451105224896,"score_spread":0.23540752214184454,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3198097706","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8955435,0.014124,0.060398225,0.0046038213,0.0006435993,0.00015338945,0.0013816576,0.0023676541,0.020784093],"genre_scores_gemma":[0.96339095,0.0009082777,0.028682511,0.0007756888,0.00021470615,0.000055863657,0.0023582017,0.00047669635,0.0031371508],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9902829,0.0070871636,0.0004771365,0.0011549836,0.0007474764,0.00025033884],"domain_scores_gemma":[0.926897,0.062310763,0.0020108123,0.0047284,0.003121085,0.0009319554],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01477402,0.0011882122,0.00072934414,0.00096183404,0.0015505353,0.0021782937,0.00077618205,0.0017974371,0.0028709755],"category_scores_gemma":[0.06323505,0.00038660775,0.00046863733,0.0011330027,0.0017221787,0.0034333577,0.0022013572,0.0025994417,0.0015428425],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0064236084,0.0010948476,0.0966262,0.0011571402,0.0009899919,0.0006217054,0.0020312956,0.16048306,0.029457338,0.00583178,0.020593734,0.67468923],"study_design_scores_gemma":[0.00057420245,0.004750503,0.09067219,0.0008032775,0.0008749265,0.0011694074,0.0021915436,0.761952,0.08859598,0.022099659,0.02606781,0.00024855701],"about_ca_topic_score_codex":0.009754391,"about_ca_topic_score_gemma":0.021013701,"teacher_disagreement_score":0.01477402,"about_ca_system_score_codex":0.0010631821,"about_ca_system_score_gemma":0.0009897004,"threshold_uncertainty_score":0.078133464},"labels":[],"label_agreement":null},{"id":"W3198384647","doi":"10.52098/acj.202141","title":"Cloud computing architecture for Tagging Arabic Text Using Hybrid Model","year":2021,"lang":"en","type":"article","venue":"Applied computing Journal","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Memorial University of Newfoundland","funders":"","keywords":"Computer science; Cloud computing; Natural language processing; Artificial intelligence; Syntax; Semantics (computer science); Arabic; Architecture; Information extraction; The Internet; Information retrieval; Speech recognition; World Wide Web; Programming language; Linguistics; Operating system","score_opus":0.019392701252485614,"score_gpt":0.27867519193989343,"score_spread":0.25928249068740783,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3198384647","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11377426,0.0011310497,0.8491025,0.0012266064,0.00047482565,0.0005752664,0.0011291169,0.020459259,0.012127161],"genre_scores_gemma":[0.75707215,0.00065110443,0.22690257,0.00057049625,0.00008764777,0.000412707,0.0022165987,0.00029780963,0.011788877],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994523,0.00008069024,0.00005585956,0.00016453164,0.00012901226,0.00011762741],"domain_scores_gemma":[0.99947554,0.00007164418,0.000037827715,0.00009720526,0.0002604058,0.000057358815],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005754207,0.00073318114,0.0006623817,0.0009787476,0.001310222,0.0016839555,0.0018409524,0.00076826295,0.0027447909],"category_scores_gemma":[0.0009869246,0.0002671044,0.00076235924,0.0015072203,0.00032833562,0.0022890423,0.0010406808,0.000522596,0.0019229935],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0035495146,0.0011794852,0.017009709,0.0004904491,0.00045074694,0.0019612575,0.00086405355,0.23521769,0.05643082,0.03452459,0.059262425,0.58905923],"study_design_scores_gemma":[0.000035422854,0.000091045236,0.0007647678,0.000015998416,0.000048541435,0.00015197326,0.00010103646,0.97571814,0.009945155,0.0046338444,0.008452644,0.000041404863],"about_ca_topic_score_codex":0.02233642,"about_ca_topic_score_gemma":0.01831664,"teacher_disagreement_score":0.02233642,"about_ca_system_score_codex":0.0014533389,"about_ca_system_score_gemma":0.002083295,"threshold_uncertainty_score":0.04441285},"labels":[],"label_agreement":null},{"id":"W3198457396","doi":"10.18653/v1/2021.emnlp-main.259","title":"Learning from Multiple Noisy Augmented Data Sets for Better Cross-Lingual Spoken Language Understanding","year":2021,"lang":"en","type":"article","venue":"Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Benchmark (surveying); Focus (optics); Noise (video); Code (set theory); Artificial intelligence; Training set; Machine learning; Spoken language; Resource (disambiguation); Noise reduction; Natural language processing; Speech recognition; Image (mathematics); Set (abstract data type)","score_opus":0.14898561212540867,"score_gpt":0.4560963868956432,"score_spread":0.3071107747702345,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3198457396","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.061603796,0.0016030018,0.9151431,0.0007920721,0.00040109036,0.00019486608,0.0022261783,0.014572403,0.0034634336],"genre_scores_gemma":[0.35176852,0.00066493405,0.62317735,0.0006669614,0.00019140079,0.0005908122,0.01607111,0.001329834,0.005539086],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99556446,0.0018552758,0.00026199626,0.0015002948,0.00058381853,0.00023411977],"domain_scores_gemma":[0.991971,0.0037860852,0.00030542005,0.0025725116,0.0011967443,0.00016817024],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0042676004,0.0027015463,0.0020762486,0.0017303984,0.0010560991,0.0031226226,0.0027068355,0.0026442823,0.004371716],"category_scores_gemma":[0.0153661985,0.001053637,0.0024899526,0.0018555411,0.0014065579,0.0058330297,0.006328544,0.005297908,0.005076002],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00059673615,0.0007786389,0.005311682,0.0007123644,0.0005924959,0.0005149361,0.0017321609,0.10535949,0.04347101,0.0058473935,0.019664919,0.8154182],"study_design_scores_gemma":[0.000066357614,0.00029503735,0.0023280468,0.00013104409,0.00012695693,0.0002544672,0.00097438646,0.9354633,0.025449885,0.019628841,0.015158095,0.00012359682],"about_ca_topic_score_codex":0.005940411,"about_ca_topic_score_gemma":0.010348893,"teacher_disagreement_score":0.005940411,"about_ca_system_score_codex":0.00073327584,"about_ca_system_score_gemma":0.0018832154,"threshold_uncertainty_score":0.022569537},"labels":[],"label_agreement":null},{"id":"W3198711991","doi":"10.1162/tacl_a_00478","title":"It’s not Rocket Science: Interpreting Figurative Language in Narratives","year":2022,"lang":"en","type":"article","venue":"Transactions of the Association for Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":35,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Literal and figurative language; Computer science; Natural language processing; Principle of compositionality; Narrative; Generative grammar; Interpretation (philosophy); Artificial intelligence; Linguistics; Context (archaeology); Expression (computer science); Programming language; History","score_opus":0.012093882901237876,"score_gpt":0.3034295858987213,"score_spread":0.29133570299748346,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3198711991","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7994957,0.0023724863,0.16230823,0.0040067197,0.00028059634,0.00034729514,0.0073010004,0.004466206,0.019421685],"genre_scores_gemma":[0.9297994,0.00037170088,0.06238274,0.0002950877,0.00003712642,0.00006407498,0.0051535526,0.00015754184,0.0017388762],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9988996,0.00070530886,0.00004643949,0.00022707337,0.00007719911,0.000044423818],"domain_scores_gemma":[0.9926323,0.005790607,0.00044184583,0.0007236254,0.0002859941,0.00012561097],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019888207,0.0009794856,0.00023556306,0.0010159275,0.000521685,0.0026538495,0.0010459985,0.0013094777,0.0035422123],"category_scores_gemma":[0.015115927,0.0003454076,0.00072251644,0.0006158512,0.001275472,0.0048800167,0.0010582352,0.0014322174,0.0013064706],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017627584,0.0006217144,0.09854395,0.003207957,0.00043235757,0.0043288446,0.051030107,0.15198609,0.036406465,0.06904928,0.045773596,0.53685683],"study_design_scores_gemma":[0.000100528654,0.0002337553,0.021364786,0.0007934897,0.00013292822,0.002029757,0.011199162,0.7836898,0.029771745,0.06796018,0.08251569,0.00020806614],"about_ca_topic_score_codex":0.0049745715,"about_ca_topic_score_gemma":0.008166929,"teacher_disagreement_score":0.0049745715,"about_ca_system_score_codex":0.0011422249,"about_ca_system_score_gemma":0.0006394879,"threshold_uncertainty_score":0.01184988},"labels":[],"label_agreement":null},{"id":"W3198775993","doi":"10.1007/978-3-030-86159-9_12","title":"Contextualized Knowledge Base Sense Embeddings in Word Sense Disambiguation","year":2021,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Computer science; Knowledge base; Embedding; Natural language processing; Artificial intelligence; Ambiguity; Representation (politics); Word (group theory); Base (topology); Knowledge representation and reasoning; Task (project management); Linguistics; Mathematics; Programming language","score_opus":0.01915145585384681,"score_gpt":0.2885088080800382,"score_spread":0.2693573522261914,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3198775993","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02350989,0.0038200212,0.96360564,0.0003506174,0.0004648143,0.00009607196,0.0013031733,0.0033512062,0.0034985822],"genre_scores_gemma":[0.2725435,0.0023221122,0.7127912,0.00026875298,0.00022630428,0.00019889268,0.0055736857,0.00090457767,0.0051710405],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9985127,0.0004727767,0.00019426481,0.00042152376,0.0003033475,0.00009542823],"domain_scores_gemma":[0.9977794,0.0011231742,0.000099496945,0.0004583844,0.0004723883,0.00006724273],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014882499,0.00083619787,0.0012496446,0.0029836134,0.00086057506,0.0022800765,0.0015512953,0.0012485903,0.0055565005],"category_scores_gemma":[0.00575673,0.0008943468,0.0008360796,0.004748838,0.00083295366,0.0070503,0.0030031975,0.0016222702,0.0033081404],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00042882873,0.00022166259,0.00114069,0.00055712636,0.0001887235,0.00025062056,0.00042669819,0.03218051,0.010678342,0.05119678,0.0190555,0.8836745],"study_design_scores_gemma":[0.000051559833,0.0001421449,0.0015229217,0.00027780342,0.00018051811,0.0005152909,0.0006167442,0.67595726,0.021051591,0.26442748,0.03516296,0.0000936378],"about_ca_topic_score_codex":0.0026450017,"about_ca_topic_score_gemma":0.0054483833,"teacher_disagreement_score":0.0055565005,"about_ca_system_score_codex":0.0005160718,"about_ca_system_score_gemma":0.0010420395,"threshold_uncertainty_score":0.018588305},"labels":[],"label_agreement":null},{"id":"W3200285169","doi":"10.18653/v1/2021.emnlp-main.779","title":"PICARD: Parsing Incrementally for Constrained Auto-Regressive Decoding from Language Models","year":2021,"lang":"en","type":"preprint","venue":"Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; University of Toronto","funders":"","keywords":"Computer science; Decoding methods; Parsing; Language model; Natural language processing; Artificial intelligence; Code (set theory); Programming language; Compiler; Algorithm; Set (abstract data type)","score_opus":0.07064381815737095,"score_gpt":0.41447000363355646,"score_spread":0.3438261854761855,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3200285169","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007871318,0.00033334832,0.89914805,0.00026699548,0.00014303892,0.00013670104,0.0011475304,0.08873478,0.0022182432],"genre_scores_gemma":[0.14289214,0.0003146261,0.827105,0.0006312913,0.0001262331,0.000502458,0.00692154,0.0158948,0.005611944],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99823964,0.00059366913,0.000115124356,0.0005737827,0.00034004205,0.00013773405],"domain_scores_gemma":[0.9947523,0.0035426621,0.00014328993,0.0009580164,0.00049850816,0.00010524105],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019269593,0.0030578387,0.0013845079,0.0012316915,0.0007987178,0.0026646508,0.0031267204,0.0022084592,0.014550982],"category_scores_gemma":[0.015300361,0.0019355732,0.0020474,0.0010493022,0.0012563666,0.003214001,0.003295782,0.004347242,0.008778794],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00066918816,0.00022550125,0.0023981088,0.00082505273,0.00045979206,0.0006857183,0.000682593,0.24307163,0.028873613,0.027164657,0.07495188,0.6199923],"study_design_scores_gemma":[0.00005583789,0.00004607507,0.00017682115,0.000037222526,0.00004491465,0.00013528508,0.000056083274,0.9584664,0.013253034,0.01973841,0.0079432735,0.000046698016],"about_ca_topic_score_codex":0.009828807,"about_ca_topic_score_gemma":0.026092893,"teacher_disagreement_score":0.014550982,"about_ca_system_score_codex":0.0011609967,"about_ca_system_score_gemma":0.0035152729,"threshold_uncertainty_score":0.04867786},"labels":[],"label_agreement":null},{"id":"W3200657555","doi":"10.1007/s42803-022-00062-7","title":"CKMorph: a comprehensive morphological analyzer for Central Kurdish","year":2023,"lang":"en","type":"article","venue":"International Journal of Digital Humanities","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Computer science; Morpheme; Adjective; Natural language processing; Spectrum analyzer; Artificial intelligence; Adverb; Set (abstract data type); Context (archaeology); Noun; Verb; Word (group theory); Part of speech; Preprocessor; Lexicon; Test set; Linguistics; Programming language","score_opus":0.05056650297410732,"score_gpt":0.3146939083321877,"score_spread":0.2641274053580804,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3200657555","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.21086356,0.0019261076,0.39775333,0.00041166088,0.00046755024,0.00068689784,0.07055904,0.2641836,0.053148285],"genre_scores_gemma":[0.3573615,0.0010064907,0.5000878,0.00034760372,0.00013125605,0.00044113965,0.0816581,0.03228394,0.02668226],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99955755,0.000041424715,0.000086177584,0.000118616634,0.00012993405,0.00006633799],"domain_scores_gemma":[0.9993056,0.0001433622,0.000079709265,0.00013634808,0.00028423857,0.000050771447],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00044354433,0.0016335734,0.0006218226,0.004273941,0.00088688673,0.0017921304,0.0007823104,0.0006128253,0.03555769],"category_scores_gemma":[0.0013519627,0.0007348323,0.00053970126,0.0023777776,0.0003934032,0.0030478812,0.001995359,0.00062913407,0.024770107],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011160187,0.00014084039,0.009102424,0.0016478352,0.00013496354,0.002128612,0.0013360856,0.0010558956,0.15570906,0.006404013,0.09905323,0.722171],"study_design_scores_gemma":[0.0002619128,0.0003241723,0.076071724,0.00043635757,0.00035536604,0.008913867,0.0042659277,0.03087897,0.1919991,0.015682109,0.67042685,0.00038367984],"about_ca_topic_score_codex":0.0014775632,"about_ca_topic_score_gemma":0.0030926676,"teacher_disagreement_score":0.03555769,"about_ca_system_score_codex":0.0003462843,"about_ca_system_score_gemma":0.0010853383,"threshold_uncertainty_score":0.118952334},"labels":[],"label_agreement":null},{"id":"W3201262565","doi":"10.1075/li.00058.ior","title":"Minimodel of semantic synthesis of Russian sentences","year":2021,"lang":"en","type":"article","venue":"Lingvisticae Investigationes","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Lexicalization; Lexicographical order; Computer science; Representation (politics); Natural language processing; Artificial intelligence; Matching (statistics); Linguistics; Mathematics; Philosophy","score_opus":0.01747218787358303,"score_gpt":0.2576618983789042,"score_spread":0.24018971050532115,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3201262565","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0879826,0.0002479312,0.8890057,0.00059922406,0.000056579327,0.00014855109,0.0008882162,0.00096207927,0.02010911],"genre_scores_gemma":[0.749392,0.00015213026,0.23834465,0.00010861057,0.000044294815,0.0002662425,0.0015353663,0.00021322122,0.009943544],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994087,0.00022771425,0.000041070245,0.00016245099,0.000100761594,0.00005926689],"domain_scores_gemma":[0.99964154,0.00017076399,0.00003455627,0.0000601277,0.00007472573,0.000018340499],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007072476,0.00044695445,0.000458172,0.00070794043,0.00059894903,0.0017056575,0.0007853474,0.0008588278,0.005584404],"category_scores_gemma":[0.0016700688,0.0002624354,0.0011954033,0.00041152647,0.000979424,0.0018924781,0.0009279422,0.00067157764,0.0008039997],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007604223,0.00004082057,0.00040018684,0.00014089525,0.000031356478,0.00022593208,0.00053499715,0.08900186,0.0073438403,0.8786515,0.0009377386,0.022614809],"study_design_scores_gemma":[0.000031608368,0.000059079015,0.00019885958,0.00003110524,0.000028484083,0.00007681943,0.00018280259,0.46500483,0.004519181,0.51993066,0.009919508,0.00001710303],"about_ca_topic_score_codex":0.0026418704,"about_ca_topic_score_gemma":0.0033512723,"teacher_disagreement_score":0.005584404,"about_ca_system_score_codex":0.0015335432,"about_ca_system_score_gemma":0.0008555007,"threshold_uncertainty_score":0.018681645},"labels":[],"label_agreement":null},{"id":"W3201505982","doi":"10.1007/978-981-16-2380-6_51","title":"On Profiling Space Reduction Efficiency in Vector Space Modeling-Based Natural Language Processing","year":2021,"lang":"en","type":"book-chapter","venue":"Lecture notes in networks and systems","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Trois-Rivières; Université du Québec à Montréal","funders":"","keywords":"Automatic summarization; Computer science; Space (punctuation); Reduction (mathematics); Salient; Profiling (computer programming); Artificial intelligence; Natural language processing; Data mining; Algorithm; Mathematics; Programming language","score_opus":0.010596772815700352,"score_gpt":0.24570913121493496,"score_spread":0.2351123583992346,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3201505982","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.040442023,0.0015900425,0.9449462,0.0006464392,0.0001305266,0.000113175935,0.0004942681,0.006656076,0.0049811737],"genre_scores_gemma":[0.4603081,0.0009233841,0.5257942,0.00047112667,0.0001446402,0.00024852122,0.0024339003,0.0014662577,0.00820984],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99602807,0.0016344525,0.00022803176,0.0005286271,0.0011387065,0.00044224822],"domain_scores_gemma":[0.9895243,0.0068286383,0.00028957112,0.002053836,0.0011622958,0.00014138286],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002194123,0.0012317754,0.0014678898,0.0012632699,0.00090711523,0.0028910516,0.0022872232,0.00087721244,0.006172732],"category_scores_gemma":[0.014774854,0.00063683337,0.0011535216,0.0029522367,0.000863338,0.0064174365,0.002152868,0.0024232375,0.00253528],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013610241,0.0005526494,0.0029740487,0.0004424261,0.00013394626,0.00013056594,0.00047502026,0.16805685,0.015521977,0.06884794,0.022140892,0.71936274],"study_design_scores_gemma":[0.000025192761,0.00005907837,0.00032104267,0.000020946292,0.000031608808,0.000041531686,0.000082641156,0.9591308,0.006944604,0.031032166,0.0022950242,0.000015317313],"about_ca_topic_score_codex":0.009839305,"about_ca_topic_score_gemma":0.009843691,"teacher_disagreement_score":0.009839305,"about_ca_system_score_codex":0.0013239818,"about_ca_system_score_gemma":0.001919244,"threshold_uncertainty_score":0.02064991},"labels":[],"label_agreement":null},{"id":"W3202143479","doi":"10.1016/j.procs.2021.08.123","title":"Inferring the Number and Order of Embedded Topics Across Documents","year":2021,"lang":"en","type":"article","venue":"Procedia Computer Science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Order (exchange); Information retrieval; Data science; Theoretical computer science","score_opus":0.01020790319764575,"score_gpt":0.3080445696120024,"score_spread":0.29783666641435663,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3202143479","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.44844505,0.0013651852,0.54277575,0.00048097462,0.00007462788,0.00021370352,0.0023845136,0.0015530442,0.0027070285],"genre_scores_gemma":[0.67168355,0.00061745476,0.32282704,0.00003858257,0.00010594843,0.00011229394,0.0029239708,0.00015131987,0.0015399337],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986663,0.00024351894,0.00013790886,0.00057145394,0.00025094565,0.00012990301],"domain_scores_gemma":[0.99131596,0.0053822864,0.00087210373,0.0008043151,0.0013838332,0.00024141949],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015503723,0.0006956477,0.0007274318,0.0058546383,0.001017456,0.0022464017,0.000686525,0.0009908803,0.0009468817],"category_scores_gemma":[0.011244746,0.00061545573,0.00066514267,0.0034694492,0.00058956695,0.0027015551,0.0010699191,0.0013413853,0.0008337227],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011330977,0.0004740901,0.12080553,0.00069709664,0.00033092685,0.00050103554,0.0032145195,0.08622718,0.055698406,0.018787248,0.0075985026,0.7045324],"study_design_scores_gemma":[0.0000690894,0.00012563176,0.052783553,0.00010694807,0.00019347979,0.00045775864,0.0017464539,0.8503007,0.03222925,0.053249948,0.008633458,0.00010368046],"about_ca_topic_score_codex":0.0061705913,"about_ca_topic_score_gemma":0.009291489,"teacher_disagreement_score":0.0061705913,"about_ca_system_score_codex":0.00086311175,"about_ca_system_score_gemma":0.0016178355,"threshold_uncertainty_score":0.012269378},"labels":[],"label_agreement":null},{"id":"W3202882906","doi":"","title":"METIS-II: the German to English MT system","year":2007,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Metis; German; History; Linguistics; Computer science; Archaeology; World Wide Web; Philosophy","score_opus":0.007916549751680718,"score_gpt":0.2734316183348972,"score_spread":0.2655150685832165,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3202882906","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.181896,0.0036079865,0.26366305,0.003589601,0.0011616617,0.0018330732,0.14268555,0.23184012,0.16972291],"genre_scores_gemma":[0.46988368,0.00089011865,0.268657,0.0009280222,0.00033238504,0.0009676458,0.19409297,0.0080057215,0.056242477],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99947923,0.00013631287,0.00005786428,0.00013598293,0.00013060827,0.000059979695],"domain_scores_gemma":[0.9997036,0.000060191327,0.00003005475,0.000067320696,0.00009663038,0.000042186755],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00058605697,0.0010812912,0.0008001823,0.0011736718,0.0007107134,0.0014270296,0.00089254975,0.0006990846,0.016932113],"category_scores_gemma":[0.0013232045,0.00044653963,0.00028745446,0.0011845941,0.0003836398,0.0015986337,0.0013329225,0.0006769503,0.012060488],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016464033,0.00027011338,0.0027235383,0.0016203001,0.0001950553,0.001792052,0.0010094405,0.007509775,0.11256778,0.03530688,0.41190392,0.42345467],"study_design_scores_gemma":[0.0010713601,0.00071417977,0.010811843,0.00022895036,0.0002381,0.002469678,0.0008439204,0.08913183,0.12045879,0.020610377,0.7532255,0.0001954523],"about_ca_topic_score_codex":0.0039250003,"about_ca_topic_score_gemma":0.0059518875,"teacher_disagreement_score":0.016932113,"about_ca_system_score_codex":0.0007517035,"about_ca_system_score_gemma":0.0009027851,"threshold_uncertainty_score":0.056643546},"labels":[],"label_agreement":null},{"id":"W32031805","doi":"10.1007/978-3-319-06483-3_14","title":"A Consensus Approach for Annotation Projection in an Advanced Dialog Context","year":2014,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Nuance Communications (Canada); Université de Montréal","funders":"","keywords":"Computer science; Annotation; Dialog box; Projection (relational algebra); Exploit; Context (archaeology); Universalization; Artificial intelligence; Process (computing); Natural language processing; World Wide Web; Programming language; Algorithm","score_opus":0.021150735817669256,"score_gpt":0.2822775895014908,"score_spread":0.26112685368382155,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W32031805","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0016928045,0.000069895585,0.9952099,0.00014518075,0.000051679664,0.00010885738,0.00008935579,0.00069394027,0.0019383823],"genre_scores_gemma":[0.09760063,0.00015079971,0.8944666,0.00015891694,0.000098046046,0.00049323007,0.0007180998,0.0004750092,0.005838642],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9896187,0.003910987,0.0008746827,0.0025758173,0.0024319342,0.00058788137],"domain_scores_gemma":[0.9870863,0.005456393,0.00035115596,0.0026364918,0.0039060141,0.00056357717],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008826677,0.0013919054,0.0019696231,0.0029514788,0.004162404,0.004916458,0.004951713,0.0035384912,0.011521936],"category_scores_gemma":[0.020569434,0.0017888398,0.0025940533,0.0032053005,0.0029165403,0.010311521,0.011062401,0.004338166,0.00419674],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00057253276,0.00029271396,0.00095676223,0.0006600274,0.00018677747,0.0006645421,0.00432598,0.045269907,0.015498547,0.530236,0.015795583,0.38554063],"study_design_scores_gemma":[0.000052300613,0.0001260628,0.00028658658,0.00015288393,0.000128228,0.00023391696,0.0010983687,0.4906816,0.013932586,0.47113496,0.022053884,0.000118637676],"about_ca_topic_score_codex":0.0064705363,"about_ca_topic_score_gemma":0.007467748,"teacher_disagreement_score":0.011521936,"about_ca_system_score_codex":0.0016033706,"about_ca_system_score_gemma":0.0047435784,"threshold_uncertainty_score":0.04668045},"labels":[],"label_agreement":null},{"id":"W3205231356","doi":"10.18653/v1/2022.findings-acl.135","title":"Morphosyntactic Tagging with Pre-trained Language Models for Arabic and its Dialects","year":2022,"lang":"en","type":"article","venue":"Findings of the Association for Computational Linguistics: ACL 2022","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"York University; New York University Abu Dhabi","keywords":"Transformer; Arabic; Computer science; Modern Standard Arabic; Natural language processing; Artificial intelligence; Language model; Training set; Resource (disambiguation); Linguistics; Engineering; Voltage","score_opus":0.008924129542520672,"score_gpt":0.25229433331955636,"score_spread":0.2433702037770357,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3205231356","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.42444018,0.004453862,0.4374241,0.0011931247,0.0014773089,0.00043995076,0.015982319,0.091139264,0.02344993],"genre_scores_gemma":[0.6342845,0.0012259,0.3018605,0.0006494557,0.0001262914,0.0002666144,0.046810754,0.003675458,0.0111005455],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99890196,0.000272595,0.00008983855,0.00051294436,0.00012330967,0.00009931322],"domain_scores_gemma":[0.9966467,0.0016506803,0.00011110572,0.00076817663,0.0007044837,0.00011883266],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015900575,0.0024461974,0.0008345883,0.0015592136,0.001009802,0.0023421014,0.0018375908,0.0012216264,0.006922849],"category_scores_gemma":[0.006235406,0.0007230107,0.0012335207,0.0012926352,0.00064310094,0.0035100887,0.0019128477,0.002592636,0.009834396],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010802882,0.00053900864,0.01312763,0.0010274561,0.000710432,0.00091757823,0.0014145594,0.10359788,0.057953253,0.003359026,0.039569054,0.77670383],"study_design_scores_gemma":[0.00015914497,0.00032363157,0.0084121395,0.0002231477,0.00045676986,0.0008174787,0.0012915544,0.8382786,0.09793703,0.010123746,0.041729093,0.0002476256],"about_ca_topic_score_codex":0.014101338,"about_ca_topic_score_gemma":0.021663705,"teacher_disagreement_score":0.014101338,"about_ca_system_score_codex":0.0012892233,"about_ca_system_score_gemma":0.0013878378,"threshold_uncertainty_score":0.028038502},"labels":[],"label_agreement":null},{"id":"W320549274","doi":"","title":"Word Blocks Help Students Learn to Write Sentences","year":2000,"lang":"en","type":"article","venue":"ETC.: A Review of General Semantics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Sentence; Adjective; Linguistics; Grammar; Computer science; Subject (documents); Verb; Object (grammar); Artificial intelligence; Reading (process); Natural language processing; Noun; Philosophy; World Wide Web","score_opus":0.012833592178711425,"score_gpt":0.31397084101839084,"score_spread":0.3011372488396794,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W320549274","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.527376,0.019313606,0.036935426,0.03975328,0.00883207,0.00294586,0.0031788417,0.01589609,0.3457688],"genre_scores_gemma":[0.43393826,0.017210497,0.09573968,0.019118281,0.0010054258,0.003532614,0.0053465,0.0019179272,0.42219082],"study_design_codex":"not_applicable","study_design_gemma":"observational","domain_scores_codex":[0.99965084,0.00008621358,0.00002935032,0.00006299603,0.0001273305,0.000043371423],"domain_scores_gemma":[0.99880826,0.00042907428,0.00011603472,0.00008108675,0.00025803794,0.00030745682],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009717978,0.0011598134,0.0006404687,0.00047372788,0.0007382671,0.0013508186,0.0009304216,0.0010729971,0.058065522],"category_scores_gemma":[0.0035057883,0.0003546777,0.0004257748,0.0002634746,0.0005694917,0.0018623113,0.0014815687,0.0023436039,0.023872133],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004112212,0.003692243,0.003327617,0.0011257344,0.000025581357,0.0008582945,0.0050666435,0.00024245527,0.011070757,0.00487715,0.48893735,0.4803649],"study_design_scores_gemma":[0.0005399195,0.003888112,0.015367078,0.0011164849,0.00008608085,0.0013046359,0.0058712712,0.000898217,0.016083892,0.023745697,0.93097895,0.00011971127],"about_ca_topic_score_codex":0.0008139163,"about_ca_topic_score_gemma":0.002178163,"teacher_disagreement_score":0.058065522,"about_ca_system_score_codex":0.00037474418,"about_ca_system_score_gemma":0.0007752728,"threshold_uncertainty_score":0.1942485},"labels":[],"label_agreement":null},{"id":"W3206689743","doi":"10.1609/aaai.v36i10.21323","title":"Non-autoregressive Translation with Layer-Wise Prediction and Deep Supervision","year":2022,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"DeepMind; Compute Canada; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Machine translation; Autoregressive model; Inference; Computer science; Transformer; Translation (biology); Artificial intelligence; BLEU; Machine learning; Artificial neural network; Econometrics; Mathematics; Engineering; Voltage","score_opus":0.0423277031369836,"score_gpt":0.27712261504520214,"score_spread":0.23479491190821855,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3206689743","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.028847024,0.000791526,0.9558993,0.0006366577,0.00016729532,0.0000441655,0.00041541594,0.010020923,0.0031776077],"genre_scores_gemma":[0.59602934,0.00069856405,0.3899154,0.0006215579,0.0001546311,0.00012685747,0.0022250288,0.0008105575,0.009418105],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999355,0.00020821848,0.00004157683,0.00019840493,0.000121128774,0.00007557603],"domain_scores_gemma":[0.99877983,0.0005356051,0.000081240796,0.00033267296,0.00023023505,0.000040421208],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014790496,0.0012805478,0.00094479905,0.00046928035,0.00036132854,0.0010189813,0.0016574621,0.0009317367,0.0037028277],"category_scores_gemma":[0.0040768213,0.0006318384,0.00086679915,0.00093192945,0.0005970473,0.0029029024,0.0008473796,0.0024492305,0.0030955211],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005713195,0.00024493333,0.0020303228,0.000279758,0.00027123804,0.00023672223,0.00014310445,0.4186108,0.022998648,0.015538066,0.016393077,0.522682],"study_design_scores_gemma":[0.000021228372,0.00003695814,0.00014628774,0.000008317203,0.000029976689,0.00004081513,0.000008987125,0.9868786,0.0057700523,0.0058199954,0.0012300942,0.0000087162625],"about_ca_topic_score_codex":0.008760117,"about_ca_topic_score_gemma":0.015741238,"teacher_disagreement_score":0.008760117,"about_ca_system_score_codex":0.000643138,"about_ca_system_score_gemma":0.0016774439,"threshold_uncertainty_score":0.017418265},"labels":[],"label_agreement":null},{"id":"W3207306342","doi":"10.18653/v1/2022.acl-long.448","title":"Compositional Generalization in Dependency Parsing","year":2022,"lang":"en","type":"article","venue":"Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Institute for Advanced Research; McGill University; Mila - Quebec Artificial Intelligence Institute","funders":"Natural Sciences and Engineering Research Council of Canada; Mitacs; McGill University; Canadian Institute for Advanced Research","keywords":"Parsing; Computer science; Dependency grammar; Dependency (UML); Artificial intelligence; Natural language processing; Generalization; Test set; Set (abstract data type); Divergence (linguistics); Programming language; Mathematics","score_opus":0.0065491978464662405,"score_gpt":0.2383371770636452,"score_spread":0.23178797921717897,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3207306342","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2563904,0.0014826918,0.69352025,0.0019087134,0.00031635826,0.00056018354,0.008253627,0.021208746,0.016359018],"genre_scores_gemma":[0.682481,0.00045602437,0.2886207,0.0010030952,0.00015565765,0.0005918663,0.020310378,0.0027488326,0.0036324838],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9920821,0.0035125345,0.00046462682,0.0022350592,0.0012697644,0.00043578443],"domain_scores_gemma":[0.9743824,0.016193036,0.0009189369,0.0054644393,0.0026505697,0.00039063088],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009917457,0.0014536594,0.0013149306,0.002710076,0.00212796,0.0025523177,0.001996306,0.0020630262,0.004985412],"category_scores_gemma":[0.03498561,0.0008361242,0.0023066623,0.0028874509,0.0021822741,0.006937485,0.0045903698,0.0038165129,0.0022466916],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012198571,0.00046239045,0.03833993,0.0011220325,0.0005583718,0.0011166764,0.0029464378,0.17938484,0.02497991,0.07404023,0.06070839,0.6151209],"study_design_scores_gemma":[0.00012850981,0.00025984042,0.012724795,0.00011546889,0.00017274704,0.0009485262,0.0005897931,0.7279741,0.021654382,0.20870222,0.026570063,0.00015953378],"about_ca_topic_score_codex":0.00637715,"about_ca_topic_score_gemma":0.0115612,"teacher_disagreement_score":0.009917457,"about_ca_system_score_codex":0.0022251988,"about_ca_system_score_gemma":0.0030500859,"threshold_uncertainty_score":0.052449107},"labels":[],"label_agreement":null},{"id":"W3208011254","doi":"","title":"Bilingual Methods for Adaptive Training Data Selection for Machine Translation","year":2016,"lang":"en","type":"article","venue":"Conference of the Association for Machine Translation in the Americas","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":31,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Machine translation; Computer science; Artificial intelligence; Selection (genetic algorithm); Convolutional neural network; Translation (biology); Sentence; Natural language processing; Machine learning; Task (project management); Adaptation (eye); Artificial neural network; BLEU; Domain adaptation; Speech recognition","score_opus":0.16171809590177152,"score_gpt":0.41267138165388,"score_spread":0.2509532857521085,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3208011254","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011768917,0.0002639098,0.98501426,0.00008550853,0.00005313985,0.00007884419,0.0001335867,0.0014774331,0.0011244424],"genre_scores_gemma":[0.228935,0.00029896077,0.7624675,0.00035816833,0.00015346357,0.00070517074,0.002405896,0.00067133066,0.0040044934],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99787855,0.000964703,0.0001751418,0.0004955498,0.00039104786,0.000095112395],"domain_scores_gemma":[0.99777067,0.0009444737,0.0001613732,0.0005348318,0.0005115939,0.000077030796],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028221447,0.0009930979,0.00084656564,0.0011924356,0.0007422733,0.0005675161,0.0014254983,0.00068512204,0.0037649632],"category_scores_gemma":[0.005065753,0.00049872097,0.0007340288,0.0013790198,0.0005079124,0.001598856,0.0017652696,0.0011009449,0.0019353044],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00066771306,0.0003731507,0.0031654998,0.0003269176,0.00020751153,0.00019923768,0.00034366187,0.09207065,0.04484657,0.014462554,0.0074822875,0.8358543],"study_design_scores_gemma":[0.00010276428,0.00015598723,0.0010881448,0.000024866016,0.00004665242,0.00015074876,0.000080007034,0.94958454,0.025147047,0.013132958,0.010453126,0.00003317232],"about_ca_topic_score_codex":0.0018776844,"about_ca_topic_score_gemma":0.004704964,"teacher_disagreement_score":0.0037649632,"about_ca_system_score_codex":0.0005802523,"about_ca_system_score_gemma":0.0011536843,"threshold_uncertainty_score":0.014925122},"labels":[],"label_agreement":null},{"id":"W3208412037","doi":"","title":"Demonstration of the Spanish to English METIS-II MT system.","year":2007,"lang":"en","type":"article","venue":"IEEE Transactions on Medical Imaging","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Metis; Computer science; Natural language processing; World Wide Web","score_opus":0.006598225678916645,"score_gpt":0.2595688154234082,"score_spread":0.25297058974449155,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3208412037","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2720355,0.0017534512,0.29513952,0.006580535,0.0033789554,0.0021158042,0.0318404,0.16340241,0.22375342],"genre_scores_gemma":[0.6605634,0.0005019038,0.23854005,0.0014934554,0.00041682727,0.00093123846,0.021143598,0.0040313476,0.07237818],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99962175,0.000121957746,0.000037520786,0.00007431277,0.000097882425,0.000046641435],"domain_scores_gemma":[0.998765,0.00032448917,0.000029461637,0.00017514781,0.00055716827,0.00014877522],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010853166,0.0006070944,0.00045902954,0.00042437718,0.00059884053,0.0010684875,0.00088677567,0.0008543835,0.03488551],"category_scores_gemma":[0.0033179955,0.00020625742,0.00018713695,0.0004933296,0.0002696971,0.0011118911,0.0010799993,0.0006660368,0.021075714],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002984219,0.0008108873,0.006232163,0.0010605871,0.0001401099,0.0051932563,0.0023417687,0.0036719374,0.14020446,0.008635702,0.35665178,0.4720733],"study_design_scores_gemma":[0.0019982485,0.0014410163,0.022464251,0.00034892437,0.00018935057,0.008197734,0.0036947036,0.14437315,0.1927291,0.013558653,0.6107065,0.00029828952],"about_ca_topic_score_codex":0.0099388305,"about_ca_topic_score_gemma":0.011105866,"teacher_disagreement_score":0.03488551,"about_ca_system_score_codex":0.00032774862,"about_ca_system_score_gemma":0.0009345382,"threshold_uncertainty_score":0.11670363},"labels":[],"label_agreement":null},{"id":"W3209416046","doi":"10.1080/15434303.2021.1992629","title":"The Relationship between Word Difficulty and Frequency: A Response to Hashimoto (2021)","year":2021,"lang":"en","type":"article","venue":"Language Assessment Quarterly","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Word lists by frequency; Correlation; Vocabulary; Rank (graph theory); Word (group theory); Linguistics; Psychology; Range (aeronautics); Mathematics; Statistics; Sentence; Philosophy","score_opus":0.01621099591469831,"score_gpt":0.3205802786497973,"score_spread":0.304369282735099,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3209416046","genre_codex":"commentary","genre_gemma":"commentary","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"commentary","genre_consensus":"commentary","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11678341,0.0051522423,0.016411297,0.8313345,0.014884158,0.00027846915,0.00064881943,0.00025030202,0.014256749],"genre_scores_gemma":[0.4585101,0.0037971628,0.012588124,0.49230295,0.009501652,0.0005226714,0.00041154533,0.00023278738,0.02213305],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9949883,0.0019398951,0.0007243869,0.00072119845,0.0014138268,0.00021239452],"domain_scores_gemma":[0.9269078,0.05599312,0.0021626698,0.0014588896,0.011771689,0.0017058499],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011509527,0.0006895202,0.0005166484,0.001049753,0.0026294484,0.0016023528,0.0009629316,0.007035044,0.005616477],"category_scores_gemma":[0.08191672,0.00045225097,0.00066993636,0.0007401052,0.0018927384,0.0027221683,0.0025178432,0.00870917,0.0024284963],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011831387,0.0005507195,0.07132431,0.00081617065,0.0001333145,0.00290801,0.03837986,0.0010307401,0.013759308,0.03185965,0.62646866,0.21158625],"study_design_scores_gemma":[0.00020635458,0.0021158182,0.10981544,0.0011999169,0.00018328385,0.00880321,0.049387448,0.005274378,0.011625504,0.04394077,0.76671463,0.0007333368],"about_ca_topic_score_codex":0.0048644273,"about_ca_topic_score_gemma":0.0056924964,"teacher_disagreement_score":0.011509527,"about_ca_system_score_codex":0.0024014579,"about_ca_system_score_gemma":0.0014489327,"threshold_uncertainty_score":0.06086892},"labels":[],"label_agreement":null},{"id":"W3209428854","doi":"10.5281/zenodo.5138041","title":"Data Citation in Practice - Cocoon a French repository dedicated to oral resources","year":2021,"lang":"en","type":"article","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Humber Polytechnic","funders":"European Commission","keywords":"Citation; Business; Data science; Computer science; World Wide Web","score_opus":0.04629649423272179,"score_gpt":0.30137806803661754,"score_spread":0.25508157380389573,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3209428854","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"reproducibility","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"reproducibility","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0110220015,0.038111094,0.054859728,0.034068134,0.015290689,0.0010409355,0.35371307,0.0560783,0.43581605],"genre_scores_gemma":[0.10865396,0.03197036,0.13237841,0.004373694,0.0084746275,0.0019502433,0.29431832,0.037884213,0.3799961],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99414694,0.0015318634,0.0006598442,0.0004914397,0.002814652,0.0003552396],"domain_scores_gemma":[0.959749,0.015039264,0.0028130866,0.0073726,0.012150888,0.0028751881],"candidate_categories":["metaresearch","scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.007868346,0.0012441907,0.0014588361,0.027837815,0.0025264542,0.011365775,0.0012770528,0.0015788354,0.1984704],"category_scores_gemma":[0.043850966,0.0005871637,0.00071880064,0.05530957,0.0011956659,0.006549269,0.004461303,0.0014053643,0.05431439],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001282827,0.000027840139,0.0007653666,0.00083208526,0.000020162415,0.00019443901,0.00068168645,0.00031631635,0.000694774,0.015729148,0.8192793,0.16133052],"study_design_scores_gemma":[0.00001643595,0.0000113736405,0.00080678344,0.00044596367,0.000008053114,0.00010040988,0.00022117632,0.00036116567,0.00025102892,0.001847912,0.99590707,0.000022533515],"about_ca_topic_score_codex":0.03770278,"about_ca_topic_score_gemma":0.04209536,"teacher_disagreement_score":0.99213165,"about_ca_system_score_codex":0.0042220703,"about_ca_system_score_gemma":0.010762533,"threshold_uncertainty_score":0.6639496},"labels":[],"label_agreement":null},{"id":"W3210115313","doi":"10.18653/v1/2021.wnut-1.23","title":"Noisy UGC Translation at the Character Level: Revisiting Open-Vocabulary Capabilities and Robustness of Char-Based Models","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Nautical Research Society","funders":"Agence Nationale de la Recherche","keywords":"Computer science; Robustness (evolution); Machine translation; Artificial intelligence; Vocabulary; Natural language processing; Character (mathematics); Translation (biology); Bridging (networking); Machine learning; Linguistics; Mathematics","score_opus":0.07650691870858589,"score_gpt":0.2976728666687269,"score_spread":0.221165947960141,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3210115313","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.23808913,0.0018328317,0.74024266,0.0022925467,0.00034275162,0.00015370233,0.00063128036,0.0052706855,0.011144482],"genre_scores_gemma":[0.92853934,0.00046550386,0.065186545,0.0005022848,0.00014524662,0.00012375535,0.00101923,0.0007301639,0.003287913],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9963676,0.001971397,0.00021630364,0.00076795125,0.0004608505,0.00021594706],"domain_scores_gemma":[0.97847795,0.015232706,0.0008153083,0.003503706,0.0015940113,0.00037624445],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0061743967,0.0013768307,0.0014258669,0.0010143395,0.00092230586,0.0034255143,0.002261016,0.0022841957,0.003490718],"category_scores_gemma":[0.039645,0.0006678066,0.000912555,0.0009204605,0.0022550162,0.0064218286,0.002964218,0.0032879862,0.002399383],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00072854833,0.00021224876,0.0038552908,0.00054027257,0.00023790874,0.00040389018,0.0009449346,0.8095265,0.018476924,0.02723718,0.00348842,0.134348],"study_design_scores_gemma":[0.000011062032,0.00006484594,0.00020292832,0.000028357501,0.000016316255,0.000050959305,0.000050467785,0.9829516,0.003037186,0.0128828045,0.0006872107,0.000016326576],"about_ca_topic_score_codex":0.006744715,"about_ca_topic_score_gemma":0.004539257,"teacher_disagreement_score":0.006744715,"about_ca_system_score_codex":0.0011496821,"about_ca_system_score_gemma":0.0013116709,"threshold_uncertainty_score":0.03265375},"labels":[],"label_agreement":null},{"id":"W3210814509","doi":"","title":"Transferring markup tags in statistical machine translation: a two-stream approach.","year":2013,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Markup language; Computer science; Translation (biology); Natural language processing; Machine translation; Artificial intelligence; World Wide Web; XML; Chemistry","score_opus":0.014304014147523225,"score_gpt":0.25840915631435063,"score_spread":0.2441051421668274,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3210814509","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0014317547,0.00057973963,0.99550015,0.0003685077,0.00011855714,0.0000788938,0.000062075465,0.0008984077,0.000961849],"genre_scores_gemma":[0.04921637,0.0016505992,0.9430946,0.00036425403,0.00029583566,0.0003053546,0.00055899605,0.0006075583,0.0039063944],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9953165,0.0024475546,0.00032075564,0.0006564372,0.0011047727,0.00015399524],"domain_scores_gemma":[0.98900324,0.0063090147,0.00080469245,0.0023408474,0.0013844483,0.00015772092],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0051255547,0.0013627628,0.0009758389,0.0036369937,0.0011733649,0.003919002,0.002604435,0.0023574356,0.0026217028],"category_scores_gemma":[0.020179408,0.0011720458,0.0016187963,0.004826496,0.0027902594,0.0054492075,0.0034528952,0.0035709587,0.004102198],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002611626,0.00028222552,0.0012090775,0.0008331038,0.00019051768,0.00063928455,0.0016247975,0.040381968,0.017446049,0.15771724,0.011304816,0.76810974],"study_design_scores_gemma":[0.00006152678,0.00017186264,0.0007517839,0.00017456201,0.000107250686,0.0008242314,0.00047572306,0.5103115,0.037136473,0.39345637,0.056402646,0.00012615966],"about_ca_topic_score_codex":0.0019050264,"about_ca_topic_score_gemma":0.0023033142,"teacher_disagreement_score":0.0051255547,"about_ca_system_score_codex":0.001063297,"about_ca_system_score_gemma":0.0022249985,"threshold_uncertainty_score":0.027106822},"labels":[],"label_agreement":null},{"id":"W3210898385","doi":"10.5281/zenodo.2536218","title":"Dataset of discussion threads from Meneame","year":2019,"lang":"en","type":"dataset","venue":"Figshare","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science","score_opus":0.0379756133860593,"score_gpt":0.31143559976643054,"score_spread":0.2734599863803712,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3210898385","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0052957865,0.0009716255,0.00047078996,0.0005784479,0.00020545883,0.00014667147,0.9805433,0.0009503745,0.010837488],"genre_scores_gemma":[0.0072610048,0.00017971269,0.0014459001,0.00014884475,0.000083294086,0.00050384033,0.9842131,0.0001179059,0.0060462803],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9982748,0.00060123444,0.00018640107,0.00036669648,0.00041107947,0.00015989103],"domain_scores_gemma":[0.9956026,0.0013923551,0.00053770194,0.00072645827,0.0010440855,0.0006968368],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011337564,0.001275924,0.0008112171,0.004516811,0.0016020043,0.0021675145,0.001708291,0.0018495438,0.041461192],"category_scores_gemma":[0.008278058,0.00038827612,0.0009271539,0.0049613197,0.00035028646,0.0017841349,0.0022436162,0.0013951787,0.04278691],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028885843,0.00012424534,0.00647044,0.00094995904,0.00006467407,0.00007676692,0.0005553139,0.00027408724,0.0003500177,0.0016955663,0.97560287,0.013547254],"study_design_scores_gemma":[0.00012348208,0.00003615203,0.018933099,0.00026888118,0.000024697028,0.00006154901,0.0005613775,0.0007310108,0.00047022055,0.0011842371,0.97757316,0.000032141772],"about_ca_topic_score_codex":0.02099059,"about_ca_topic_score_gemma":0.06157746,"teacher_disagreement_score":0.041461192,"about_ca_system_score_codex":0.0019431906,"about_ca_system_score_gemma":0.0020946804,"threshold_uncertainty_score":0.13870156},"labels":[],"label_agreement":null},{"id":"W3211384195","doi":"10.18653/v1/2021.emnlp-main.130","title":"Translation-based Supervision for Policy Generation in Simultaneous Neural Machine Translation","year":2021,"lang":"en","type":"article","venue":"Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Machine translation; Artificial intelligence; Translation (biology); Security token; Oracle; Reinforcement learning; Heuristic; Action (physics); Inference; Machine learning; Quality (philosophy); Sentence; Natural language processing; Programming language","score_opus":0.07715591203049524,"score_gpt":0.42118249622465037,"score_spread":0.3440265841941551,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3211384195","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04783933,0.00020762181,0.94709104,0.00028231554,0.000059693713,0.00010716379,0.00006522608,0.0025220471,0.0018255002],"genre_scores_gemma":[0.80670124,0.00007836626,0.19095181,0.00022003044,0.000049032595,0.00026739034,0.00020674504,0.0002248858,0.001300484],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99821824,0.00085870316,0.00010274641,0.00043896635,0.0002599096,0.00012151597],"domain_scores_gemma":[0.99009657,0.0068315254,0.00071970036,0.0012139612,0.00089199736,0.0002462267],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034074537,0.0008492752,0.0010377216,0.00041941274,0.00065733626,0.00066397985,0.0015025609,0.0011441164,0.002449974],"category_scores_gemma":[0.01604421,0.00066008314,0.0004260174,0.0004715534,0.0015695475,0.0020224757,0.0013770038,0.002343032,0.0007340631],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006749986,0.00046356436,0.0021647818,0.00025738354,0.000069693924,0.00019243754,0.00045655272,0.6885932,0.011897061,0.016713846,0.0029887408,0.2755277],"study_design_scores_gemma":[0.000027363596,0.00006393868,0.00010757601,0.0000084568055,0.0000062672125,0.00002236248,0.0000133869835,0.98891544,0.0028190145,0.00761265,0.00039633433,0.000007221902],"about_ca_topic_score_codex":0.002956854,"about_ca_topic_score_gemma":0.004953816,"teacher_disagreement_score":0.0034074537,"about_ca_system_score_codex":0.0010145685,"about_ca_system_score_gemma":0.0022827033,"threshold_uncertainty_score":0.01802051},"labels":[],"label_agreement":null},{"id":"W3211887606","doi":"10.26615/978-954-452-072-4_057","title":"Semi-Supervised and Unsupervised Sense Annotation via Translations","year":2021,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Machine Intelligence Institute","keywords":"Annotation; Computer science; Artificial intelligence; Natural language processing; Unsupervised learning","score_opus":0.012009234891178774,"score_gpt":0.2500861603837959,"score_spread":0.2380769254926171,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3211887606","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.051436737,0.0007294609,0.9183975,0.00042865868,0.00031088362,0.0004436467,0.0058149467,0.013230712,0.009207372],"genre_scores_gemma":[0.20705423,0.0004890072,0.7570994,0.0002880333,0.000120207456,0.0007882642,0.026951598,0.002277299,0.0049320105],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98921204,0.0042967135,0.0011922362,0.003015351,0.002005706,0.00027798032],"domain_scores_gemma":[0.97650903,0.010080491,0.0016644078,0.006783621,0.0046196836,0.00034270543],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004310771,0.0020819332,0.0016155089,0.0052692485,0.0017040345,0.0030097682,0.0023910496,0.0012052694,0.004924562],"category_scores_gemma":[0.020486755,0.0010487675,0.0013744595,0.005433585,0.0023446765,0.0058112643,0.0054163425,0.0024331121,0.0058216704],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00076560315,0.0006002672,0.006634807,0.0022812525,0.00044138942,0.0009512885,0.0031757331,0.024710072,0.08694502,0.031495247,0.040788323,0.801211],"study_design_scores_gemma":[0.00028124131,0.00050281675,0.009499656,0.00062884146,0.0003034967,0.0018064998,0.003750353,0.49506932,0.20468663,0.16439897,0.11865286,0.0004193406],"about_ca_topic_score_codex":0.0017730397,"about_ca_topic_score_gemma":0.004579254,"teacher_disagreement_score":0.0052692485,"about_ca_system_score_codex":0.0007385024,"about_ca_system_score_gemma":0.0026514349,"threshold_uncertainty_score":0.022797823},"labels":[],"label_agreement":null},{"id":"W3212319665","doi":"10.5539/elt.v14n12p23","title":"“Please Let me Use Google Translate”: Thai EFL Students’ Behavior and Attitudes toward Google Translate Use in English Writing","year":2021,"lang":"en","type":"article","venue":"English Language Teaching","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"Kasetsart University","keywords":"Paragraph; Psychology; Syntax; Sentence; Mathematics education; Grammar; Quality (philosophy); Linguistics; Computer science; World Wide Web; Natural language processing","score_opus":0.021588056162319454,"score_gpt":0.31324746139870774,"score_spread":0.2916594052363883,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3212319665","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99847287,0.000057118657,0.00008563543,0.00014556329,0.0000059766553,0.00001065206,0.000015931091,0.0000068363943,0.001199466],"genre_scores_gemma":[0.99735117,0.00019466346,0.00019759625,0.00022426363,0.000005145601,0.00002034203,0.00003718223,0.000009253324,0.0019604873],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9985896,0.00048794207,0.00011986924,0.00012322399,0.00044533983,0.00023403525],"domain_scores_gemma":[0.9940832,0.0015283122,0.00208715,0.0001593976,0.000963519,0.0011783775],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014691708,0.00037623005,0.00030796084,0.0007878931,0.0015004905,0.0033454292,0.0004105199,0.0007775427,0.0029379618],"category_scores_gemma":[0.006142821,0.0003105815,0.00034000285,0.0006568212,0.0013192011,0.0014153328,0.0011758264,0.001051685,0.0010801969],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015910956,0.00065071194,0.53261507,0.00035209826,0.000039366387,0.0031123736,0.4140939,0.0001056586,0.0074628363,0.00023100055,0.0017850416,0.039392788],"study_design_scores_gemma":[0.000016291082,0.0006668978,0.23354517,0.00019094335,0.00005550265,0.0022076734,0.75047845,0.0006572909,0.0022841827,0.0002529451,0.009551696,0.000093008675],"about_ca_topic_score_codex":0.004143199,"about_ca_topic_score_gemma":0.0066914028,"teacher_disagreement_score":0.004143199,"about_ca_system_score_codex":0.0004918311,"about_ca_system_score_gemma":0.00084319455,"threshold_uncertainty_score":0.009828508},"labels":[],"label_agreement":null},{"id":"W3212627999","doi":"10.26615/978-954-452-072-4_017","title":"Cross-Lingual Wolastoqey-English Definition Modelling","year":2021,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of New Brunswick","funders":"","keywords":"Computer science; Natural language processing; Word (group theory); Task (project management); Artificial intelligence; Encoding (memory); Baseline (sea); Language model; Linguistics","score_opus":0.030160438474449235,"score_gpt":0.2840604957220903,"score_spread":0.2539000572476411,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3212627999","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15830055,0.0010864269,0.7825602,0.0006173683,0.00045586663,0.00088856963,0.00852693,0.022432797,0.025131281],"genre_scores_gemma":[0.4898682,0.00044311493,0.46751156,0.00039656862,0.00005061325,0.00061034237,0.02591094,0.003756653,0.011451991],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988587,0.00036385155,0.00012131456,0.00035917177,0.00022236815,0.00007460607],"domain_scores_gemma":[0.9979151,0.00090687483,0.00014247328,0.00038496187,0.0005763541,0.0000741528],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012064328,0.0009415595,0.00058764743,0.0012912012,0.0004302978,0.0013767272,0.0010723415,0.0006685128,0.0070270444],"category_scores_gemma":[0.0046511414,0.00046067088,0.000866807,0.0008717805,0.0004269258,0.002307168,0.0022488933,0.0011399427,0.0032513088],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006529536,0.00049271446,0.016883062,0.0030143429,0.00027264853,0.0023197725,0.0029135284,0.028172748,0.0956575,0.052314945,0.0761215,0.7211844],"study_design_scores_gemma":[0.000156745,0.00048668298,0.01016731,0.00034484375,0.0002117439,0.0038607582,0.0021016817,0.4208737,0.1678918,0.037614744,0.3561055,0.0001844839],"about_ca_topic_score_codex":0.0016163582,"about_ca_topic_score_gemma":0.0039190096,"teacher_disagreement_score":0.99838364,"about_ca_system_score_codex":0.0005304044,"about_ca_system_score_gemma":0.0007772756,"threshold_uncertainty_score":0.023507774},"labels":[],"label_agreement":null},{"id":"W3212964142","doi":"10.26615/978-954-452-072-4_183","title":"AutoChart: A Dataset for Chart-to-Text Generation Task","year":2021,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Saskatchewan","funders":"","keywords":"Computer science; Task (project management); Chart; Data science; Natural language processing; Artificial intelligence; Information retrieval; Engineering; Systems engineering","score_opus":0.03007901274681364,"score_gpt":0.3094329610311797,"score_spread":0.27935394828436605,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3212964142","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.035671204,0.0008923009,0.015538352,0.00080581807,0.00038829766,0.0008525888,0.9206771,0.018533202,0.0066410704],"genre_scores_gemma":[0.026899727,0.0002735464,0.03361417,0.00017063171,0.00006121262,0.0009671994,0.93518096,0.00041972875,0.0024128603],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9988518,0.0002507719,0.00018041919,0.00030804815,0.00033203128,0.00007700175],"domain_scores_gemma":[0.99495715,0.0022733398,0.0004077184,0.00083323725,0.0011855254,0.00034306655],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009907736,0.0012701463,0.00045502587,0.0036565026,0.00088918064,0.0010422271,0.0017198905,0.0020228878,0.008472155],"category_scores_gemma":[0.007467921,0.00025190413,0.00088314473,0.0031995284,0.0004067993,0.001415729,0.001101684,0.0013367548,0.006661089],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00064441864,0.0007358025,0.008416928,0.0024308325,0.00008566344,0.0007625341,0.00047428007,0.005715386,0.007576597,0.0042455764,0.8668365,0.10207543],"study_design_scores_gemma":[0.00082503713,0.00053041754,0.030944427,0.00044464844,0.000098288205,0.0009157524,0.0010967378,0.06078416,0.02083743,0.010573229,0.8727242,0.00022572713],"about_ca_topic_score_codex":0.009721459,"about_ca_topic_score_gemma":0.022689551,"teacher_disagreement_score":0.009721459,"about_ca_system_score_codex":0.0012255365,"about_ca_system_score_gemma":0.0019847427,"threshold_uncertainty_score":0.028342128},"labels":[],"label_agreement":null},{"id":"W3213216381","doi":"10.18653/v1/2021.findings-emnlp.275","title":"Sometimes We Want Ungrammatical Translations","year":2021,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; Mila - Quebec Artificial Intelligence Institute","funders":"McGill University","keywords":"Robustness (evolution); Computer science; Machine translation; Spelling; Word order; Artificial intelligence; Natural language processing; Linguistics","score_opus":0.015520709270369304,"score_gpt":0.27652168494178825,"score_spread":0.26100097567141894,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3213216381","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.41265348,0.003185519,0.43301624,0.026829602,0.0024247998,0.0002973146,0.0065557756,0.021758819,0.09327844],"genre_scores_gemma":[0.82927334,0.0012187556,0.12860705,0.0084957145,0.00045436068,0.0002206314,0.002888366,0.0056427773,0.02319901],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9972875,0.0008313911,0.00026430134,0.0006485423,0.0008193013,0.0001490531],"domain_scores_gemma":[0.9807945,0.011349508,0.0018438987,0.003994046,0.001718642,0.00029940522],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021373488,0.0010485158,0.0008224561,0.00069811026,0.0011195318,0.0021258339,0.00067082857,0.0020919393,0.018557576],"category_scores_gemma":[0.027707467,0.00054063427,0.00045039607,0.0012646947,0.0020310963,0.0039993036,0.0019778279,0.0022939339,0.009890552],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0036221824,0.00027645644,0.014031102,0.003917316,0.00045318424,0.012637545,0.011164925,0.009873222,0.33834112,0.13927457,0.10380915,0.36259916],"study_design_scores_gemma":[0.00023312963,0.00050183316,0.0125319455,0.00059734815,0.00024135993,0.018238386,0.007460199,0.037221808,0.2791534,0.33720693,0.3062786,0.0003350848],"about_ca_topic_score_codex":0.00026080263,"about_ca_topic_score_gemma":0.0005632237,"teacher_disagreement_score":0.018557576,"about_ca_system_score_codex":0.00038970026,"about_ca_system_score_gemma":0.00042005634,"threshold_uncertainty_score":0.062081277},"labels":[],"label_agreement":null},{"id":"W3213416405","doi":"10.5121/ijnlc.2021.10504","title":"Built to Scale: A Corpus-Based Analysis of Adjective Scales in the Mcgill Pain Questionnaire","year":2021,"lang":"en","type":"article","venue":"International Journal on Natural Language Computing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Adjective; Adjective check list; Psychology; McGill Pain Questionnaire; Scale (ratio); Linguistics; Natural language processing; Visual analogue scale; Medicine; Social psychology; Computer science; Physical therapy; Noun; Cartography; Philosophy; Geography","score_opus":0.009581752518038654,"score_gpt":0.31488394594207353,"score_spread":0.30530219342403486,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3213416405","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6436012,0.0027397126,0.19553728,0.0020527483,0.0014802528,0.0043801093,0.104679376,0.006955875,0.038573463],"genre_scores_gemma":[0.59858054,0.00086379953,0.2748244,0.0005051845,0.00019778253,0.0039545116,0.111985624,0.0014235453,0.0076646945],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99682754,0.0015620821,0.00025740912,0.00051626394,0.0007385441,0.00009815452],"domain_scores_gemma":[0.98598427,0.010738065,0.00057496235,0.0008854148,0.0016454003,0.0001719157],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003195557,0.00070960517,0.00041715073,0.00453955,0.001246077,0.0015731787,0.00087595545,0.0005279716,0.0049496526],"category_scores_gemma":[0.021091914,0.00029791897,0.0006634782,0.004950043,0.0009147544,0.001370751,0.0018847092,0.001139325,0.0020001354],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009996757,0.000953864,0.08261113,0.0049449224,0.00038760228,0.0029644014,0.04574548,0.0054083974,0.0375496,0.025848234,0.17324151,0.6193452],"study_design_scores_gemma":[0.00043628574,0.00051729096,0.38388976,0.0009024752,0.0004350314,0.002701853,0.03211575,0.07901113,0.01390164,0.019208647,0.4663983,0.00048187032],"about_ca_topic_score_codex":0.012051009,"about_ca_topic_score_gemma":0.021154897,"teacher_disagreement_score":0.012051009,"about_ca_system_score_codex":0.0011055161,"about_ca_system_score_gemma":0.0017704521,"threshold_uncertainty_score":0.023961723},"labels":[],"label_agreement":null},{"id":"W3213496413","doi":"10.5281/zenodo.3558710","title":"Lynx D2.5 Report on Lynx acquired vocabularies","year":2019,"lang":"en","type":"article","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canarie","funders":"European Commission","keywords":"Geography","score_opus":0.020643288750885967,"score_gpt":0.2534793829124178,"score_spread":0.2328360941615318,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3213496413","genre_codex":"dataset","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03181882,0.0031862406,0.071370326,0.0050516375,0.0013475223,0.0053843423,0.50926477,0.015657715,0.3569186],"genre_scores_gemma":[0.021228392,0.001572454,0.06894659,0.0006057788,0.00013961222,0.0049451725,0.78111374,0.007615569,0.11383269],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9846751,0.003017532,0.0019418406,0.0015365846,0.007930624,0.00089823926],"domain_scores_gemma":[0.9746695,0.004586242,0.001210163,0.003547977,0.014642813,0.0013434245],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.022937154,0.0013349894,0.0011472508,0.008742545,0.0027534112,0.008081397,0.0032397304,0.0014483132,0.08740374],"category_scores_gemma":[0.03785869,0.0013471884,0.0011157108,0.004930007,0.0011835474,0.008625165,0.011487398,0.0027610445,0.07566892],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004090566,0.00022881117,0.00536367,0.0023911018,0.0000708734,0.0004144913,0.006924008,0.0025112468,0.00910333,0.06382819,0.6599404,0.24881484],"study_design_scores_gemma":[0.00003475017,0.00005600683,0.0030334857,0.0005400879,0.00001591396,0.00007965529,0.001135581,0.0005963967,0.0024238324,0.0026189564,0.9894091,0.000056207755],"about_ca_topic_score_codex":0.08209824,"about_ca_topic_score_gemma":0.043532755,"teacher_disagreement_score":0.08740374,"about_ca_system_score_codex":0.006684544,"about_ca_system_score_gemma":0.018663213,"threshold_uncertainty_score":0.29239464},"labels":[],"label_agreement":null},{"id":"W3214201940","doi":"10.26615/978-954-452-072-4_079","title":"Now, It’s Personal : The Need for Personalized Word Sense Disambiguation","year":2021,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of New Brunswick","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"WordNet; Word-sense disambiguation; Computer science; Lemma (botany); Word (group theory); Natural language processing; Artificial intelligence; SemEval; Information retrieval; Linguistics","score_opus":0.0243879449188451,"score_gpt":0.29399558126898234,"score_spread":0.26960763635013724,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3214201940","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.23690972,0.0068265945,0.6777383,0.018790942,0.0023243956,0.00042269053,0.01546526,0.024385426,0.017136633],"genre_scores_gemma":[0.5487866,0.0025034987,0.4194926,0.003767965,0.00095890136,0.00023531438,0.015921425,0.001650387,0.006683261],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99565566,0.0013524052,0.00041820394,0.0018347318,0.0005960139,0.00014294405],"domain_scores_gemma":[0.9779248,0.006281386,0.0010638029,0.011907018,0.0018905504,0.0009324924],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0068421704,0.0010269328,0.0014867081,0.0032890178,0.0018840223,0.004895637,0.0018041511,0.0014511144,0.0021120422],"category_scores_gemma":[0.022515355,0.00079861877,0.0011958327,0.0037780472,0.0015904163,0.019655813,0.0033176988,0.004123491,0.0037830395],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010074133,0.0005384617,0.061579477,0.0007789108,0.00042281154,0.0005155551,0.0030613795,0.026905067,0.021681463,0.021840347,0.07705917,0.7846099],"study_design_scores_gemma":[0.00015922362,0.0003737121,0.0224807,0.00040358678,0.0003033174,0.002166542,0.0047932165,0.30608022,0.033578213,0.32089335,0.30839166,0.00037619498],"about_ca_topic_score_codex":0.0033027425,"about_ca_topic_score_gemma":0.009557002,"teacher_disagreement_score":0.0068421704,"about_ca_system_score_codex":0.001159566,"about_ca_system_score_gemma":0.0018187282,"threshold_uncertainty_score":0.036185265},"labels":[],"label_agreement":null},{"id":"W3217588902","doi":"","title":"Extraction of nominative entities, an opportunity for the cultural sector ?","year":2015,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Humanities; Political science; Art","score_opus":0.1100434543754612,"score_gpt":0.3633509801469527,"score_spread":0.2533075257714915,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3217588902","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19360282,0.006341976,0.6357589,0.013194256,0.00055673014,0.0009199224,0.007519449,0.0032385099,0.1388675],"genre_scores_gemma":[0.50425977,0.0043505835,0.41857338,0.0014913182,0.00013133723,0.00040182885,0.010675738,0.000986953,0.05912909],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9972101,0.001095327,0.00023975859,0.00038402865,0.00092673895,0.00014407298],"domain_scores_gemma":[0.991676,0.00311636,0.0005758945,0.0018500027,0.0026563348,0.00012535119],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005083868,0.0006340004,0.00050794875,0.0036507708,0.0027952858,0.0062298123,0.0015403624,0.0014541986,0.008484457],"category_scores_gemma":[0.014713667,0.00051838055,0.0007428235,0.0067651034,0.002208746,0.009628572,0.002518841,0.0011183822,0.0030476958],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002850439,0.0000796834,0.026276762,0.0024894583,0.00017065935,0.002073008,0.037130807,0.0031779679,0.030599674,0.17095923,0.029240701,0.6975171],"study_design_scores_gemma":[0.000033089353,0.00006394724,0.024588414,0.0009796183,0.00016841268,0.0022753116,0.023078425,0.011719006,0.037781723,0.055737365,0.84341496,0.00015977095],"about_ca_topic_score_codex":0.061705355,"about_ca_topic_score_gemma":0.10240263,"teacher_disagreement_score":0.061705355,"about_ca_system_score_codex":0.003339639,"about_ca_system_score_gemma":0.0053407047,"threshold_uncertainty_score":0.12269235},"labels":[],"label_agreement":null},{"id":"W32688152","doi":"10.1016/j.watres.2020.116164","title":"Determining Word Sense Dominance Using a Thesaurus.","year":2006,"lang":"en","type":"article","venue":"Conference of the European Chapter of the Association for Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":44,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Fundamental Research Funds for the Central Universities; National Natural Science Foundation of China","keywords":"Thesaurus; Natural language processing; Computer science; Word (group theory); Word-sense disambiguation; Artificial intelligence; Dominance (genetics); Similarity (geometry); Information retrieval; Linguistics; WordNet","score_opus":0.029496322816885148,"score_gpt":0.2619333875922081,"score_spread":0.23243706477532294,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W32688152","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.89175063,0.0040892423,0.043779597,0.00051762594,0.00087585195,0.0011001976,0.020995662,0.0018984724,0.034992747],"genre_scores_gemma":[0.89617044,0.001042163,0.08085817,0.00034585764,0.00019811599,0.0010523597,0.01104377,0.00024097803,0.0090481425],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99830127,0.0003211786,0.00031092908,0.0005820124,0.00038099277,0.000103653954],"domain_scores_gemma":[0.99557674,0.0016983097,0.0008108443,0.00024403693,0.0013217855,0.0003482919],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001320191,0.00081513287,0.00055228174,0.0056779706,0.0008243749,0.0018791363,0.00040520643,0.00067460537,0.010046752],"category_scores_gemma":[0.010002251,0.00022510298,0.00077206094,0.00402789,0.0005197334,0.0021120121,0.0012149655,0.00042802928,0.004307393],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002208998,0.00030965483,0.24383745,0.0031579912,0.00072456955,0.0010686979,0.004709458,0.0007442642,0.12941532,0.0021089849,0.014446081,0.5972686],"study_design_scores_gemma":[0.00032850407,0.003224435,0.80999047,0.0009482104,0.0014750422,0.007850568,0.015082538,0.031499077,0.040886175,0.013135272,0.07511059,0.00046922854],"about_ca_topic_score_codex":0.0032323631,"about_ca_topic_score_gemma":0.0061230045,"teacher_disagreement_score":0.010046752,"about_ca_system_score_codex":0.00039381898,"about_ca_system_score_gemma":0.0007971484,"threshold_uncertainty_score":0.033609748},"labels":[],"label_agreement":null},{"id":"W332101547","doi":"","title":"Verb-Noun Compounds in Chinese","year":2003,"lang":"en","type":"article","venue":"Southwest journal of linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Adjunct; Verb; Linguistics; Noun; Philosophy; Mathematics; Combinatorics","score_opus":0.01110030578235564,"score_gpt":0.28102065541156973,"score_spread":0.2699203496292141,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W332101547","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.55541503,0.0020213353,0.019732922,0.0021912237,0.00031844238,0.00046372807,0.001804136,0.0007898221,0.4172633],"genre_scores_gemma":[0.980917,0.00053375127,0.004174363,0.00011516632,0.00005758099,0.00007599074,0.00071261235,0.00007930867,0.01333421],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9995401,0.00006864897,0.000037978294,0.00014320124,0.00015927212,0.00005082212],"domain_scores_gemma":[0.9997024,0.0001070472,0.000055710087,0.00004182708,0.00006962433,0.000023447656],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00045726533,0.0007507099,0.00045437305,0.0010420921,0.0023131259,0.0014726025,0.0005222843,0.0004156132,0.01450501],"category_scores_gemma":[0.0008857057,0.0003038221,0.00031074378,0.002677366,0.0016001253,0.0019287493,0.0010631426,0.00065390806,0.0013611246],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002896667,0.000093076196,0.012631417,0.0010916286,0.000051230618,0.005776549,0.012482161,0.0024370563,0.026701568,0.8082408,0.014408928,0.11579593],"study_design_scores_gemma":[0.0003966867,0.00039753653,0.059766755,0.00022573145,0.00023612342,0.0058744946,0.00775108,0.036300708,0.035244297,0.21176824,0.6417923,0.0002460994],"about_ca_topic_score_codex":0.021464491,"about_ca_topic_score_gemma":0.022302879,"teacher_disagreement_score":0.021464491,"about_ca_system_score_codex":0.0039937194,"about_ca_system_score_gemma":0.0030280654,"threshold_uncertainty_score":0.04852408},"labels":[],"label_agreement":null},{"id":"W33523939","doi":"10.1126/sciadv.abc0671","title":"The problems in a Question Answering system in the academic domain","year":2007,"lang":"en","type":"book-chapter","venue":"Science Advances","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"H2020 Excellent Science; Environment Canada","keywords":"Domain (mathematical analysis); Work (physics); Government (linguistics); Question answering; Library science; Political science; Engineering management; Operations research; Computer science; Engineering; Information retrieval; Philosophy; Mechanical engineering; Linguistics; Mathematics","score_opus":0.017432651939020643,"score_gpt":0.3079986525131947,"score_spread":0.29056600057417403,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W33523939","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02014718,0.0026271972,0.9138266,0.032475933,0.00043618798,0.0002769731,0.0012004297,0.004430965,0.02457854],"genre_scores_gemma":[0.24541,0.0019512969,0.7164992,0.0056367205,0.0012107058,0.0006019794,0.003305077,0.0010175705,0.024367439],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9876187,0.0066127363,0.0010802639,0.0023968434,0.001882345,0.00040903472],"domain_scores_gemma":[0.964605,0.028000947,0.000824645,0.0027841327,0.0032824818,0.00050280575],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013995083,0.0008316173,0.0014558453,0.0023635174,0.0033363774,0.008517441,0.0030453568,0.004923008,0.015942447],"category_scores_gemma":[0.055665556,0.00082928373,0.0012921165,0.0041010627,0.004633829,0.02242173,0.006069566,0.0036571005,0.006597818],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00030640827,0.00022941513,0.004318551,0.0009934524,0.00008967946,0.0006586888,0.0039285845,0.021761922,0.002967614,0.46064308,0.07789893,0.4262038],"study_design_scores_gemma":[0.000039262035,0.000045207133,0.0008411169,0.00015955506,0.00004850841,0.00058739,0.0015525777,0.17813084,0.0042981585,0.7036897,0.11055702,0.00005071261],"about_ca_topic_score_codex":0.0038917642,"about_ca_topic_score_gemma":0.0024050772,"teacher_disagreement_score":0.015942447,"about_ca_system_score_codex":0.0026406215,"about_ca_system_score_gemma":0.0025296465,"threshold_uncertainty_score":0.07401395},"labels":[],"label_agreement":null},{"id":"W33781319","doi":"10.1186/s13195-021-00802-x","title":"Word-Order Relaxations & Restrictions within a Dependency Grammar","year":2001,"lang":"en","type":"article","venue":"International Workshop/Conference on Parsing Technologies","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Vetenskapsrådet; Weston Brain Institute","keywords":"Dependency (UML); Computer science; Word order; Dependency grammar; Natural language processing; Link grammar; Word grammar; Grammar; Artificial intelligence; Word (group theory); Head-driven phrase structure grammar; Linguistics; Generative grammar; Emergent grammar; Relational grammar","score_opus":0.04095957313498641,"score_gpt":0.3138369396712023,"score_spread":0.27287736653621586,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W33781319","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15680437,0.0007686942,0.7813206,0.002915509,0.00045721178,0.0002799113,0.0032993362,0.0033043178,0.05085011],"genre_scores_gemma":[0.690768,0.00057190587,0.28008878,0.00047824622,0.00028706336,0.0002724206,0.0064573246,0.0024121609,0.018664021],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9980057,0.0006439458,0.0002571656,0.0006143042,0.00025410045,0.00022476377],"domain_scores_gemma":[0.99317485,0.0030503895,0.00044365934,0.0025126385,0.0006006627,0.00021786682],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021337261,0.00062260515,0.00081355375,0.0011171077,0.0016489681,0.00245223,0.0014111431,0.0011336214,0.012362982],"category_scores_gemma":[0.00857987,0.0011868039,0.0019935719,0.0010930831,0.0026949584,0.0068543814,0.0027411843,0.0027447322,0.0035475348],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002833845,0.00013303311,0.0050343443,0.0003425527,0.00013389545,0.0018211567,0.0026597197,0.017042125,0.01580689,0.849831,0.012294914,0.09461698],"study_design_scores_gemma":[0.00003472189,0.000058754824,0.001965031,0.00007741522,0.00009221665,0.00094344234,0.0005317485,0.04635811,0.0072095096,0.89758897,0.045062862,0.000077283694],"about_ca_topic_score_codex":0.0036444713,"about_ca_topic_score_gemma":0.006166513,"teacher_disagreement_score":0.012362982,"about_ca_system_score_codex":0.0009652637,"about_ca_system_score_gemma":0.0017612224,"threshold_uncertainty_score":0.041358292},"labels":[],"label_agreement":null},{"id":"W343722495","doi":"","title":"575 Tlingit verbs: A study of Tlingit verb paradigms","year":2013,"lang":"en","type":"dissertation","venue":"Americanae (AECID Library)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Linguistics; Verb; Prefix; Computer science; History; Philosophy","score_opus":0.007411211923000111,"score_gpt":0.24619460893938971,"score_spread":0.2387833970163896,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W343722495","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9813131,0.00013017925,0.0022757577,0.0002105262,0.000009603938,0.00010190075,0.000059350954,0.000014049119,0.015885578],"genre_scores_gemma":[0.9914437,0.00023130125,0.0027110425,0.00009875933,0.000007168628,0.00018602228,0.00012019714,0.00006013868,0.005141761],"study_design_codex":"qualitative","study_design_gemma":"not_applicable","domain_scores_codex":[0.99762625,0.0013422715,0.00019089271,0.00027419953,0.00039344112,0.00017304023],"domain_scores_gemma":[0.9902478,0.0069233873,0.0011739575,0.0005005067,0.00083166786,0.0003227269],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0045741205,0.000441667,0.0004187879,0.0019306099,0.0030165566,0.0055541983,0.0009629104,0.0009077148,0.0032249878],"category_scores_gemma":[0.01179067,0.0005731141,0.00022380099,0.0017227739,0.004037493,0.005517222,0.0028269324,0.002761734,0.000597396],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000035930123,0.0001486182,0.016326128,0.00010816081,0.0000023044176,0.0008163636,0.95548326,0.00003857034,0.0034130164,0.012154483,0.00044176052,0.011031378],"study_design_scores_gemma":[0.000012598732,0.00015905123,0.024824321,0.00024134581,0.000005790883,0.0018167278,0.9337185,0.0007617179,0.0017499715,0.0037222267,0.03295065,0.000037195477],"about_ca_topic_score_codex":0.003656541,"about_ca_topic_score_gemma":0.008495532,"teacher_disagreement_score":0.0055541983,"about_ca_system_score_codex":0.0017568392,"about_ca_system_score_gemma":0.0014877305,"threshold_uncertainty_score":0.024190545},"labels":[],"label_agreement":null},{"id":"W35449742","doi":"10.1155/2022/3626726","title":"The Second Language Acquisition of Past Tense Marker in English by L1 Speakers of Chinese LE PASSÉ DANS L'ACQUISITION DE L'ANGLAIS EN TANT QU'UNE DEUXIÈME LANGUE PAR LES LOCUTEUR DU CHINOIS","year":2009,"lang":"en","type":"article","venue":"Canadian social science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"Taif University","keywords":"Linguistics; Past tense; Humanities; Second-language acquisition; Psychology; Philosophy; Verb","score_opus":0.0020777687325513926,"score_gpt":0.21142889899277936,"score_spread":0.20935113026022797,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W35449742","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9752465,0.0004212616,0.0013411191,0.0005203154,0.000049864982,0.000036188154,0.00023546927,0.000033981232,0.022115303],"genre_scores_gemma":[0.9947418,0.0001728029,0.0007243265,0.00008847128,0.0000115958355,0.000023148372,0.00012028444,0.000038537906,0.004078915],"study_design_codex":"qualitative","study_design_gemma":"observational","domain_scores_codex":[0.99930716,0.00022525487,0.0000468255,0.00015810314,0.00016693414,0.00009578299],"domain_scores_gemma":[0.9970822,0.0016005909,0.0003822071,0.00016340255,0.00053373916,0.00023790855],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016208335,0.00038134804,0.0002692529,0.00050560763,0.0009360295,0.0021210567,0.00035229986,0.00039807605,0.0085004615],"category_scores_gemma":[0.0051737283,0.00037300785,0.0001837229,0.00039476334,0.0012164187,0.002106514,0.0008154662,0.0012082977,0.0009874132],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00082095887,0.00031013452,0.22073917,0.0006885306,0.00009540038,0.004425783,0.5762427,0.00022380169,0.07995979,0.016091747,0.004894665,0.09550736],"study_design_scores_gemma":[0.00009191967,0.0005649578,0.7900423,0.00052635104,0.00016755797,0.007200358,0.13958338,0.0030275187,0.014525057,0.004284228,0.039678592,0.00030779152],"about_ca_topic_score_codex":0.026113695,"about_ca_topic_score_gemma":0.028191293,"teacher_disagreement_score":0.026113695,"about_ca_system_score_codex":0.0008094874,"about_ca_system_score_gemma":0.0012017179,"threshold_uncertainty_score":0.051923394},"labels":[],"label_agreement":null},{"id":"W36344173","doi":"10.71781/10330","title":"Désambiguïsation de corpus monolingues par des approches de type Lesk","year":2003,"lang":"fr","type":"dissertation","venue":"Biochemical and Biophysical Research Communications","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Humanities; Philosophy; Art","score_opus":0.10736707571816777,"score_gpt":0.40468188302748553,"score_spread":0.29731480730931775,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W36344173","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.121049516,0.0048188167,0.8028678,0.0038921717,0.003707655,0.00064511766,0.008206313,0.023132686,0.031679895],"genre_scores_gemma":[0.23967762,0.0019058665,0.70554715,0.00067755964,0.0005345249,0.00048521117,0.009772499,0.007753686,0.033645935],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9961694,0.001368174,0.00045275709,0.0009720745,0.0008384847,0.00019901802],"domain_scores_gemma":[0.99112165,0.0037480467,0.0003703425,0.0015779241,0.0030409698,0.0001411147],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031504221,0.0014067079,0.0012896531,0.003199609,0.0023777594,0.0038617656,0.0011117561,0.001228686,0.015540685],"category_scores_gemma":[0.015586765,0.000758429,0.0009526391,0.0023922648,0.0011592441,0.0035424263,0.001912948,0.002280096,0.00888055],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015034408,0.00021472252,0.0043460233,0.001959106,0.00028648716,0.001727864,0.00700871,0.0026902962,0.15435305,0.026676297,0.058010038,0.7412239],"study_design_scores_gemma":[0.0002678598,0.00040624026,0.011824895,0.00066275586,0.0006522916,0.0039519593,0.006493039,0.053488027,0.26284686,0.032144178,0.62682354,0.00043835328],"about_ca_topic_score_codex":0.009013569,"about_ca_topic_score_gemma":0.012528669,"teacher_disagreement_score":0.015540685,"about_ca_system_score_codex":0.0012052286,"about_ca_system_score_gemma":0.0031702996,"threshold_uncertainty_score":0.05198878},"labels":[],"label_agreement":null},{"id":"W36560989","doi":"","title":"Intégration de l'alignement de mots dans le concordancier bilingue TransSearch","year":2009,"lang":"fr","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; Computer Research Institute of Montréal","funders":"","keywords":"Machine translation; Computer science; Sentence; Word (group theory); Bilingual dictionary; Natural language processing; Artificial intelligence; World Wide Web; Linguistics","score_opus":0.016310318578867215,"score_gpt":0.2643899253127555,"score_spread":0.2480796067338883,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W36560989","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15276994,0.0015342338,0.774685,0.0019253627,0.000891426,0.00035656497,0.004613794,0.023636911,0.03958677],"genre_scores_gemma":[0.43801713,0.0005362307,0.50591177,0.0004401185,0.00020896191,0.00030289116,0.0070722057,0.0057068365,0.04180388],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9969709,0.0008471681,0.00022165217,0.00096615154,0.00080478354,0.00018928545],"domain_scores_gemma":[0.9953082,0.0016256326,0.00022935012,0.0008396739,0.0018455833,0.00015157701],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025870649,0.0012047511,0.00090736814,0.0027323738,0.0014224742,0.0034436337,0.001126517,0.0018072915,0.01756615],"category_scores_gemma":[0.008372568,0.00091813545,0.00097312964,0.002284785,0.00096709834,0.0033872584,0.002410748,0.0018485464,0.008417541],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012260699,0.00023761352,0.010311083,0.0011269251,0.00018720621,0.0007624727,0.0038295654,0.008246702,0.1389993,0.03429962,0.023341876,0.7774316],"study_design_scores_gemma":[0.00012781862,0.0006570913,0.02777626,0.00060755544,0.0006661869,0.0015661463,0.0034193986,0.26019245,0.30162552,0.04232669,0.36080146,0.00023331518],"about_ca_topic_score_codex":0.013928427,"about_ca_topic_score_gemma":0.015199251,"teacher_disagreement_score":0.01756615,"about_ca_system_score_codex":0.0011646831,"about_ca_system_score_gemma":0.002001599,"threshold_uncertainty_score":0.058764637},"labels":[],"label_agreement":null},{"id":"W36903255","doi":"10.3390/molecules28052010","title":"Hierarchical Probabilistic Neural Network Language Model.","year":2005,"lang":"en","type":"article","venue":"International Conference on Artificial Intelligence and Statistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":840,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"National Center for Advancing Translational Sciences","keywords":"Computer science; Language model; Artificial intelligence; Cluster analysis; Hierarchy; Probabilistic logic; Embedding; Generalization; Artificial neural network; Hierarchical clustering; Hierarchical database model; Natural language processing; Machine learning; Data mining; Mathematics","score_opus":0.07106443107823439,"score_gpt":0.3506428053863268,"score_spread":0.2795783743080924,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W36903255","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014922515,0.004231634,0.940626,0.0040014205,0.00064633467,0.00036156562,0.010717679,0.0069753216,0.017517649],"genre_scores_gemma":[0.5465831,0.0035725185,0.3872391,0.0016856828,0.00060963206,0.0016515101,0.018091343,0.00097638805,0.03959071],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99882823,0.00043267652,0.000076406555,0.000334791,0.0001770589,0.00015078041],"domain_scores_gemma":[0.996842,0.0023295556,0.00017488145,0.00018273263,0.0003912897,0.00007940917],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019775538,0.0013484361,0.0018157656,0.0017241291,0.0008701258,0.0018760095,0.0025384778,0.0020075992,0.020254672],"category_scores_gemma":[0.00674375,0.00081790675,0.0021098033,0.0017512193,0.00082487345,0.0027780952,0.0015914077,0.0027392064,0.0074202875],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00027318046,0.00019700793,0.003107936,0.00038798232,0.00028541067,0.0002804672,0.00018388186,0.69001555,0.00046890537,0.068959475,0.03389309,0.20194711],"study_design_scores_gemma":[0.000016811375,0.000013660853,0.000137465,0.000016965865,0.00001862465,0.00003164914,0.000013381892,0.96176505,0.000077027595,0.03472379,0.003176377,0.000009245978],"about_ca_topic_score_codex":0.028210966,"about_ca_topic_score_gemma":0.025649654,"teacher_disagreement_score":0.028210966,"about_ca_system_score_codex":0.0016869258,"about_ca_system_score_gemma":0.0021677539,"threshold_uncertainty_score":0.06775862},"labels":[],"label_agreement":null},{"id":"W39682441","doi":"","title":"Document Semantic Annotation for Intelligent Tutoring Systems: A Concept Mapping Approach.","year":2007,"lang":"en","type":"article","venue":"PolyPublie (École Polytechnique de Montréal)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal; Université de Montréal","funders":"","keywords":"Computer science; Parsing; Domain (mathematical analysis); Natural language processing; Artificial intelligence; Knowledge acquisition; Domain knowledge; Scratch; Natural language; Intelligent tutoring system; Semantic annotation; Ontology; Annotation; Information retrieval; Programming language","score_opus":0.01580231317639407,"score_gpt":0.258389131135175,"score_spread":0.2425868179587809,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W39682441","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0059785554,0.00097108044,0.9767952,0.00089347496,0.00016480469,0.00018217928,0.0013167255,0.005326428,0.008371509],"genre_scores_gemma":[0.061860926,0.0007986129,0.92724484,0.00015829834,0.000068709065,0.00021011801,0.0040110447,0.0003647086,0.005282752],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988537,0.0005019915,0.00010145909,0.00020033722,0.00029948735,0.000043054486],"domain_scores_gemma":[0.9978296,0.0010462903,0.00014170307,0.00037615697,0.0005102481,0.00009597349],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001334317,0.0004925726,0.00036247273,0.0030014936,0.0009457691,0.0023234664,0.0013760162,0.0012616833,0.005624136],"category_scores_gemma":[0.006599068,0.00028577365,0.0005486408,0.0029121106,0.00072376104,0.0037392285,0.0013959835,0.0010285724,0.0030567704],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001479699,0.00019040961,0.001299977,0.00083071116,0.00005184557,0.0004662583,0.0012012937,0.009266965,0.01811912,0.12459156,0.048495542,0.7953383],"study_design_scores_gemma":[0.00006808131,0.00013117674,0.0020964015,0.0004127059,0.00014169756,0.001424037,0.0009417482,0.26039597,0.05160773,0.24384889,0.43883938,0.00009228979],"about_ca_topic_score_codex":0.0024315186,"about_ca_topic_score_gemma":0.0029999274,"teacher_disagreement_score":0.005624136,"about_ca_system_score_codex":0.0010224008,"about_ca_system_score_gemma":0.0016122704,"threshold_uncertainty_score":0.018814623},"labels":[],"label_agreement":null},{"id":"W40123913","doi":"","title":"AUTOMATICALLY CONSTRUCTING A LEXICON OF VERB PHRASE IDIOMATIC COMBINATIONS","year":2006,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":88,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Natural language processing; Computer science; Lexicon; Artificial intelligence; Linguistics; Flexibility (engineering); Phrase; Verb; Representation (politics); Class (philosophy); Mathematics","score_opus":0.006476881362674912,"score_gpt":0.24630763970731312,"score_spread":0.2398307583446382,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W40123913","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14031449,0.001084185,0.80123734,0.0008035279,0.00032559191,0.0007888006,0.012514248,0.01991653,0.023015313],"genre_scores_gemma":[0.3037088,0.0005811913,0.66906244,0.00022317459,0.00010599463,0.00056959415,0.021471448,0.0020554173,0.00222196],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99834085,0.00045938868,0.00024891327,0.00052080257,0.0003350993,0.00009488125],"domain_scores_gemma":[0.99627006,0.0021265962,0.00036866014,0.00037114858,0.00075016054,0.00011341402],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001048708,0.0014456718,0.0015827593,0.0063264575,0.0012512065,0.0033602288,0.0014501212,0.00095770176,0.0053560436],"category_scores_gemma":[0.008826998,0.0015288594,0.00095304265,0.004314817,0.0008616865,0.0049140626,0.0022176923,0.0015447381,0.0039739152],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00044309805,0.00037899535,0.017847719,0.0025092827,0.00033972503,0.0026253155,0.003045393,0.006791958,0.12704928,0.076259725,0.04869823,0.7140113],"study_design_scores_gemma":[0.0002827586,0.00040951246,0.02770223,0.0008595316,0.00076172064,0.008857287,0.004391519,0.49646991,0.10278308,0.14369899,0.21329203,0.00049131113],"about_ca_topic_score_codex":0.001863464,"about_ca_topic_score_gemma":0.0042042118,"teacher_disagreement_score":0.0063264575,"about_ca_system_score_codex":0.0010397696,"about_ca_system_score_gemma":0.0015744839,"threshold_uncertainty_score":0.017917752},"labels":[],"label_agreement":null},{"id":"W4200162648","doi":"10.22148/001c.30697","title":"On Organizing a Shared Task for the Digital Humanities – Conclusions and Future Paths","year":2021,"lang":"en","type":"article","venue":"Journal of Cultural Analytics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Narrative; Annotation; Task (project management); Digital humanities; Narratology; Computer science; World Wide Web; Linguistics; Artificial intelligence; Engineering","score_opus":0.020825326980333212,"score_gpt":0.2705781422478529,"score_spread":0.24975281526751972,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4200162648","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02576315,0.03019768,0.4336745,0.43221977,0.0073982603,0.00073236215,0.0012226937,0.0013689435,0.067422576],"genre_scores_gemma":[0.34078637,0.023643587,0.55587536,0.023170728,0.005048407,0.0021117267,0.005197471,0.0019308433,0.04223558],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.97751004,0.01567601,0.00084061903,0.0021217028,0.0023127268,0.0015388505],"domain_scores_gemma":[0.945227,0.025101915,0.0015639646,0.012371295,0.008892772,0.006842928],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03953398,0.001104676,0.0010504088,0.003225555,0.009899044,0.021582976,0.005256612,0.0068789353,0.012815391],"category_scores_gemma":[0.047343023,0.0007119502,0.0015704519,0.005241407,0.01621589,0.06750826,0.024362199,0.008265768,0.0043209875],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018113888,0.00021035095,0.0020047342,0.0010299964,0.000051513627,0.00026435728,0.029872388,0.0015289098,0.0006001709,0.6798361,0.10715435,0.17726593],"study_design_scores_gemma":[0.000024366711,0.000036767768,0.00094101654,0.0011991503,0.000017811737,0.00016185889,0.0481543,0.0036419295,0.000710513,0.6209242,0.324125,0.00006308929],"about_ca_topic_score_codex":0.007048781,"about_ca_topic_score_gemma":0.0069525866,"teacher_disagreement_score":0.03953398,"about_ca_system_score_codex":0.005967786,"about_ca_system_score_gemma":0.009893275,"threshold_uncertainty_score":0.2090782},"labels":[],"label_agreement":null},{"id":"W4200246069","doi":"10.5539/ijel.v12n1p98","title":"Word Stress in Qassimi Arabic: A Constraint-Based Analysis","year":2021,"lang":"en","type":"article","venue":"International Journal of English Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Constraint (computer-aided design); Formalism (music); Word (group theory); Arabic; Mathematics; Linguistics; Stress (linguistics); Natural language processing; Computer science; Arithmetic; Artificial intelligence; Combinatorics; Speech recognition; Philosophy; Literature; Art","score_opus":0.012355418148970022,"score_gpt":0.2957423054716122,"score_spread":0.28338688732264217,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4200246069","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.31731692,0.0009776611,0.63851494,0.0013003665,0.00010868103,0.0002491687,0.0013187964,0.00043301855,0.039780505],"genre_scores_gemma":[0.8211382,0.0004486835,0.17251234,0.00010736154,0.00007755051,0.00012483684,0.0007106222,0.00016760643,0.004712841],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.999582,0.00011328809,0.000048137244,0.000089695466,0.00012762552,0.000039220155],"domain_scores_gemma":[0.9991841,0.00036438546,0.00009027545,0.0000984624,0.00022348482,0.000039278148],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005833116,0.0005465952,0.00038350705,0.0015136935,0.0012881046,0.0022131589,0.0007585652,0.00055874826,0.004598318],"category_scores_gemma":[0.001883617,0.00043060037,0.0010381666,0.0014533645,0.0019223035,0.002076544,0.0012141602,0.0012044837,0.00061761873],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000086312546,0.00005238756,0.0039468496,0.00019083152,0.000040843697,0.00067247974,0.003975215,0.020574642,0.011943623,0.91335666,0.0018145477,0.043345753],"study_design_scores_gemma":[0.00004333287,0.00013186038,0.009042107,0.00016495645,0.00007722091,0.0006514595,0.0025107812,0.27211046,0.009705645,0.6596646,0.045773853,0.00012376012],"about_ca_topic_score_codex":0.010017859,"about_ca_topic_score_gemma":0.008972763,"teacher_disagreement_score":0.010017859,"about_ca_system_score_codex":0.0018369997,"about_ca_system_score_gemma":0.0012056441,"threshold_uncertainty_score":0.019919097},"labels":[],"label_agreement":null},{"id":"W4200492986","doi":"10.22148/001c.30701","title":"Annotation Guidelines for Narrative Levels and Narrative Acts v2","year":2021,"lang":"en","type":"article","venue":"Journal of Cultural Analytics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Narrative; Annotation; Narrative inquiry; Political science; Literature; Computer science; Art; Artificial intelligence","score_opus":0.10205961958090254,"score_gpt":0.3980552257095357,"score_spread":0.2959956061286332,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4200492986","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005142769,0.0010530681,0.72654784,0.0155213205,0.0025378107,0.0046442146,0.008883615,0.008004191,0.2276651],"genre_scores_gemma":[0.04936887,0.0014299694,0.8141453,0.0051533324,0.00072219124,0.011511608,0.013852761,0.0047013112,0.09911459],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.98566264,0.007202274,0.0029017734,0.0011630157,0.0025354289,0.0005348222],"domain_scores_gemma":[0.95225894,0.021457944,0.0019294001,0.005598537,0.018019618,0.0007355388],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012792355,0.0011945462,0.00074016233,0.0057812384,0.004029702,0.005592704,0.0027067608,0.005800948,0.038361214],"category_scores_gemma":[0.049913265,0.0015165666,0.0007305204,0.0035229095,0.0043164934,0.0069509437,0.00421565,0.004929483,0.030307073],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016190807,0.00012585476,0.00076663686,0.0018295061,0.00001001812,0.0006363688,0.019965285,0.001008014,0.008099186,0.44189382,0.36646584,0.15903758],"study_design_scores_gemma":[0.000016931028,0.000029462022,0.00074603996,0.0012737734,0.0000067399474,0.00034948587,0.002091797,0.0018438516,0.0029708422,0.048270095,0.94233906,0.000061977385],"about_ca_topic_score_codex":0.009141242,"about_ca_topic_score_gemma":0.011295958,"teacher_disagreement_score":0.038361214,"about_ca_system_score_codex":0.0029548486,"about_ca_system_score_gemma":0.0044499743,"threshold_uncertainty_score":0.12833107},"labels":[],"label_agreement":null},{"id":"W4200572262","doi":"10.22148/001c.30703","title":"Annotation Guideline No. 7 (revised): Guidelines for annotation of narrative structure","year":2021,"lang":"en","type":"article","venue":"Journal of Cultural Analytics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Annotation; Narrative; Focalization; Framing (construction); Computer science; Linguistics; Key (lock); Perspective (graphical); Literature; Artificial intelligence; History; Philosophy; Art","score_opus":0.04336450762914995,"score_gpt":0.37492497431710475,"score_spread":0.3315604666879548,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4200572262","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004532201,0.0025722806,0.6334045,0.0211079,0.009134285,0.033135828,0.09670415,0.029170591,0.17023832],"genre_scores_gemma":[0.012080667,0.0023993456,0.7513723,0.00796793,0.0009961447,0.05873864,0.08823609,0.015095636,0.06311328],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9286116,0.03181873,0.022043202,0.0047203028,0.010458792,0.0023474933],"domain_scores_gemma":[0.7323054,0.08538932,0.009830862,0.04437568,0.12510617,0.0029925532],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.07117966,0.0018674689,0.0021784625,0.013144298,0.0061040213,0.009724089,0.008404501,0.008584215,0.07239889],"category_scores_gemma":[0.2243035,0.0035351606,0.002658565,0.010655702,0.004861898,0.007146131,0.009451032,0.007473012,0.09584729],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021882898,0.0001222349,0.00066123053,0.0065622125,0.00006369599,0.000546558,0.018439336,0.00048029659,0.012238333,0.029931812,0.8340214,0.09671408],"study_design_scores_gemma":[0.00004730147,0.00002491562,0.0010094706,0.0042370027,0.000041142113,0.00028839477,0.001662949,0.00047025777,0.003629008,0.009756838,0.9787243,0.000108394095],"about_ca_topic_score_codex":0.02104102,"about_ca_topic_score_gemma":0.02976321,"teacher_disagreement_score":0.07239889,"about_ca_system_score_codex":0.0056313304,"about_ca_system_score_gemma":0.019372506,"threshold_uncertainty_score":0.3764385},"labels":[],"label_agreement":null},{"id":"W4200633747","doi":"10.1609/aaai.v36i11.21443","title":"Word Embeddings via Causal Inference: Gender Bias Reducing and Semantic Information Preserving","year":2022,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Economic and Social Research Council; Social Sciences and Humanities Research Council of Canada","keywords":"Debiasing; Computer science; Word (group theory); Natural language processing; Artificial intelligence; Inference; Causal inference; Focus (optics); Semantic similarity; Oracle; Cognitive psychology; Psychology; Linguistics; Cognitive science; Mathematics","score_opus":0.07089615856085248,"score_gpt":0.31148861091298646,"score_spread":0.240592452352134,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4200633747","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04328872,0.0006518725,0.94940025,0.00061040616,0.0001801016,0.00009056526,0.00052123494,0.0021997783,0.0030571201],"genre_scores_gemma":[0.6219087,0.00071133627,0.36627105,0.0007162453,0.00029241445,0.00016870664,0.0022046524,0.0006335439,0.0070933895],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9984603,0.00058718503,0.000111424946,0.0003930107,0.00032120713,0.00012691629],"domain_scores_gemma":[0.9963238,0.0014092149,0.00042531887,0.0010866342,0.00066656014,0.000088497975],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023987372,0.0010435778,0.0006696698,0.001566325,0.0006683016,0.0010608202,0.00102704,0.001010506,0.0034934643],"category_scores_gemma":[0.013208948,0.000388932,0.0007963364,0.0014642007,0.0010102951,0.0032306106,0.0020461455,0.0014775039,0.0015554952],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000363797,0.00024090064,0.006903694,0.00024204933,0.00012772657,0.00031826692,0.00083765405,0.02650423,0.014445042,0.0515242,0.013285193,0.8852072],"study_design_scores_gemma":[0.000112167756,0.00020238834,0.003672195,0.00012030538,0.00017737372,0.00059983414,0.0007237704,0.68460155,0.028269662,0.26115048,0.020266624,0.0001037386],"about_ca_topic_score_codex":0.0020673552,"about_ca_topic_score_gemma":0.003997539,"teacher_disagreement_score":0.0034934643,"about_ca_system_score_codex":0.00055817154,"about_ca_system_score_gemma":0.0014477754,"threshold_uncertainty_score":0.012685895},"labels":[],"label_agreement":null},{"id":"W42051811","doi":"","title":"Waterloo at NTCIR-3: Using Self-supervised Word Segmentation","year":2002,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Artificial intelligence; Text segmentation; Natural language processing; Segmentation; Word (group theory); Task (project management); Character (mathematics); Machine translation; Information retrieval","score_opus":0.02440267527869397,"score_gpt":0.2597840563686933,"score_spread":0.23538138108999931,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W42051811","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13361977,0.0033847305,0.50615275,0.0019724432,0.0016258998,0.0039398368,0.044471245,0.2349445,0.069888875],"genre_scores_gemma":[0.26132718,0.00074535195,0.5640775,0.0008956865,0.00029872922,0.0020632544,0.12641367,0.00633345,0.03784518],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99618995,0.0016687916,0.00027575542,0.00075499044,0.0008585224,0.00025204886],"domain_scores_gemma":[0.9964805,0.0009977145,0.00016229447,0.0006550174,0.0014539653,0.00025047665],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035740123,0.0016261355,0.0014453564,0.0033899606,0.0016511611,0.0020295687,0.0014125475,0.0011693053,0.011723632],"category_scores_gemma":[0.0060129417,0.00066568103,0.00068709016,0.001751547,0.0008244959,0.0036031015,0.0018421629,0.000823083,0.0102799],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013509522,0.000694398,0.0057549067,0.0018546105,0.00033088218,0.0007393436,0.0016057135,0.0066153277,0.122299634,0.0045584682,0.30255044,0.5516453],"study_design_scores_gemma":[0.0013406558,0.001146727,0.01968316,0.0002503001,0.00046782577,0.0013954989,0.0022563152,0.30807838,0.29349825,0.011023159,0.36026725,0.00059245864],"about_ca_topic_score_codex":0.044881508,"about_ca_topic_score_gemma":0.055138938,"teacher_disagreement_score":0.044881508,"about_ca_system_score_codex":0.001193986,"about_ca_system_score_gemma":0.002636195,"threshold_uncertainty_score":0.08924055},"labels":[],"label_agreement":null},{"id":"W4205570659","doi":"10.1002/9781444324044.refs","title":"References","year":2010,"lang":"en","type":"other","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Atomic Energy of Canada Limited; Defense Advanced Research Projects Agency; University of Cambridge","keywords":"Computer science","score_opus":0.01182738381344872,"score_gpt":0.2729167193913131,"score_spread":0.2610893355778644,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4205570659","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0004705288,0.0044990373,0.008582641,0.0041090986,0.006743467,0.00022000755,0.021769887,0.0032305112,0.95037484],"genre_scores_gemma":[0.0018629794,0.0033417197,0.0066136685,0.001573096,0.0007389404,0.00009430766,0.02156927,0.0016827438,0.9625233],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9990784,0.00011137192,0.00004198451,0.000096573815,0.0006117126,0.000059899172],"domain_scores_gemma":[0.9978073,0.00031305203,0.00006264248,0.00029663768,0.0013544144,0.00016585304],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0007661252,0.0007474031,0.0007622227,0.007066331,0.0013707477,0.0025835263,0.0017418425,0.000945565,0.5227389],"category_scores_gemma":[0.005417329,0.0002543986,0.0006402752,0.006479923,0.0003803799,0.002311611,0.0015225019,0.0011535359,0.48664877],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000008915131,0.00001948031,0.000053904147,0.00009302536,0.0000018278977,0.000024802044,0.00001804227,0.000080017555,0.000093219765,0.0030159352,0.9365785,0.060012225],"study_design_scores_gemma":[0.0000020809089,0.000002065166,0.00010098824,0.000086412605,0.0000037042653,0.000033272227,0.000018978457,0.000071588984,0.00016040808,0.001524456,0.99799144,0.0000045287375],"about_ca_topic_score_codex":0.017701864,"about_ca_topic_score_gemma":0.030857751,"teacher_disagreement_score":0.47726113,"about_ca_system_score_codex":0.0015490707,"about_ca_system_score_gemma":0.0021323538,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W4206165639","doi":"10.1145/775107.775138","title":"Discovering word senses from text","year":2002,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":120,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Word (group theory); Computer science; Cluster analysis; Precision and recall; Set (abstract data type); Artificial intelligence; Similarity (geometry); Centroid; Natural language processing; Feature (linguistics); Feature vector; Element (criminal law); Space (punctuation); Cluster (spacecraft); Domain (mathematical analysis); Information retrieval; Mathematics; Linguistics; Image (mathematics)","score_opus":0.016333199237866095,"score_gpt":0.23864345945404686,"score_spread":0.22231026021618078,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4206165639","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.420488,0.0035553544,0.5399799,0.000856318,0.00035313706,0.0010403069,0.016503455,0.0074890736,0.009734431],"genre_scores_gemma":[0.356296,0.0014379752,0.6149664,0.00021272995,0.0001546019,0.0005830666,0.023580195,0.0005921966,0.0021767763],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99626213,0.0008185006,0.0005463578,0.0012786031,0.00092691457,0.00016744949],"domain_scores_gemma":[0.99162227,0.0045166654,0.00093389774,0.0010519669,0.0016583177,0.00021679392],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017522232,0.0012192827,0.0011702464,0.013209365,0.0013878815,0.0022629232,0.001090056,0.0011108535,0.0016529968],"category_scores_gemma":[0.014694427,0.0005952127,0.0013945047,0.0058559733,0.00096472184,0.004321162,0.0021658372,0.0008536202,0.0016581811],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007029556,0.00038751872,0.05773765,0.0024504296,0.00048251796,0.0023541562,0.004287329,0.0110694375,0.061914556,0.014102001,0.021430872,0.82308054],"study_design_scores_gemma":[0.00025519237,0.00080326607,0.08306502,0.001238367,0.0009964285,0.011259894,0.012552222,0.43329802,0.15335666,0.13278788,0.1698433,0.0005437078],"about_ca_topic_score_codex":0.0019350456,"about_ca_topic_score_gemma":0.004160596,"teacher_disagreement_score":0.013209365,"about_ca_system_score_codex":0.00066466734,"about_ca_system_score_gemma":0.0014212176,"threshold_uncertainty_score":0.009266794},"labels":[],"label_agreement":null},{"id":"W4206648524","doi":"10.1109/iisec54230.2021.9672408","title":"Meticulous Use of Turkish Informatics Terms to Deter Defective Turkish","year":2021,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Turkish; Informatics; Computer science; Political science; Linguistics; Law; Philosophy","score_opus":0.015728383604285093,"score_gpt":0.25852815637669724,"score_spread":0.24279977277241216,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4206648524","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19932668,0.011847489,0.48688275,0.037239928,0.004570909,0.0015159358,0.00261423,0.006661958,0.24934007],"genre_scores_gemma":[0.47757256,0.004555801,0.46385708,0.003603033,0.00042389514,0.00045657568,0.0031580399,0.0024550548,0.04391792],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9943217,0.0021753425,0.0011342607,0.00041977275,0.0014988171,0.00045007688],"domain_scores_gemma":[0.98345506,0.0035909652,0.0021537715,0.0023499767,0.007896877,0.00055343006],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0046936898,0.0007603829,0.00067724776,0.0056988457,0.0030092746,0.004951995,0.0015354047,0.0011443442,0.008130121],"category_scores_gemma":[0.017086111,0.00048551994,0.00052779174,0.003203607,0.0030954827,0.008535416,0.004606084,0.0030074953,0.005291594],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003361459,0.00013975786,0.010241802,0.0016274134,0.00003776005,0.0016358783,0.03295866,0.0010167663,0.028414823,0.23341238,0.123142555,0.567036],"study_design_scores_gemma":[0.000019926332,0.00013155983,0.006237502,0.00082080904,0.000056166733,0.0037199806,0.023197014,0.0023443976,0.020273492,0.029904013,0.9131381,0.00015708874],"about_ca_topic_score_codex":0.0053742602,"about_ca_topic_score_gemma":0.0112695955,"teacher_disagreement_score":0.008130121,"about_ca_system_score_codex":0.0022188472,"about_ca_system_score_gemma":0.0064677964,"threshold_uncertainty_score":0.027198017},"labels":[],"label_agreement":null},{"id":"W4206885548","doi":"10.18653/v1/w19-57","title":"Proceedings of the 16th Meeting on the Mathematics of Language","year":2019,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canada Research Chairs; University of Toronto","funders":"Stony Brook University; National Science Foundation","keywords":"Computer science; Mathematics education; Programming language; Mathematics","score_opus":0.015293893621284813,"score_gpt":0.26346594467212603,"score_spread":0.24817205105084122,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4206885548","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016388955,0.13201235,0.2460613,0.09074629,0.2096916,0.00029447666,0.003081929,0.001278477,0.3004446],"genre_scores_gemma":[0.1670778,0.056008715,0.16743442,0.0079589635,0.06417632,0.00081307016,0.007083704,0.0020910695,0.5273559],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9972045,0.0009973674,0.00025594418,0.0007337492,0.000567431,0.00024097996],"domain_scores_gemma":[0.9968194,0.001046928,0.00011229985,0.00071985263,0.00083048583,0.00047108074],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0047216103,0.0014717297,0.0016556042,0.0019609344,0.0015375321,0.007780926,0.0019130034,0.0016593032,0.04236255],"category_scores_gemma":[0.00677173,0.00057575776,0.0021938828,0.0016200668,0.002706816,0.0055237534,0.003940766,0.0067202575,0.013934821],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023297149,0.00012336804,0.0007184525,0.00064899167,0.00018473784,0.00027124266,0.00047188657,0.003099268,0.002167239,0.4285867,0.41102707,0.15246809],"study_design_scores_gemma":[0.0000356638,0.000054816093,0.00065772637,0.0002210914,0.00005664672,0.0002293068,0.000119146294,0.0039961953,0.0010351101,0.1530146,0.8405413,0.000038409227],"about_ca_topic_score_codex":0.0024595452,"about_ca_topic_score_gemma":0.003367976,"teacher_disagreement_score":0.04236255,"about_ca_system_score_codex":0.0059742527,"about_ca_system_score_gemma":0.0025968899,"threshold_uncertainty_score":0.14171684},"labels":[],"label_agreement":null},{"id":"W4207037623","doi":"10.31219/osf.io/m567r","title":"A corpus-based approach to map target vowel asymmetry in Brazilian Veneto metaphony","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Saint Mary's University; Université Laval","funders":"McGill University; Ball State University","keywords":"Variation (astronomy); Vowel; Linguistics; Variety (cybernetics); Computer science; Natural language processing; Artificial intelligence; Speech recognition","score_opus":0.014149672245924125,"score_gpt":0.2690469867725536,"score_spread":0.2548973145266295,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4207037623","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.90874463,0.00092874205,0.056816343,0.00025035284,0.000053160493,0.00048371777,0.009495677,0.0007970201,0.022430353],"genre_scores_gemma":[0.9398098,0.000269909,0.049673796,0.00004940514,0.00002312293,0.000612364,0.0069868434,0.00022606479,0.0023487245],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9992606,0.00020416545,0.000055209774,0.00028317788,0.00012908559,0.0000678099],"domain_scores_gemma":[0.9971137,0.0011763567,0.00033007443,0.0006138836,0.00066302525,0.00010292775],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010231497,0.00034721207,0.0003388442,0.0046032337,0.0012246409,0.0010018288,0.00054763036,0.00039778472,0.0037060906],"category_scores_gemma":[0.0059122685,0.00030098803,0.00024083843,0.005108642,0.0008707245,0.00062607665,0.0016200591,0.00042939835,0.0005023415],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00094739563,0.00038963946,0.15938121,0.0022282854,0.00019289214,0.0024197614,0.04481508,0.009167101,0.23842117,0.04252853,0.012402749,0.48710632],"study_design_scores_gemma":[0.000110347646,0.000273612,0.7916733,0.00024124976,0.00021986914,0.0018383216,0.016003544,0.03908818,0.028959543,0.011364354,0.11008703,0.00014074914],"about_ca_topic_score_codex":0.040503968,"about_ca_topic_score_gemma":0.069802105,"teacher_disagreement_score":0.040503968,"about_ca_system_score_codex":0.00092243234,"about_ca_system_score_gemma":0.0011309723,"threshold_uncertainty_score":0.080536425},"labels":[],"label_agreement":null},{"id":"W4210301664","doi":"10.1075/lab.21017.arc","title":"Phonological parsing via an integrated I-language","year":2022,"lang":"en","type":"article","venue":"Linguistic Approaches to Bilingualism","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"","keywords":"Parsing; Computer science; Natural language processing; Property (philosophy); Linguistics; Syntax; Phonology; Grammar; Artificial intelligence; Context (archaeology); Philosophy; History","score_opus":0.07898159509638313,"score_gpt":0.29193880521881505,"score_spread":0.2129572101224319,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4210301664","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15207072,0.00032119037,0.76810735,0.0017797013,0.00005757802,0.00006375847,0.0005172863,0.0048949863,0.07218745],"genre_scores_gemma":[0.90863526,0.00011685344,0.087169506,0.0002292643,0.000024946557,0.000029917632,0.0002212818,0.0004614207,0.0031116554],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99910885,0.00029453053,0.000049675018,0.00030942634,0.00016574028,0.00007170783],"domain_scores_gemma":[0.99721575,0.0011292128,0.00028933314,0.0009031902,0.0003743903,0.000088107925],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010145096,0.00036606728,0.00032448783,0.0011833981,0.0005188556,0.0043149646,0.0010009792,0.0007341397,0.008411713],"category_scores_gemma":[0.0050468384,0.0004296856,0.0006128866,0.0009000114,0.0031366006,0.0056767324,0.0025467798,0.0013553533,0.0016037723],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001727871,0.00008152946,0.008151967,0.00020664511,0.000049561964,0.00072449836,0.0046824217,0.005031937,0.02977968,0.8197497,0.0027639347,0.12860535],"study_design_scores_gemma":[0.000035550478,0.000091867805,0.007042678,0.000119027805,0.000090792666,0.0014091678,0.0013666396,0.09104692,0.026406487,0.8452517,0.027047712,0.00009143236],"about_ca_topic_score_codex":0.0026721687,"about_ca_topic_score_gemma":0.0014655859,"teacher_disagreement_score":0.008411713,"about_ca_system_score_codex":0.0010626444,"about_ca_system_score_gemma":0.0010666314,"threshold_uncertainty_score":0.028139949},"labels":[],"label_agreement":null},{"id":"W4210884608","doi":"10.1007/s13369-022-06588-w","title":"Improving Neural Machine Translation for Low Resource Algerian Dialect by Transductive Transfer Learning Strategy","year":2022,"lang":"en","type":"article","venue":"Arabian Journal for Science and Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Trois-Rivières","funders":"Direction Générale de la Recherche Scientifique et du Développement Technologique","keywords":"Transfer of learning; Computer science; Artificial intelligence; Machine translation; Translation (biology); Natural language processing; Arabic; Sequence (biology); Resource (disambiguation); Machine learning; Linguistics; Chemistry","score_opus":0.009304821083073348,"score_gpt":0.23237242085521384,"score_spread":0.2230675997721405,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4210884608","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.29951212,0.0028584665,0.65855336,0.0012577834,0.0012990687,0.00017871601,0.0015336849,0.010787304,0.024019506],"genre_scores_gemma":[0.75690675,0.0008650994,0.22148809,0.00032964454,0.00022197301,0.000114731214,0.0036175812,0.00067136943,0.0157848],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996592,0.00010562028,0.000032534404,0.00010074167,0.000056283232,0.00004573888],"domain_scores_gemma":[0.99952734,0.0001626574,0.000023596795,0.00008646466,0.00017907817,0.000020826541],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004688172,0.0006944173,0.00070214394,0.0006865387,0.00067255535,0.0010213412,0.0005487579,0.0006268791,0.005384417],"category_scores_gemma":[0.0012824049,0.00016607299,0.0005432159,0.0008052756,0.0002587067,0.0010668875,0.00080868363,0.00086915906,0.003557705],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00045950687,0.0003384135,0.0015831231,0.00048083582,0.00017019146,0.0008151127,0.000354969,0.058042333,0.0763531,0.0139763355,0.026356824,0.82106924],"study_design_scores_gemma":[0.00008072493,0.00029038714,0.0022016352,0.000047650923,0.00016888035,0.00047285896,0.00035597678,0.90867335,0.051842045,0.017005263,0.018809823,0.00005136173],"about_ca_topic_score_codex":0.004109311,"about_ca_topic_score_gemma":0.0047322237,"teacher_disagreement_score":0.005384417,"about_ca_system_score_codex":0.00035819106,"about_ca_system_score_gemma":0.00097396347,"threshold_uncertainty_score":0.018012643},"labels":[],"label_agreement":null},{"id":"W4211156271","doi":"10.2200/s00239ed1v01y200912hlt006","title":"Semantic Role Labeling","year":2011,"lang":"en","type":"article","venue":"Institutional Research Information System (Università degli Studi di Trento)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":70,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"FrameNet; Computer science; Semantic role labeling; Natural language processing; Artificial intelligence; Parsing; Task (project management); Inference; Annotation; Context (archaeology)","score_opus":0.07674148792415966,"score_gpt":0.3072401343124997,"score_spread":0.23049864638834003,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4211156271","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.001945139,0.003881793,0.69400907,0.00653988,0.00258383,0.00036925575,0.0025917,0.0048250826,0.2832543],"genre_scores_gemma":[0.0904789,0.008295106,0.72888213,0.004064563,0.0019001397,0.0008285338,0.012550046,0.0040087844,0.14899182],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99704355,0.0010029126,0.00022761665,0.00067196536,0.000813177,0.00024068754],"domain_scores_gemma":[0.9975489,0.0010164996,0.00010527134,0.0006934041,0.00053274364,0.00010323935],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031463217,0.0014200156,0.0008686557,0.0030287234,0.002915093,0.0052059987,0.0033237797,0.0023379214,0.0457732],"category_scores_gemma":[0.006784969,0.00089842716,0.001313416,0.0027421634,0.0030145166,0.013560338,0.0033967416,0.0035106293,0.023154885],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003754634,0.00002591855,0.00017608082,0.0002339081,0.0000088042425,0.00011072052,0.000867035,0.00070310925,0.0011831439,0.8018875,0.076152936,0.118613295],"study_design_scores_gemma":[0.000006153306,0.000009907033,0.000125284,0.00020800659,0.000008973608,0.0003114982,0.00041060237,0.0034494873,0.0016370428,0.27081186,0.72299886,0.000022303306],"about_ca_topic_score_codex":0.0020664346,"about_ca_topic_score_gemma":0.0024600388,"teacher_disagreement_score":0.0457732,"about_ca_system_score_codex":0.0029756748,"about_ca_system_score_gemma":0.002370837,"threshold_uncertainty_score":0.15312666},"labels":[],"label_agreement":null},{"id":"W4211256942","doi":"10.2200/s00866ed1v01y201807dtm049","title":"Natural Language Data Management and Interfaces","year":2018,"lang":"en","type":"article","venue":"Synthesis lectures on data management","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Natural (archaeology); Computer science; Geology","score_opus":0.028126999865088882,"score_gpt":0.30841588384572977,"score_spread":0.2802888839806409,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4211256942","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0023828656,0.010639064,0.9052719,0.010806255,0.004439154,0.000109282664,0.0012642234,0.006083323,0.059003938],"genre_scores_gemma":[0.074900486,0.019029535,0.5486487,0.0029079353,0.010841737,0.00054232107,0.0075524887,0.0067137554,0.328863],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9981651,0.00032906805,0.00015358804,0.0005968392,0.0006409629,0.0001144119],"domain_scores_gemma":[0.99680257,0.0015083613,0.00007959246,0.0006382401,0.00077643746,0.0001948007],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003480668,0.0012214605,0.0012208226,0.0020009323,0.0010055241,0.007911374,0.0017367641,0.0013306024,0.062164698],"category_scores_gemma":[0.0071204784,0.0009107218,0.00129067,0.0032392272,0.0019279536,0.013728419,0.0022537163,0.0039437097,0.023803007],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007184661,0.000088651905,0.00020547057,0.00042120923,0.000026514077,0.000055285593,0.0006145036,0.0013942295,0.0052758567,0.31824687,0.2652693,0.40833026],"study_design_scores_gemma":[0.000022852128,0.00003362029,0.00036460094,0.00019694364,0.000030357243,0.00018224597,0.0002060351,0.011776441,0.0052649267,0.31788933,0.6639935,0.000039216615],"about_ca_topic_score_codex":0.0019522504,"about_ca_topic_score_gemma":0.0017173233,"teacher_disagreement_score":0.062164698,"about_ca_system_score_codex":0.0029396901,"about_ca_system_score_gemma":0.0016492312,"threshold_uncertainty_score":0.20796162},"labels":[],"label_agreement":null},{"id":"W4213305977","doi":"10.7820/vli.v10.2.mizumoto","title":"Comparisons of word lists on new word level checker","year":2021,"lang":"en","type":"article","venue":"Vocabulary Learning and Instruction","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"Japan Society for the Promotion of Science","keywords":"Word (group theory); Computer science; Natural language processing; Linguistics; Artificial intelligence; Philosophy","score_opus":0.025717479921638794,"score_gpt":0.2767666383681596,"score_spread":0.2510491584465208,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4213305977","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4092222,0.0017162893,0.42503396,0.00064824923,0.0019001534,0.0019217964,0.021653622,0.07693979,0.060963973],"genre_scores_gemma":[0.5491101,0.00046072673,0.3942962,0.00074717606,0.0002005946,0.0016544338,0.028585372,0.009077246,0.015868094],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9897251,0.0026283038,0.0022063635,0.0017294089,0.0033315967,0.00037926162],"domain_scores_gemma":[0.9547529,0.026299596,0.0026800754,0.006472718,0.008846566,0.0009481318],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0057815127,0.0009741817,0.0011667262,0.0061576734,0.0006848416,0.003932679,0.0015337411,0.0008700316,0.024868574],"category_scores_gemma":[0.056995153,0.00055143016,0.0007103061,0.0041247457,0.000766936,0.0071397196,0.0042603137,0.0013436953,0.009327418],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0032729204,0.00067152054,0.024220645,0.0015960854,0.00018340527,0.0007535581,0.0027793315,0.003568123,0.029337969,0.02040148,0.039151765,0.87406325],"study_design_scores_gemma":[0.0010554816,0.002844918,0.09248458,0.0013761547,0.0005048329,0.003555439,0.0063263737,0.13174812,0.20181054,0.07236985,0.48509303,0.0008306551],"about_ca_topic_score_codex":0.0016082061,"about_ca_topic_score_gemma":0.0018735495,"teacher_disagreement_score":0.024868574,"about_ca_system_score_codex":0.00092357036,"about_ca_system_score_gemma":0.0011880645,"threshold_uncertainty_score":0.08319366},"labels":[],"label_agreement":null},{"id":"W4213364197","doi":"10.18653/v1/2021.mwe-1","title":"Proceedings of the 17th Workshop on Multiword Expressions (MWE 2021)","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Agence Nationale de la Recherche","keywords":"Computer science","score_opus":0.020441662368084736,"score_gpt":0.2868270865364936,"score_spread":0.2663854241684088,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4213364197","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019220507,0.04877127,0.69213176,0.029845588,0.04030094,0.00092529383,0.015557439,0.024658047,0.12858908],"genre_scores_gemma":[0.052629832,0.020311857,0.43465048,0.007426485,0.0066355076,0.0012121161,0.076368466,0.01551405,0.38525122],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.995736,0.0017191708,0.00031774028,0.0010171267,0.00088744564,0.00032248153],"domain_scores_gemma":[0.99378407,0.0022073863,0.00015284424,0.0017629323,0.0014524601,0.0006403932],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0060537565,0.0020441427,0.0026866742,0.0021595766,0.0014368793,0.0068047694,0.0034155224,0.0031078602,0.12154159],"category_scores_gemma":[0.010861643,0.0009999764,0.0020709785,0.0025252118,0.0012127201,0.010087553,0.0063683423,0.0044244677,0.0815786],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00059135683,0.0003464711,0.00042815323,0.0005977809,0.0001300755,0.00030956205,0.00042180048,0.0013600377,0.0068060667,0.019706115,0.58962363,0.37967902],"study_design_scores_gemma":[0.00006912005,0.00006869657,0.00070080155,0.000304837,0.000059825255,0.00032890434,0.0002093714,0.007152417,0.0042574187,0.02998483,0.9568242,0.000039481492],"about_ca_topic_score_codex":0.0021063657,"about_ca_topic_score_gemma":0.0038475792,"teacher_disagreement_score":0.12154159,"about_ca_system_score_codex":0.0014386189,"about_ca_system_score_gemma":0.0021843193,"threshold_uncertainty_score":0.40659714},"labels":[],"label_agreement":null},{"id":"W4214694936","doi":"10.5430/elr.v9n4p52","title":"Reviewer Acknowledgements for English Linguistics Research, Vol. 9, No. 4","year":2020,"lang":"en","type":"article","venue":"English Linguistics Research","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Library science; Suite; Sociology; History; Computer science; Archaeology","score_opus":0.10715902394289835,"score_gpt":0.41851746083928676,"score_spread":0.3113584368963884,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4214694936","genre_codex":"editorial","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00023640186,0.005374007,0.0039536906,0.13499308,0.84527075,0.0009795259,0.0008923363,0.00083037384,0.007469806],"genre_scores_gemma":[0.016214313,0.018960213,0.020219153,0.17934167,0.57555753,0.008954109,0.0039053415,0.005500237,0.17134748],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.92058086,0.020016387,0.015399759,0.0062649264,0.035462637,0.00227552],"domain_scores_gemma":[0.17572829,0.03336818,0.016345939,0.012117419,0.7524312,0.010009029],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.06794453,0.0030232454,0.004274292,0.012675603,0.0067099403,0.012010727,0.005687301,0.007570564,0.09776914],"category_scores_gemma":[0.53700566,0.0013818832,0.0029260474,0.007292083,0.002945902,0.009965563,0.005848505,0.00920302,0.06389409],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000019800316,0.000003849482,0.00011923263,0.00040409304,0.000009574259,0.0000432557,0.00019601303,0.000007158339,0.000040873023,0.00040504892,0.9889949,0.009756249],"study_design_scores_gemma":[0.00003559904,0.000015505833,0.0004595568,0.0020573272,0.000042777436,0.0003392179,0.000765295,0.00012368172,0.00012026187,0.0011829851,0.9948036,0.00005418162],"about_ca_topic_score_codex":0.0035989918,"about_ca_topic_score_gemma":0.005708762,"teacher_disagreement_score":0.09776914,"about_ca_system_score_codex":0.0061489344,"about_ca_system_score_gemma":0.018029356,"threshold_uncertainty_score":0.35932928},"labels":[],"label_agreement":null},{"id":"W4214706066","doi":"10.5430/wjel.v12n1p185","title":"Evaluation of Google Image Translate in Rendering Arabic Signage into English","year":2022,"lang":"en","type":"article","venue":"World Journal of English Language","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Rendering (computer graphics); Lexis; Arabic; Artificial intelligence; Natural language processing; Information retrieval; Linguistics","score_opus":0.01291833824376038,"score_gpt":0.28765927858977264,"score_spread":0.27474094034601226,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4214706066","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9569493,0.0012681645,0.009464852,0.00048225163,0.0004246701,0.00050746446,0.0030581572,0.009884188,0.017960846],"genre_scores_gemma":[0.9544801,0.00063094945,0.023661237,0.00031039794,0.000095289644,0.00017924442,0.009979305,0.0013586229,0.0093049165],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.996938,0.0012626312,0.0003234394,0.00035516228,0.0009539324,0.00016672559],"domain_scores_gemma":[0.98731625,0.0074109496,0.00040305298,0.001295529,0.0031528256,0.0004214463],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002803748,0.0015181653,0.0009858842,0.0016451918,0.0006521645,0.0020310576,0.0011534387,0.0013755183,0.004047597],"category_scores_gemma":[0.022298658,0.0002360334,0.0007115839,0.0013225331,0.0010541847,0.0021803044,0.001308103,0.0009241161,0.0041408306],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.017459342,0.0040575783,0.035702,0.0092683565,0.0009690472,0.0053343973,0.014099319,0.03851673,0.07666486,0.0055642314,0.08012656,0.71223766],"study_design_scores_gemma":[0.0020786275,0.014177918,0.1671533,0.0011418345,0.0013485714,0.007285033,0.019245083,0.47592017,0.1877729,0.0051027033,0.11784386,0.00093004364],"about_ca_topic_score_codex":0.014784547,"about_ca_topic_score_gemma":0.013222705,"teacher_disagreement_score":0.014784547,"about_ca_system_score_codex":0.0007318178,"about_ca_system_score_gemma":0.00083345507,"threshold_uncertainty_score":0.029396951},"labels":[],"label_agreement":null},{"id":"W4214810120","doi":"10.1093/llc/fqac006","title":"Towards a linked open data resource for direct speech acts in Greek and Latin epic","year":2022,"lang":"en","type":"article","venue":"Digital Scholarship in the Humanities","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Mount Allison University","funders":"European Social Fund; Vlaamse regering; Social Sciences and Humanities Research Council of Canada; Universität Rostock; Mount Allison University; Fonds Wetenschappelijk Onderzoek","keywords":"EPIC; Metadata; Scholarship; Linked data; Computer science; Literature; World Wide Web; Art; Political science; Semantic Web","score_opus":0.14228743085660875,"score_gpt":0.33403860886793274,"score_spread":0.191751178011324,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4214810120","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14869997,0.0021473754,0.5259481,0.012789401,0.0013443765,0.006035656,0.19255002,0.028064346,0.08242073],"genre_scores_gemma":[0.23079935,0.0013862835,0.4341424,0.0017570218,0.0004997708,0.0052066585,0.30517173,0.003028973,0.01800792],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.985329,0.0059149535,0.0024582273,0.0019886252,0.0037385204,0.00057068915],"domain_scores_gemma":[0.9225535,0.024908204,0.004750705,0.029351234,0.012998851,0.005437614],"candidate_categories":["open_science"],"consensus_categories":[],"category_scores_codex":[0.025539372,0.00059149944,0.0010541817,0.017331747,0.0033048796,0.009800053,0.0027999843,0.0022033795,0.008329809],"category_scores_gemma":[0.0660102,0.00065843255,0.0008008753,0.013322485,0.00274047,0.015285484,0.019826455,0.002552614,0.0062906113],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011336542,0.0014146137,0.030556874,0.0029694587,0.00018828912,0.0018332388,0.037018802,0.004481532,0.021638492,0.29717094,0.13718612,0.4644081],"study_design_scores_gemma":[0.0001064662,0.00018633401,0.01360325,0.0014713368,0.00008348948,0.00034517626,0.015519547,0.010792716,0.013259395,0.06439661,0.88003963,0.00019607825],"about_ca_topic_score_codex":0.007662799,"about_ca_topic_score_gemma":0.006831735,"teacher_disagreement_score":0.9972,"about_ca_system_score_codex":0.0024794517,"about_ca_system_score_gemma":0.0068621663,"threshold_uncertainty_score":0.13506669},"labels":[],"label_agreement":null},{"id":"W4214826913","doi":"10.1109/taslp.2022.3155281","title":"Dealing With Hierarchical Types and Label Noise in Fine-Grained Entity Typing","year":2022,"lang":"en","type":"article","venue":"IEEE/ACM Transactions on Audio Speech and Language Processing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Fundamental Research Funds for the Central Universities; State Key Laboratory of Software Development Environment; National Natural Science Foundation of China","keywords":"Benchmarking; Computer science; Noise (video); Hierarchical organization; Hierarchical database model; Artificial intelligence; Data mining; Pattern recognition (psychology); Machine learning; Image (mathematics)","score_opus":0.011227548454533245,"score_gpt":0.2616009364783938,"score_spread":0.25037338802386055,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4214826913","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08449117,0.0008206661,0.90970993,0.00040671296,0.00010958971,0.000093783994,0.00043207293,0.0031159343,0.00082023226],"genre_scores_gemma":[0.61932045,0.00046642742,0.37166896,0.00058202737,0.00019270818,0.00020108066,0.0028324649,0.0007509791,0.003984886],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9928243,0.0023270727,0.00049616356,0.0023378308,0.0014727686,0.00054181216],"domain_scores_gemma":[0.9602042,0.02657518,0.003459297,0.006711457,0.0023815248,0.00066834176],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010006743,0.001584977,0.002123253,0.0025701616,0.0015092023,0.0030324173,0.004145791,0.0035211574,0.0011762573],"category_scores_gemma":[0.030938145,0.0013714612,0.0014597581,0.0034990928,0.0023620045,0.007446857,0.0029688256,0.0050116135,0.001047705],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011385108,0.0006582161,0.058228716,0.00058338296,0.0002875884,0.0014074158,0.0026044336,0.3735707,0.025408233,0.023444861,0.010463117,0.5022048],"study_design_scores_gemma":[0.000023597404,0.00008759694,0.0051305047,0.00005372622,0.000049906754,0.00045314882,0.00020523916,0.95897186,0.009278608,0.022233007,0.0034605882,0.0000522317],"about_ca_topic_score_codex":0.010369438,"about_ca_topic_score_gemma":0.01886428,"teacher_disagreement_score":0.010369438,"about_ca_system_score_codex":0.0016846755,"about_ca_system_score_gemma":0.0026689223,"threshold_uncertainty_score":0.052921355},"labels":[],"label_agreement":null},{"id":"W4214859661","doi":"10.31234/osf.io/m397u","title":"Concreteness ratings for 62 thousand English multiword expressions","year":2022,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Concreteness; Psychology; Meaning (existential); Linguistics; Computer science; Word (group theory); Cognitive psychology; Natural language processing","score_opus":0.0208887967583278,"score_gpt":0.30851232475741847,"score_spread":0.2876235279990907,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4214859661","genre_codex":"empirical","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.961575,0.0009337291,0.0053302175,0.0002591977,0.00025206007,0.00024626905,0.01250107,0.0005385524,0.018363878],"genre_scores_gemma":[0.9292503,0.00078168995,0.019537471,0.00025320463,0.0001628667,0.0010056759,0.030141857,0.0004790786,0.018387893],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99679345,0.0006746941,0.0007485202,0.00037599896,0.001280635,0.00012680548],"domain_scores_gemma":[0.98123574,0.009911238,0.002155135,0.0013575426,0.0044640256,0.0008764],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001977665,0.0006129485,0.00043783683,0.00190151,0.00046882723,0.0009111797,0.0002964793,0.0008045312,0.008764706],"category_scores_gemma":[0.023492437,0.00017750532,0.0004942215,0.0013708436,0.00061279954,0.0014753873,0.0012188635,0.000750732,0.0035282567],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0045570084,0.0009834219,0.27771467,0.0048071,0.00049771235,0.0016224281,0.022660688,0.0024093068,0.09760463,0.0071112188,0.11934922,0.4606827],"study_design_scores_gemma":[0.00015364263,0.0013789363,0.830207,0.00047191588,0.00020390925,0.002687979,0.009889739,0.0073951473,0.021393329,0.004228571,0.12158501,0.0004048558],"about_ca_topic_score_codex":0.0008261506,"about_ca_topic_score_gemma":0.0021085127,"teacher_disagreement_score":0.008764706,"about_ca_system_score_codex":0.00032534494,"about_ca_system_score_gemma":0.00019594355,"threshold_uncertainty_score":0.029320896},"labels":[],"label_agreement":null},{"id":"W4220669679","doi":"10.5539/ijel.v12n3p46","title":"Causative Constructions in Modern Standard Arabic","year":2022,"lang":"en","type":"article","venue":"International Journal of English Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Causative; Linguistics; Verb; Arabic; Grammar; Subject (documents); Section (typography); Event (particle physics); Psychology; Biology; Computer science; Philosophy","score_opus":0.01128533384180114,"score_gpt":0.2901676730412227,"score_spread":0.27888233919942157,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4220669679","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5452218,0.012628519,0.16418824,0.0032859307,0.00040829828,0.0001572613,0.0007293285,0.0011538938,0.2722268],"genre_scores_gemma":[0.9747182,0.0016264942,0.016985789,0.00015086164,0.00008925411,0.000036379122,0.00013910199,0.00011346613,0.006140415],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99940026,0.00023000289,0.000051239942,0.000099494086,0.0001661449,0.000052795025],"domain_scores_gemma":[0.99913883,0.00032441816,0.00018815963,0.00009106193,0.00023476771,0.00002267889],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006673116,0.00052149437,0.00020764697,0.0020398838,0.0022031371,0.0017627026,0.00032251613,0.00059453293,0.005697522],"category_scores_gemma":[0.0016000911,0.000274635,0.0003022285,0.0015344467,0.002180411,0.0023538778,0.0010781181,0.00078806165,0.00084287074],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006579695,0.00003595267,0.0055464897,0.0004105235,0.000028500202,0.0019857849,0.031139217,0.0008458678,0.01055316,0.8644279,0.0033313346,0.08162938],"study_design_scores_gemma":[0.000038288705,0.00011599495,0.026346732,0.0006114333,0.00011432138,0.011993737,0.024483074,0.0102334665,0.01683798,0.39993066,0.50917053,0.00012368683],"about_ca_topic_score_codex":0.001809291,"about_ca_topic_score_gemma":0.0020812894,"teacher_disagreement_score":0.005697522,"about_ca_system_score_codex":0.0017691422,"about_ca_system_score_gemma":0.00067210075,"threshold_uncertainty_score":0.019060075},"labels":[],"label_agreement":null},{"id":"W4220771852","doi":"10.29173/cais1281","title":"A design approach for a distributed system to handle Chinese","year":2022,"lang":"en","type":"article","venue":"Proceedings of the Annual Conference of CAIS / Actes du congrès annuel de l ACSI","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Alphabet; Computer science; Information retrieval; Theoretical computer science; Natural language processing; Artificial intelligence; Linguistics","score_opus":0.02137387241242492,"score_gpt":0.2535068351991103,"score_spread":0.2321329627866854,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4220771852","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0051581846,0.00006448647,0.98530465,0.00034216073,0.00007971374,0.00040924613,0.000056035526,0.0022205524,0.006365002],"genre_scores_gemma":[0.12335297,0.00014392809,0.85466325,0.00047290197,0.00007323725,0.0014219208,0.00025480366,0.00028433723,0.019332714],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99933726,0.00014369513,0.00005534697,0.00016530945,0.00022339741,0.000074949894],"domain_scores_gemma":[0.9993136,0.00010107208,0.000030821833,0.00015341674,0.00030694602,0.000094064555],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012930459,0.0004181353,0.00036196652,0.00045924354,0.0013415362,0.0019215965,0.0018268845,0.0010773697,0.0064750407],"category_scores_gemma":[0.0015366486,0.00044947158,0.0005742999,0.000490706,0.00070197973,0.0020731762,0.0013326342,0.0010773224,0.0031788612],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00078317075,0.0006536208,0.0029446983,0.00097624154,0.00020478918,0.001730517,0.003658749,0.050917942,0.18332075,0.26213306,0.03452733,0.45814914],"study_design_scores_gemma":[0.00056761125,0.0013291318,0.0014233931,0.00019735348,0.00033268085,0.0016263119,0.0010337773,0.47180095,0.08079348,0.07654207,0.36418507,0.00016815287],"about_ca_topic_score_codex":0.003142966,"about_ca_topic_score_gemma":0.004469532,"teacher_disagreement_score":0.0064750407,"about_ca_system_score_codex":0.0009956729,"about_ca_system_score_gemma":0.0020914935,"threshold_uncertainty_score":0.021661162},"labels":[],"label_agreement":null},{"id":"W4220933044","doi":"10.21203/rs.3.rs-1412249/v1","title":"Openvar: Functional Annotation Of Variants In Non-Canonical Open Reading Frames","year":2022,"lang":"en","type":"preprint","venue":"Research Square","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Sherbrooke","funders":"Canadian Institutes of Health Research; Ministère de l'Économie, de la Science et de l'Innovation - Québec","keywords":"Annotation; Open reading frame; Computer science; Reading (process); Non canonical; Artificial intelligence; Natural language processing; Biology; Genetics; Linguistics; Philosophy; Gene","score_opus":0.08826492797028386,"score_gpt":0.43830757852372015,"score_spread":0.3500426505534363,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4220933044","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029910913,0.0006862294,0.86507714,0.00079328526,0.0010068424,0.00015507794,0.01610797,0.07611309,0.010149539],"genre_scores_gemma":[0.33892614,0.00085616333,0.5371033,0.0005677603,0.00085812923,0.00039787657,0.067431934,0.035098795,0.018759917],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9972857,0.00045637053,0.00028989182,0.0011022424,0.00057891913,0.0002868833],"domain_scores_gemma":[0.9937115,0.0029454448,0.00047619964,0.0016581807,0.0009906375,0.00021798287],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024919382,0.0023526317,0.0013651003,0.0026732143,0.0018421025,0.0038175974,0.003083025,0.0025649946,0.02006811],"category_scores_gemma":[0.007592625,0.0015649855,0.0021032938,0.0026713628,0.0017101942,0.006998399,0.003493554,0.0033268204,0.010383879],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0028352372,0.00036191492,0.010331789,0.0023589795,0.00021243017,0.0028870231,0.0039767385,0.005610401,0.05340081,0.39252472,0.1523152,0.37318474],"study_design_scores_gemma":[0.0002300134,0.00024460923,0.0048381877,0.0007197715,0.00034692092,0.0020799898,0.0017799987,0.05608033,0.07212603,0.4701616,0.39100346,0.00038915785],"about_ca_topic_score_codex":0.0022747286,"about_ca_topic_score_gemma":0.0026453615,"teacher_disagreement_score":0.02006811,"about_ca_system_score_codex":0.00084165734,"about_ca_system_score_gemma":0.0012836979,"threshold_uncertainty_score":0.0671345},"labels":[],"label_agreement":null},{"id":"W4220936221","doi":"10.5430/wjel.v12n1p334","title":"The L1 Semantic Retrieval of L2 Words: Evidence from Advanced L2 Learners’ Reaction Times","year":2022,"lang":"en","type":"article","venue":"World Journal of English Language","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Lemma (botany); Natural language processing; Meaning (existential); Vocabulary; Linguistics; Second language; Artificial intelligence; Translation (biology); Psychology","score_opus":0.009704467551901977,"score_gpt":0.26471066745116895,"score_spread":0.255006199899267,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4220936221","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9989135,0.000055535787,0.0002512595,0.00001310759,0.0000033186975,0.0000093415765,0.000054692988,0.000008380685,0.0006909093],"genre_scores_gemma":[0.9986822,0.00008854999,0.0003789675,0.00004067238,0.0000068522545,0.000032561897,0.000119172364,0.0000164613,0.000634491],"study_design_codex":"bench_or_experimental","study_design_gemma":"observational","domain_scores_codex":[0.99909174,0.0002763027,0.00008749406,0.0002194982,0.00025768246,0.000067147834],"domain_scores_gemma":[0.99058527,0.0060448158,0.0016066054,0.00062507065,0.0006941306,0.00044409826],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00138386,0.0005297096,0.00041450103,0.0004912452,0.00015280505,0.001093082,0.0002896647,0.0006405501,0.003149711],"category_scores_gemma":[0.014677012,0.00029687668,0.00023572844,0.00033125305,0.00054021954,0.000819006,0.00062281283,0.0006576045,0.00093915605],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.013851027,0.0015562568,0.3663202,0.0010318987,0.00048699233,0.0014601072,0.029778551,0.001246097,0.5071856,0.0008634002,0.00090854336,0.07531132],"study_design_scores_gemma":[0.0003295903,0.004086686,0.92304164,0.0000461231,0.000189003,0.0016196654,0.0043200757,0.0035865903,0.05985589,0.001075297,0.0017410752,0.00010828871],"about_ca_topic_score_codex":0.0014659718,"about_ca_topic_score_gemma":0.0006880626,"teacher_disagreement_score":0.003149711,"about_ca_system_score_codex":0.00013902785,"about_ca_system_score_gemma":0.00016649459,"threshold_uncertainty_score":0.0105368495},"labels":[],"label_agreement":null},{"id":"W4221055937","doi":"10.29173/cais1288","title":"Lessons from machine-readable Chinese applied to managing retrieval of graphical and textual information","year":2022,"lang":"en","type":"article","venue":"Proceedings of the Annual Conference of CAIS / Actes du congrès annuel de l ACSI","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Information retrieval; Computer science; Natural language processing; Artificial intelligence; World Wide Web","score_opus":0.010595988666449898,"score_gpt":0.24565475615341567,"score_spread":0.23505876748696578,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4221055937","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.095421456,0.012764262,0.6603054,0.059728358,0.0020305559,0.00094374025,0.005016219,0.017981444,0.14580859],"genre_scores_gemma":[0.48740813,0.008345769,0.42631736,0.0037315458,0.0012867689,0.00047836284,0.004095742,0.003373192,0.06496306],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99846613,0.000670214,0.00015187894,0.00021176445,0.0003882439,0.00011180219],"domain_scores_gemma":[0.9831251,0.011989834,0.00023651715,0.0016533169,0.0027247046,0.0002705506],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0045315926,0.00078374083,0.0007741488,0.00214145,0.0015013037,0.0056493687,0.002941909,0.0015703547,0.018769557],"category_scores_gemma":[0.025690824,0.0006396771,0.0006135693,0.00300453,0.0035064695,0.016229162,0.0017156459,0.0018861172,0.0042024013],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005511278,0.00018144869,0.0037899148,0.0011570019,0.00006654156,0.0013633814,0.007848372,0.011225276,0.010045058,0.15056571,0.09847871,0.7147275],"study_design_scores_gemma":[0.00024395595,0.000391829,0.007247832,0.000554393,0.0002289955,0.0016427031,0.008450167,0.14613895,0.059927117,0.43598098,0.33892375,0.0002693118],"about_ca_topic_score_codex":0.0738634,"about_ca_topic_score_gemma":0.03695961,"teacher_disagreement_score":0.0738634,"about_ca_system_score_codex":0.0030942275,"about_ca_system_score_gemma":0.003959646,"threshold_uncertainty_score":0.14686692},"labels":[],"label_agreement":null},{"id":"W4221122266","doi":"10.5539/ijel.v12n3p18","title":"The Diachronic Shift of Japanese Transitive/Unaccusative Verb Pairs","year":2022,"lang":"en","type":"article","venue":"International Journal of English Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Grammaticalization; Linguistics; Adverb; Transitive relation; Adjective; Verb; Noun; Mathematics; Suffix; Philosophy; Combinatorics","score_opus":0.00924191741891605,"score_gpt":0.27345912715940673,"score_spread":0.2642172097404907,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4221122266","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.879602,0.00078023854,0.021036902,0.00026793574,0.00021076981,0.0000669139,0.00015768294,0.000229133,0.09764839],"genre_scores_gemma":[0.98893666,0.00015950529,0.0039802487,0.000078779725,0.00003317726,0.000028967099,0.00017124321,0.00015911742,0.006452296],"study_design_codex":"bench_or_experimental","study_design_gemma":"observational","domain_scores_codex":[0.9993954,0.00014163606,0.00006328911,0.0002151375,0.00011851744,0.00006603892],"domain_scores_gemma":[0.99938726,0.00016846003,0.00007525102,0.0001489831,0.00016947539,0.000050411265],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00056529883,0.00027217434,0.00023756799,0.00068318716,0.0014102363,0.0015294207,0.00041749282,0.00044897545,0.004261526],"category_scores_gemma":[0.0016754238,0.00033199,0.00032701797,0.0006961142,0.001707461,0.0023753333,0.0014836688,0.0008014506,0.0009116647],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00050565065,0.00009221035,0.022735778,0.0008479417,0.00005273132,0.0062021776,0.15866435,0.00034180444,0.39402753,0.24223875,0.0038323258,0.17045869],"study_design_scores_gemma":[0.00015191446,0.0006696263,0.20257841,0.00027986243,0.00026482518,0.021553185,0.071329705,0.0063709966,0.12199848,0.08310807,0.49133462,0.0003603216],"about_ca_topic_score_codex":0.0025789079,"about_ca_topic_score_gemma":0.0039524306,"teacher_disagreement_score":0.004261526,"about_ca_system_score_codex":0.0007724771,"about_ca_system_score_gemma":0.00039968584,"threshold_uncertainty_score":0.014256239},"labels":[],"label_agreement":null},{"id":"W4221136247","doi":"10.29173/cais1303","title":"An enhancement of Boolean retrieval systems based on term co-occurrence frequencies / Un amélioration des systèmes d’information basé sur la cooccurrence de termes","year":2022,"lang":"fr","type":"article","venue":"Proceedings of the Annual Conference of CAIS / Actes du congrès annuel de l ACSI","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"University of Regina","funders":"","keywords":"Term (time); Computer science; Mathematics; Physics","score_opus":0.021514740251472708,"score_gpt":0.26314960811638893,"score_spread":0.24163486786491623,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4221136247","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08095811,0.0042493115,0.8972634,0.0008029791,0.00049026276,0.00030216313,0.0009132075,0.010410004,0.0046104947],"genre_scores_gemma":[0.36100966,0.0016943171,0.62614316,0.00045111735,0.00051757373,0.00024536758,0.001658286,0.0004517763,0.007828632],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9984291,0.0003497775,0.00016816423,0.00025620667,0.0007064035,0.00009024925],"domain_scores_gemma":[0.99587446,0.0019359132,0.000188724,0.000456403,0.0014532136,0.00009123173],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026289057,0.00064093305,0.0013212701,0.0024045901,0.0004752512,0.0017337385,0.001085405,0.00070574414,0.006014367],"category_scores_gemma":[0.0065354095,0.00032553947,0.0007196585,0.0017089791,0.00034451226,0.002502606,0.0007190891,0.00072813506,0.002413106],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009310649,0.00028492307,0.001846933,0.0005085672,0.00017508277,0.00012702376,0.0001248259,0.0072089904,0.109135136,0.004035055,0.005951951,0.86967033],"study_design_scores_gemma":[0.0002784653,0.000885169,0.009874515,0.00009362931,0.0008806474,0.000944658,0.00013065535,0.8163489,0.1401982,0.006294637,0.023919402,0.00015113129],"about_ca_topic_score_codex":0.003942593,"about_ca_topic_score_gemma":0.0035211022,"teacher_disagreement_score":0.006014367,"about_ca_system_score_codex":0.0005593341,"about_ca_system_score_gemma":0.0007281756,"threshold_uncertainty_score":0.020120084},"labels":[],"label_agreement":null},{"id":"W4221144083","doi":"","title":"Concilier l'équité statistique et la précision en apprentissage machine interprétable grâce à la PLNE","year":2022,"lang":"fr","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal; École de Technologie Supérieure","funders":"","keywords":"Computer science","score_opus":0.011615879192490394,"score_gpt":0.27524509218264437,"score_spread":0.26362921299015396,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4221144083","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10971246,0.0016577333,0.87601614,0.0031350388,0.00028224528,0.00003733617,0.00036026735,0.0023999382,0.0063988217],"genre_scores_gemma":[0.7042464,0.000949789,0.27493587,0.00067031087,0.00068612635,0.00012879103,0.00070847094,0.0015373229,0.016136928],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98837644,0.0038251758,0.00078935793,0.0028361143,0.003660213,0.0005127122],"domain_scores_gemma":[0.90784395,0.076931976,0.0028096829,0.006639495,0.005248918,0.00052597444],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01067109,0.0014779338,0.0016424333,0.0031042972,0.0013316554,0.008094075,0.0016821577,0.00333558,0.005841669],"category_scores_gemma":[0.073070966,0.0011875938,0.0018374834,0.0017822923,0.004341933,0.008162855,0.002653756,0.0058925375,0.0017778635],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015027992,0.0002382034,0.01877575,0.0006923268,0.00056787714,0.00089285185,0.0020850904,0.25037423,0.041517004,0.34425646,0.0059724264,0.333125],"study_design_scores_gemma":[0.000042844986,0.0001398289,0.005300885,0.00010224141,0.000088714754,0.00029889253,0.00017146971,0.8600811,0.01786945,0.1112903,0.004566551,0.00004772166],"about_ca_topic_score_codex":0.010052316,"about_ca_topic_score_gemma":0.009636277,"teacher_disagreement_score":0.01067109,"about_ca_system_score_codex":0.0030805669,"about_ca_system_score_gemma":0.0028869829,"threshold_uncertainty_score":0.05643481},"labels":[],"label_agreement":null},{"id":"W4224090047","doi":"10.1609/aaai.v36i10.21343","title":"Text Revision By On-the-Fly Representation Optimization","year":2022,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"National Key Research and Development Program of China; Chinese University of Hong Kong","keywords":"Computer science; Natural language processing; Artificial intelligence; Transformer; Representation (politics); Inference; Natural language; Feature learning; Formality; Sequence (biology); Language model; Linguistics","score_opus":0.05825150952747766,"score_gpt":0.3112337834735554,"score_spread":0.25298227394607775,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4224090047","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023624584,0.0002465228,0.96276045,0.00022043813,0.00012577277,0.00016936849,0.00018877149,0.010875431,0.0017886482],"genre_scores_gemma":[0.4033385,0.00024052213,0.58090466,0.00045735057,0.00020949543,0.00034581093,0.0019050697,0.0023391317,0.010259581],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985746,0.0003208538,0.00010617034,0.00046469455,0.00039046974,0.00014315163],"domain_scores_gemma":[0.9964676,0.0013634026,0.00029632702,0.0011289178,0.0006075929,0.00013614431],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014961896,0.0015396836,0.001354997,0.00090359576,0.00058348756,0.00138254,0.0024381715,0.0012392404,0.005671051],"category_scores_gemma":[0.007951388,0.00048315316,0.0013948088,0.00075231225,0.0010561746,0.0027069899,0.002317675,0.0020270825,0.0035859493],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00047088633,0.00027943827,0.00097525446,0.00027565326,0.00008987323,0.00039288195,0.0003674972,0.19460437,0.039853062,0.011396542,0.016492696,0.7348018],"study_design_scores_gemma":[0.00007330868,0.000112378715,0.00019562595,0.000011244018,0.000040792267,0.00017948616,0.000081093014,0.9580597,0.02238117,0.01372876,0.0051150895,0.000021353764],"about_ca_topic_score_codex":0.0014172292,"about_ca_topic_score_gemma":0.0022726732,"teacher_disagreement_score":0.005671051,"about_ca_system_score_codex":0.0007054278,"about_ca_system_score_gemma":0.0012534849,"threshold_uncertainty_score":0.018971562},"labels":[],"label_agreement":null},{"id":"W4224279521","doi":"10.25159/2663-6573/9275","title":"Internal Temporal Structure of the Biblical Hebrew Verb: A Case Study of Lexical Aspect in the Verb yd'","year":2022,"lang":"en","type":"article","venue":"Journal for Semitics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; Wycliffe College","funders":"","keywords":"Linguistics; Verb; Biblical Hebrew; Rule-based machine translation; Hebrew; Computer science; Natural language processing; Psychology; Artificial intelligence; Philosophy; Hebrew Bible; Biblical studies","score_opus":0.02190315608070074,"score_gpt":0.32565246533570863,"score_spread":0.30374930925500787,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4224279521","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8075006,0.00046249022,0.025463857,0.0006365084,0.000030312067,0.000049173945,0.00009599901,0.00007783525,0.16568322],"genre_scores_gemma":[0.99129593,0.00012370771,0.0035306304,0.000036250112,0.000011799618,0.000013975851,0.00004831553,0.000031188694,0.004908122],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99947494,0.00027962585,0.000024921257,0.00006272946,0.00009670691,0.000061016948],"domain_scores_gemma":[0.99911755,0.00060160144,0.000107652995,0.00008991932,0.000059617363,0.000023523255],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00090378476,0.00015496596,0.0002096115,0.00079695205,0.0025489575,0.0025933797,0.00040884616,0.00090812065,0.002103531],"category_scores_gemma":[0.002483389,0.00020679165,0.0002341086,0.001130023,0.0036834546,0.0026408338,0.0013018437,0.0011635121,0.0001793514],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000118503805,0.00006936607,0.0070127174,0.00009309298,0.0000141826895,0.006043533,0.09569962,0.0008300642,0.009380447,0.8535364,0.0009400426,0.026261965],"study_design_scores_gemma":[0.00015859227,0.0003218839,0.09398248,0.0006297951,0.00012269843,0.020074401,0.17655177,0.027390648,0.03235773,0.32681337,0.3213773,0.00021923732],"about_ca_topic_score_codex":0.0098083485,"about_ca_topic_score_gemma":0.010600533,"teacher_disagreement_score":0.0098083485,"about_ca_system_score_codex":0.0023631384,"about_ca_system_score_gemma":0.00071791414,"threshold_uncertainty_score":0.01950252},"labels":[],"label_agreement":null},{"id":"W4225331178","doi":"10.46298/jdmdh.9123","title":"Towards an empirical evaluation of translated texts and translation quality","year":2022,"lang":"en","type":"article","venue":"Journal of Data Mining & Digital Humanities","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Trois-Rivières","funders":"","keywords":"Computer science; Machine translation; Natural language processing; Phraseology; Source text; Artificial intelligence; Context (archaeology); Quality (philosophy); Translation (biology); Linguistics","score_opus":0.27107377730309534,"score_gpt":0.42736644358020137,"score_spread":0.15629266627710603,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4225331178","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.22599944,0.0058187605,0.72107846,0.002609718,0.00031377643,0.0018129195,0.0028019608,0.0012525057,0.03831253],"genre_scores_gemma":[0.7347958,0.0014892467,0.2532876,0.0004437496,0.00033594927,0.0025817603,0.0037205047,0.0007277898,0.0026177107],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.8055444,0.13004248,0.013150205,0.011194759,0.038832694,0.0012354836],"domain_scores_gemma":[0.5494729,0.2975434,0.041035395,0.04148311,0.06808916,0.002375958],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.109679185,0.001648914,0.0015835079,0.011617981,0.0012953657,0.008555014,0.0026761284,0.0025513822,0.006669734],"category_scores_gemma":[0.4075953,0.0007758859,0.0010642126,0.013370885,0.007156247,0.010449621,0.0066799615,0.0024493656,0.0021832567],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012612636,0.0015612151,0.2139689,0.0056436285,0.001402917,0.00050243555,0.02055697,0.018458698,0.010378568,0.10493266,0.013430724,0.60790193],"study_design_scores_gemma":[0.0005740297,0.0035060982,0.4276375,0.0058784396,0.001383713,0.0015901731,0.02112822,0.16878796,0.038400542,0.23025712,0.1001939,0.0006623501],"about_ca_topic_score_codex":0.0015031133,"about_ca_topic_score_gemma":0.0009654546,"teacher_disagreement_score":0.109679185,"about_ca_system_score_codex":0.0023214922,"about_ca_system_score_gemma":0.0027661393,"threshold_uncertainty_score":0.58004594},"labels":[],"label_agreement":null},{"id":"W4225411538","doi":"10.18653/v1/2022.naacl-main.383","title":"Semantically Informed Slang Interpretation","year":2022,"lang":"en","type":"article","venue":"Proceedings of the 2022 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute; University of Toronto","funders":"Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada","keywords":"Slang; Computer science; Natural language processing; Interpretation (philosophy); Context (archaeology); Artificial intelligence; Semantic interpretation; Natural language; Machine translation; Linguistics; Programming language","score_opus":0.011456947446516935,"score_gpt":0.2585388599623224,"score_spread":0.24708191251580547,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4225411538","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022526667,0.00022916737,0.95889914,0.0004901841,0.00019830494,0.00017289225,0.0007756985,0.004956565,0.01175136],"genre_scores_gemma":[0.6630679,0.00023036123,0.32627532,0.00041224287,0.00012804255,0.00016889452,0.0029287436,0.0012642023,0.005524313],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9949523,0.0019844368,0.00034738405,0.00083727576,0.0015264628,0.0003520688],"domain_scores_gemma":[0.9949744,0.0012816065,0.00046691642,0.0018541107,0.0012886014,0.00013439782],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022939295,0.0012779868,0.0009170112,0.0015973707,0.0012299276,0.0026674708,0.0015362661,0.0014285314,0.008284203],"category_scores_gemma":[0.01189905,0.00049434917,0.0010101632,0.001505684,0.0024774666,0.0042553144,0.0043750545,0.0022385227,0.0029744217],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008004664,0.00030766518,0.0070974804,0.0013473213,0.0001611739,0.0014765083,0.00745089,0.080053285,0.065422736,0.24253641,0.034869462,0.5584766],"study_design_scores_gemma":[0.00008155664,0.00028919659,0.0026389966,0.000358459,0.00007752766,0.0010133756,0.0041507264,0.53534234,0.037414867,0.3300373,0.08838085,0.00021489116],"about_ca_topic_score_codex":0.0023209045,"about_ca_topic_score_gemma":0.0038264662,"teacher_disagreement_score":0.008284203,"about_ca_system_score_codex":0.0008995086,"about_ca_system_score_gemma":0.002030645,"threshold_uncertainty_score":0.027713418},"labels":[],"label_agreement":null},{"id":"W4225707828","doi":"10.18653/v1/2022.acl-long.512","title":"Neural reality of argument structure constructions","year":2022,"lang":"en","type":"article","venue":"Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute; University of Toronto","funders":"Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada","keywords":"Verb; Argument (complex analysis); Sentence; Linguistics; Computer science; Meaning (existential); Priming (agriculture); Mirroring; Natural language processing; Artificial intelligence; Psychology; Communication; Philosophy","score_opus":0.007626823306383453,"score_gpt":0.24520573462944859,"score_spread":0.23757891132306513,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4225707828","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.88911104,0.00044736004,0.071643054,0.0012450013,0.00009097018,0.000031524167,0.00031138974,0.00028720289,0.03683244],"genre_scores_gemma":[0.9907545,0.00014231299,0.007236664,0.00007084681,0.000013856607,0.000024116855,0.00023582947,0.00004307736,0.0014788278],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99946433,0.0001243144,0.000019616118,0.00025079888,0.00009835579,0.00004253223],"domain_scores_gemma":[0.9985851,0.00064978446,0.00020552277,0.0002859609,0.00015236421,0.00012132437],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006652111,0.00018375933,0.00026301353,0.00053283584,0.00041979033,0.0023320818,0.00054478936,0.00092918356,0.0035292376],"category_scores_gemma":[0.007559201,0.0004989664,0.00038400313,0.00028801162,0.0014219629,0.0038233048,0.0015498756,0.0013639287,0.00057740445],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00067611306,0.00017361584,0.041918058,0.0005606035,0.00018907597,0.0006127648,0.007753117,0.010823391,0.32955697,0.44028032,0.0024264122,0.1650295],"study_design_scores_gemma":[0.000090088666,0.0004711552,0.12873468,0.00012923547,0.00014189773,0.0015883333,0.0017860187,0.091172844,0.025518209,0.7375389,0.012728708,0.00009991735],"about_ca_topic_score_codex":0.0006354104,"about_ca_topic_score_gemma":0.00046655865,"teacher_disagreement_score":0.0035292376,"about_ca_system_score_codex":0.00043046128,"about_ca_system_score_gemma":0.00032052555,"threshold_uncertainty_score":0.011806548},"labels":[],"label_agreement":null},{"id":"W4226020023","doi":"10.1007/s10936-022-09863-x","title":"The Persian Lexicon Project: minimized orthographic neighbourhood effects in a dense language","year":2022,"lang":"en","type":"article","venue":"Journal of Psycholinguistic Research","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Athabasca University; University of Alberta","funders":"","keywords":"Psycholinguistics; Lexicon; Persian; Natural language processing; Linguistics; Computer science; Neighbourhood (mathematics); Artificial intelligence; Psychology; Cognition; Philosophy; Mathematics; Neuroscience","score_opus":0.036496273842947624,"score_gpt":0.40756488348840714,"score_spread":0.3710686096454595,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4226020023","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6373252,0.00040519735,0.33811846,0.0005022935,0.00008582398,0.00012398293,0.0028992638,0.0071373726,0.0134023465],"genre_scores_gemma":[0.7953601,0.00014135538,0.19424933,0.0001307654,0.000042828244,0.00015309821,0.0030002806,0.0014983573,0.0054238136],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99963593,0.00015351908,0.00001670261,0.0001070313,0.000057565754,0.00002923415],"domain_scores_gemma":[0.99931645,0.00028712908,0.00003307637,0.00020405967,0.00010793495,0.000051315008],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008150429,0.00040194514,0.00060882635,0.00057888957,0.0005691829,0.0011820862,0.0009911266,0.0004441864,0.0060381135],"category_scores_gemma":[0.0026249615,0.0002731307,0.0004087706,0.0007275035,0.0006501256,0.0017945238,0.0020527628,0.00046192034,0.001173578],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0026314792,0.00054053724,0.00999361,0.0006028783,0.00035237704,0.00086922705,0.0017122375,0.05161255,0.10409274,0.12990497,0.027371615,0.67031574],"study_design_scores_gemma":[0.0014838856,0.0009028728,0.012796907,0.000073419804,0.000586483,0.0008626766,0.0024493758,0.61987096,0.07465002,0.25266755,0.033487394,0.00016852985],"about_ca_topic_score_codex":0.005461851,"about_ca_topic_score_gemma":0.01426081,"teacher_disagreement_score":0.0060381135,"about_ca_system_score_codex":0.00022837904,"about_ca_system_score_gemma":0.0011952518,"threshold_uncertainty_score":0.020199478},"labels":[],"label_agreement":null},{"id":"W4226024751","doi":"10.1017/s1351324922000134","title":"Real-world sentence boundary detection using multitask learning: A case study on French","year":2022,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Hanbat National University","keywords":"Computer science; Sentence; Punctuation; Natural language processing; Task (project management); Artificial intelligence; Boundary (topology); Multi-task learning; Speech recognition","score_opus":0.01083990589589887,"score_gpt":0.27614783946117666,"score_spread":0.26530793356527776,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4226024751","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9574623,0.001284699,0.02884295,0.001378357,0.0001512362,0.00023538154,0.0038111156,0.002708739,0.0041251453],"genre_scores_gemma":[0.95157665,0.00023939359,0.039300818,0.00037780104,0.00010288897,0.00015068515,0.0059284847,0.0001912287,0.002132082],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99726033,0.0015348417,0.00015385561,0.00044925243,0.00043784807,0.00016385686],"domain_scores_gemma":[0.98725617,0.008560916,0.00061626226,0.0009797586,0.0020820107,0.00050492096],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002485326,0.0008659191,0.00051600044,0.0019204443,0.0015498062,0.0013158513,0.0011014926,0.0019108533,0.001264634],"category_scores_gemma":[0.012158814,0.00015175942,0.00056169444,0.0017077023,0.00075050996,0.0013251612,0.00072147825,0.0007789317,0.00074639946],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0033407833,0.003347955,0.11413224,0.0037814865,0.00070394657,0.03303708,0.01725966,0.06382222,0.08793525,0.006566803,0.09650704,0.5695656],"study_design_scores_gemma":[0.00061591616,0.002433633,0.23996113,0.0004910034,0.00044454282,0.012296382,0.019767428,0.41209218,0.12369832,0.013379518,0.17427783,0.0005421676],"about_ca_topic_score_codex":0.031283077,"about_ca_topic_score_gemma":0.043383162,"teacher_disagreement_score":0.031283077,"about_ca_system_score_codex":0.0012695241,"about_ca_system_score_gemma":0.0007627204,"threshold_uncertainty_score":0.062201977},"labels":[],"label_agreement":null},{"id":"W4226052266","doi":"10.5267/j.ijdns.2022.1.010","title":"Artificial intelligence for target symptoms of Thai herbal medicine by web scraping","year":2022,"lang":"en","type":"article","venue":"International Journal of Data and Network Science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"Walailak University","keywords":"Artificial intelligence; The Internet; Computer science; Medical knowledge; Trustworthiness; Machine learning; Traditional medicine; Medical education; World Wide Web; Medicine; Internet privacy","score_opus":0.034406122933907664,"score_gpt":0.3426756939727749,"score_spread":0.30826957103886726,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4226052266","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5524003,0.00075565674,0.4176448,0.0011928981,0.000072570096,0.00077787886,0.0020613263,0.004855497,0.020239113],"genre_scores_gemma":[0.7107009,0.00043079324,0.28190514,0.00017179378,0.00002727825,0.0002376191,0.0016403085,0.00007380118,0.0048122625],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997595,0.00007665932,0.000027725218,0.000053286585,0.00006658556,0.00001617898],"domain_scores_gemma":[0.9992017,0.00049463025,0.00006931491,0.000057296853,0.00015428205,0.000022772212],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00033437484,0.000370006,0.00027195283,0.0016598208,0.00026456354,0.00081301405,0.00030664683,0.0002649975,0.0022390129],"category_scores_gemma":[0.0023521539,0.000102793485,0.0005289491,0.001069348,0.000169429,0.0007985089,0.0003634343,0.00042200554,0.0008851801],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00031585872,0.00042018396,0.03076327,0.0005077055,0.00007504513,0.0009804369,0.0013556243,0.012903705,0.032328255,0.0027229777,0.005042708,0.9125842],"study_design_scores_gemma":[0.000062074745,0.0006430587,0.07717897,0.00029443408,0.00027242964,0.0026894712,0.0032608027,0.8107386,0.06137688,0.01637789,0.027011333,0.000093986244],"about_ca_topic_score_codex":0.0018474084,"about_ca_topic_score_gemma":0.0032819163,"teacher_disagreement_score":0.0022390129,"about_ca_system_score_codex":0.000245002,"about_ca_system_score_gemma":0.00053697306,"threshold_uncertainty_score":0.0074902773},"labels":[],"label_agreement":null},{"id":"W4226053219","doi":"10.1075/jial.22005.bow","title":"Translating for Canada, eh?","year":2021,"lang":"en","type":"article","venue":"The Journal of Internationalization and Localization","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Ottawa","funders":"","keywords":"License; Open educational resources; Computer science; Key (lock); Coronavirus disease 2019 (COVID-19); World Wide Web; Data science; Computer security","score_opus":0.010989961263098796,"score_gpt":0.26460791675210565,"score_spread":0.25361795548900684,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4226053219","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.053271327,0.0027513287,0.15923606,0.043519083,0.0041512796,0.0011838183,0.043704532,0.047016524,0.6451661],"genre_scores_gemma":[0.37475508,0.004117481,0.22095942,0.007683489,0.00052969524,0.00043658869,0.033153016,0.01059583,0.34776932],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99759656,0.0003780317,0.00016127058,0.00039651,0.0010343228,0.00043340883],"domain_scores_gemma":[0.9931463,0.000594113,0.00019035743,0.0007006844,0.0049433084,0.00042520426],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022972133,0.0008003787,0.0003920956,0.0026703926,0.004570162,0.0045109913,0.0008311499,0.0007816777,0.034753542],"category_scores_gemma":[0.0074559622,0.00033999849,0.00044497082,0.00355888,0.0021874874,0.0024330576,0.002808044,0.0013567558,0.009923119],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023831536,0.00008505058,0.004048866,0.0005423735,0.00002537501,0.0006824468,0.010117962,0.0016740977,0.011567107,0.12119864,0.49411592,0.3557039],"study_design_scores_gemma":[0.000022655002,0.0000177783,0.0028851072,0.00014817843,0.00001876225,0.00016765295,0.004337526,0.0014512296,0.0063841618,0.004436891,0.9800408,0.0000892509],"about_ca_topic_score_codex":0.90114117,"about_ca_topic_score_gemma":0.89436775,"teacher_disagreement_score":0.09885883,"about_ca_system_score_codex":0.018568536,"about_ca_system_score_gemma":0.04327982,"threshold_uncertainty_score":0.19888204},"labels":[],"label_agreement":null},{"id":"W4226084681","doi":"10.21083/nrsc.v2021i14.6246","title":"Traducteurs automatiques neuronaux comme outil didactique/pédagogique : DeepL dans l’apprentissage du français langue seconde","year":2021,"lang":"fr","type":"article","venue":"Nouvelle Revue Synergies Canada","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"University of Guelph","funders":"","keywords":"Humanities; Philosophy","score_opus":0.018209200040484784,"score_gpt":0.23697547018451662,"score_spread":0.21876627014403183,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4226084681","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10015497,0.0024505465,0.7827367,0.0084963795,0.00040686975,0.00033009137,0.0006055506,0.0063616806,0.098457195],"genre_scores_gemma":[0.6023311,0.0021417846,0.32534912,0.0011885149,0.00008685941,0.00033268303,0.0007742271,0.0012503122,0.06654534],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99654704,0.001663393,0.00022329969,0.0005313799,0.00076623634,0.00026858874],"domain_scores_gemma":[0.99441975,0.0025305504,0.0003267714,0.0010561133,0.001351555,0.00031520627],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004369919,0.00077616464,0.0004884155,0.001311774,0.0014251481,0.0056643,0.0011274413,0.0011369647,0.010429224],"category_scores_gemma":[0.008621863,0.0004646252,0.00067400455,0.0010529727,0.0041476497,0.00617437,0.0037075612,0.0023585474,0.0036189093],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000269995,0.0001723493,0.008586682,0.0012228696,0.000054447755,0.000674005,0.03741335,0.005122785,0.03585847,0.3511519,0.0116575,0.54781574],"study_design_scores_gemma":[0.000060363614,0.0003443529,0.008465412,0.0010456045,0.00014694029,0.0012693354,0.021473229,0.048793465,0.049160037,0.10629769,0.7627127,0.00023082035],"about_ca_topic_score_codex":0.039620608,"about_ca_topic_score_gemma":0.050428066,"teacher_disagreement_score":0.039620608,"about_ca_system_score_codex":0.0048133302,"about_ca_system_score_gemma":0.006407569,"threshold_uncertainty_score":0.078779995},"labels":[],"label_agreement":null},{"id":"W4226092505","doi":"","title":"Toolbox for Multimodal Learn (scikit-multimodallearn)","year":2021,"lang":"en","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Toolbox; Computer science; Programming language","score_opus":0.014286908454913852,"score_gpt":0.2548918070852912,"score_spread":0.24060489863037735,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4226092505","genre_codex":"software","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":"software","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0013866111,0.0002935288,0.11721863,0.000212743,0.00026748536,0.0002930621,0.051372994,0.8102114,0.018743556],"genre_scores_gemma":[0.034772146,0.0009921017,0.30114618,0.0013155357,0.0002528224,0.0048302836,0.30190384,0.28018188,0.07460517],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9989623,0.0001813623,0.00011792717,0.0003114306,0.00026364767,0.00016332486],"domain_scores_gemma":[0.99728787,0.0010994091,0.000101284386,0.0008329472,0.0003768142,0.00030164258],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015984914,0.0035081862,0.001793902,0.0026634082,0.00067398266,0.0025528048,0.004295132,0.0024598457,0.2827968],"category_scores_gemma":[0.008033511,0.0017911735,0.0027080444,0.001657849,0.000621884,0.0050195637,0.0075755166,0.0046440144,0.2923086],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00053513574,0.00028326546,0.00070426776,0.0018361527,0.00013594919,0.0002441749,0.00027574576,0.0024016285,0.003732068,0.006568217,0.7825572,0.20072618],"study_design_scores_gemma":[0.0008182484,0.0002130792,0.002113695,0.0009256198,0.00013930605,0.00056427886,0.00021032862,0.045368444,0.02479555,0.07451938,0.85002506,0.00030696284],"about_ca_topic_score_codex":0.0022055411,"about_ca_topic_score_gemma":0.0039335256,"teacher_disagreement_score":0.2827968,"about_ca_system_score_codex":0.0008232508,"about_ca_system_score_gemma":0.0015090909,"threshold_uncertainty_score":0.9460496},"labels":[],"label_agreement":null},{"id":"W4226129045","doi":"10.1109/tmrb.2022.3170210","title":"Surgical Procedure Understanding, Evaluation, and Interpretation: A Dictionary Factorization Approach","year":2022,"lang":"en","type":"article","venue":"IEEE Transactions on Medical Robotics and Bionics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Institutes of Health Research; Canada Foundation for Innovation; Government of Alberta","keywords":"Interpretation (philosophy); Computer science; Natural language processing; Artificial intelligence; Programming language","score_opus":0.027706515864215298,"score_gpt":0.285580246937577,"score_spread":0.2578737310733617,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4226129045","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020331776,0.00023157871,0.9770098,0.00032598464,0.000035114193,0.000119006705,0.00017766292,0.0005959899,0.0011731395],"genre_scores_gemma":[0.41471025,0.0002664006,0.58188665,0.00016826722,0.000087738576,0.00018450635,0.0006518169,0.00013959824,0.0019047309],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99885106,0.00040774202,0.000068947564,0.00028656787,0.00029368277,0.00009205654],"domain_scores_gemma":[0.997675,0.0010199937,0.00025904996,0.0003209309,0.00062572106,0.00009930405],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015577719,0.00070764235,0.0006681478,0.001413336,0.00035389327,0.0009874433,0.0008711718,0.0011206826,0.0021121528],"category_scores_gemma":[0.00706861,0.00026211783,0.00052976015,0.0007322082,0.00051527086,0.0011153745,0.00097663,0.001112884,0.0007060516],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035836967,0.00024421982,0.0051465114,0.00017622548,0.00008435601,0.00015284447,0.0003714255,0.093804345,0.022753386,0.006448385,0.00570506,0.86475486],"study_design_scores_gemma":[0.000019121984,0.00011618101,0.0019314123,0.000032336935,0.000017731587,0.00012284392,0.00009290085,0.9833928,0.0055958675,0.00708871,0.001570728,0.000019357667],"about_ca_topic_score_codex":0.003749777,"about_ca_topic_score_gemma":0.005611276,"teacher_disagreement_score":0.003749777,"about_ca_system_score_codex":0.0006173636,"about_ca_system_score_gemma":0.001098247,"threshold_uncertainty_score":0.008238375},"labels":[],"label_agreement":null},{"id":"W4226288657","doi":"10.18653/v1/2022.naacl-main.57","title":"NeuroLogic A*esque Decoding: Constrained Text Generation with Lookahead Heuristics","year":2022,"lang":"en","type":"article","venue":"Proceedings of the 2022 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":46,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Naval Information Warfare Center Pacific; Natural Sciences and Engineering Research Council of Canada; Defense Advanced Research Projects Agency; Allen Institute for Artificial Intelligence","keywords":"Rowan; Heuristics; Computer science; Decoding methods; Linguistics; Artificial intelligence; History; Philosophy; Algorithm","score_opus":0.0168325293327092,"score_gpt":0.24577551111509302,"score_spread":0.22894298178238381,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4226288657","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03138658,0.0011360964,0.9130712,0.0011638359,0.0007189371,0.0004602013,0.0013988604,0.030389544,0.02027478],"genre_scores_gemma":[0.33851078,0.00038466358,0.63855004,0.0006209728,0.00024620793,0.0003925116,0.0036044128,0.0034312462,0.014259108],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99837315,0.00063968997,0.00013795453,0.0003508277,0.00030157913,0.00019679398],"domain_scores_gemma":[0.99257433,0.005072385,0.00018456182,0.0011016995,0.0008841445,0.0001829236],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015610757,0.0015848413,0.0014969842,0.0017412306,0.0015067058,0.0028623708,0.0034505087,0.0024606918,0.023271365],"category_scores_gemma":[0.010303383,0.00088760874,0.0011933658,0.0024178927,0.0011480646,0.0044393823,0.0032219007,0.0016893897,0.007687749],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013455935,0.00045236127,0.0013017583,0.00048575774,0.00012618634,0.00060146296,0.0006653996,0.08085893,0.010481211,0.059128996,0.09311789,0.7514343],"study_design_scores_gemma":[0.00027550387,0.0000864175,0.00013283119,0.000039867944,0.000052712083,0.00016031456,0.00019792431,0.8922511,0.009622971,0.086176075,0.010958141,0.00004620345],"about_ca_topic_score_codex":0.0067587914,"about_ca_topic_score_gemma":0.010368142,"teacher_disagreement_score":0.023271365,"about_ca_system_score_codex":0.0009493202,"about_ca_system_score_gemma":0.0023984725,"threshold_uncertainty_score":0.0778504},"labels":[],"label_agreement":null},{"id":"W4229014795","doi":"10.18653/v1/2022.naacl-main.223","title":"A Few Thousand Translations Go a Long Way! Leveraging Pre-trained Models for African News Translation","year":2022,"lang":"en","type":"article","venue":"Proceedings of the 2022 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":45,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"International Development Research Centre; Rockefeller Foundation","keywords":"Blessing; Art history; Art; Theology; Philosophy; Classics","score_opus":0.026496001429269347,"score_gpt":0.2694050153342648,"score_spread":0.24290901390499547,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4229014795","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12235861,0.045244027,0.6476969,0.026048306,0.01649205,0.00043801562,0.012898765,0.086038515,0.04278486],"genre_scores_gemma":[0.46219295,0.01227112,0.38189152,0.006071209,0.0053033647,0.00047484887,0.054325875,0.008704035,0.06876514],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9977717,0.0009298298,0.00017109078,0.00059250504,0.00036217363,0.00017278675],"domain_scores_gemma":[0.9934442,0.0034344904,0.00019461766,0.0016063587,0.0011017275,0.00021869999],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004118209,0.0027469976,0.0020782077,0.0020876732,0.00150701,0.0042496147,0.0020350048,0.002703636,0.014453814],"category_scores_gemma":[0.015505927,0.0016223877,0.0017589243,0.0027498717,0.0009019708,0.009641599,0.0027465848,0.0043391567,0.028482245],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011906017,0.00027156825,0.00258473,0.0005194543,0.0005409976,0.00056754594,0.0003813801,0.05199802,0.0065444736,0.004243212,0.17423432,0.7569237],"study_design_scores_gemma":[0.00025131612,0.00041009017,0.0020915894,0.00039468633,0.00046599912,0.000664329,0.00069324655,0.8371603,0.012428362,0.04419842,0.101075985,0.00016572309],"about_ca_topic_score_codex":0.0067928396,"about_ca_topic_score_gemma":0.011769193,"teacher_disagreement_score":0.014453814,"about_ca_system_score_codex":0.0009443113,"about_ca_system_score_gemma":0.0012725634,"threshold_uncertainty_score":0.048352897},"labels":[],"label_agreement":null},{"id":"W4229647604","doi":"10.1007/978-3-662-44185-5_101152","title":"Translation Machinery","year":2015,"lang":"en","type":"book-chapter","venue":"Encyclopedia of Astrobiology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Translation (biology); Computer science; Natural language processing; Biology","score_opus":0.019885312998963046,"score_gpt":0.26105132942795406,"score_spread":0.24116601642899102,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4229647604","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0014929149,0.0041610394,0.47371057,0.0028081296,0.0023693189,0.00031532464,0.010244847,0.030194655,0.47470313],"genre_scores_gemma":[0.035408925,0.008742245,0.33450016,0.0019281568,0.0018692097,0.0006152028,0.051019233,0.014078541,0.5518384],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99878424,0.00020462822,0.00012084735,0.00040906854,0.00039340413,0.00008781089],"domain_scores_gemma":[0.9987845,0.000283376,0.000041463707,0.00054328673,0.00031295363,0.0000344811],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00082376285,0.001619935,0.0012313824,0.0031866976,0.0015687443,0.0056944513,0.0020057973,0.0013932714,0.21473819],"category_scores_gemma":[0.0032519395,0.0010138646,0.0014045688,0.0037621881,0.0012780513,0.007114938,0.0030518654,0.0024196485,0.25543275],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000047435777,0.000040704428,0.00009069338,0.00063174404,0.000023306717,0.00012308858,0.00023905656,0.00050938065,0.00423046,0.22861835,0.3271178,0.43832794],"study_design_scores_gemma":[0.000013193616,0.0000115627245,0.00010128529,0.0001239321,0.00001844586,0.00026127335,0.000056902547,0.00178835,0.00519591,0.115086176,0.877324,0.000019026475],"about_ca_topic_score_codex":0.0009826696,"about_ca_topic_score_gemma":0.0010997767,"teacher_disagreement_score":0.21473819,"about_ca_system_score_codex":0.0009000182,"about_ca_system_score_gemma":0.0016548171,"threshold_uncertainty_score":0.7183708},"labels":[],"label_agreement":null},{"id":"W4229727521","doi":"10.1145/564405.564409","title":"Resolving query translation ambiguity using a decaying co-occurrence model and syntactic dependence relations","year":2002,"lang":"en","type":"article","venue":"Proceedings of the 25th annual international ACM SIGIR conference on Research and development in information retrieval - SIGIR '02","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Translation (biology); Selection (genetic algorithm); Natural language processing; Artificial intelligence; Ambiguity; Word (group theory); Mutual information; Factor (programming language); Information retrieval; Mathematics; Programming language","score_opus":0.13103869841814536,"score_gpt":0.3638350060257669,"score_spread":0.23279630760762154,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4229727521","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.035877917,0.00052740704,0.96051645,0.00034679825,0.000041286163,0.00010870838,0.000116996925,0.00092363183,0.0015407942],"genre_scores_gemma":[0.5588571,0.00092807395,0.4344401,0.000349195,0.00018868981,0.00035871353,0.00064949464,0.0005432543,0.003685392],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9940625,0.0028388787,0.0005345459,0.0010839482,0.0012290938,0.00025106652],"domain_scores_gemma":[0.97938114,0.014278572,0.0013296158,0.0026497145,0.0020527923,0.00030815488],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0066705775,0.0011072151,0.0019060032,0.0029552602,0.0011740341,0.0021116885,0.0020216943,0.0017901617,0.0017071287],"category_scores_gemma":[0.026391756,0.0011241249,0.0015477939,0.0047645676,0.0015062102,0.006186619,0.0023333249,0.002420772,0.0013981356],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020160403,0.00076510105,0.0137437135,0.00085373136,0.00051198853,0.0014303488,0.0023877248,0.26413625,0.040548954,0.092952676,0.008588663,0.5720649],"study_design_scores_gemma":[0.000036506346,0.000077541,0.00078364,0.000015628477,0.000052821517,0.00034708244,0.00006511474,0.97490203,0.0028555356,0.019211192,0.0016076918,0.00004509631],"about_ca_topic_score_codex":0.008795406,"about_ca_topic_score_gemma":0.01025999,"teacher_disagreement_score":0.008795406,"about_ca_system_score_codex":0.0013428886,"about_ca_system_score_gemma":0.0022175456,"threshold_uncertainty_score":0.035277843},"labels":[],"label_agreement":null},{"id":"W4230719382","doi":"10.1145/1109557.1109603","title":"Implicit dictionaries with <i>O</i>(1) modifications per update and fast search","year":2006,"lang":"en","type":"article","venue":"Proceedings of the seventeenth annual ACM-SIAM symposium on Discrete algorithm - SODA '06","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Conjecture; Constant (computer programming); Computer science; Set (abstract data type); Order (exchange); Combinatorics; Binary logarithm; Search cost; Search problem; Mathematics; Discrete mathematics; Theoretical computer science; Algorithm; Programming language","score_opus":0.00571366836197134,"score_gpt":0.23407605207875373,"score_spread":0.22836238371678239,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4230719382","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1613961,0.0017960082,0.8135396,0.0019831797,0.00030702687,0.00029511593,0.00073298824,0.0035577565,0.01639222],"genre_scores_gemma":[0.45317608,0.00081249006,0.52637136,0.0006517754,0.0004369874,0.00044868593,0.0011892389,0.00081196555,0.016101316],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9974408,0.0004207319,0.00029084107,0.00053252134,0.000878884,0.00043618586],"domain_scores_gemma":[0.9824473,0.006781947,0.0020928164,0.0074220416,0.00090398197,0.0003519097],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012875594,0.00082500355,0.0016181745,0.0005995124,0.00097066705,0.002829825,0.0033820856,0.0018410209,0.0059184534],"category_scores_gemma":[0.01587225,0.0011187451,0.0007672643,0.0023874873,0.0023975554,0.015586398,0.003607992,0.002519149,0.0036497705],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0032706035,0.0006325871,0.006238894,0.0013407358,0.00012947855,0.0005587483,0.0014276849,0.12518811,0.046426788,0.22900768,0.03010303,0.5556756],"study_design_scores_gemma":[0.0007259132,0.0015928166,0.0029738636,0.00025488198,0.00017786335,0.0023578904,0.00061979133,0.60921407,0.05825653,0.27972242,0.043891527,0.00021244015],"about_ca_topic_score_codex":0.0012082164,"about_ca_topic_score_gemma":0.0022811193,"teacher_disagreement_score":0.0059184534,"about_ca_system_score_codex":0.0010403185,"about_ca_system_score_gemma":0.0018122102,"threshold_uncertainty_score":0.019799173},"labels":[],"label_agreement":null},{"id":"W4230872509","doi":"10.1016/b0-08-044854-2/05234-2","title":"Association for Computational Linguistics","year":2006,"lang":"en","type":"book-chapter","venue":"Elsevier eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1154,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Association (psychology); Linguistics; Applied linguistics; Computer science; Computational linguistics; Natural language processing; Philosophy; Epistemology","score_opus":0.012214444710845887,"score_gpt":0.2608096693829754,"score_spread":0.24859522467212952,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4230872509","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.001081613,0.019698804,0.09998356,0.013604961,0.015418985,0.00029927745,0.015090834,0.009328762,0.8254933],"genre_scores_gemma":[0.0096198665,0.020659084,0.05020132,0.0037820803,0.0030846635,0.0006778912,0.022837717,0.005694959,0.88344246],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9985436,0.0002987566,0.00021400914,0.00045044356,0.0004201784,0.0000729729],"domain_scores_gemma":[0.9949856,0.0014932349,0.0002463935,0.0016275081,0.0014056837,0.00024165616],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.0021189635,0.0018430752,0.0028695099,0.0053974255,0.0017134331,0.00550085,0.0018451491,0.0020116973,0.3387558],"category_scores_gemma":[0.00991881,0.0015023141,0.0011124129,0.008931608,0.0018561622,0.01215914,0.0042496608,0.0041300515,0.42528135],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000016784887,0.00002483557,0.00024179148,0.0005518713,0.0000197119,0.000082702136,0.0002255585,0.00014055769,0.0002959945,0.057129208,0.69408673,0.2471843],"study_design_scores_gemma":[0.000007381017,0.0000043310556,0.00018407479,0.00035662533,0.000016991069,0.00013439421,0.000059871,0.00031027256,0.00011475663,0.026112555,0.9726907,0.000008144353],"about_ca_topic_score_codex":0.0033244907,"about_ca_topic_score_gemma":0.0036797444,"teacher_disagreement_score":0.6612442,"about_ca_system_score_codex":0.0008518009,"about_ca_system_score_gemma":0.0031354483,"threshold_uncertainty_score":0.9431846},"labels":[],"label_agreement":null},{"id":"W4230939024","doi":"10.22215/etd/2015-10720","title":"Enhancing Machine Translation for English-Japanese Using Syntactic Pattern Recognition Methods","year":2015,"lang":"en","type":"dissertation","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Machine translation; Sentence; Natural language processing; Artificial intelligence; String (physics); Set (abstract data type); Transfer-based machine translation; Translation (biology); Example-based machine translation; Matching (statistics); Representation (politics); String searching algorithm; Speech recognition; Pattern matching; Mathematics; Programming language","score_opus":0.05884144728236558,"score_gpt":0.3947219384824856,"score_spread":0.33588049120012003,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4230939024","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.061276503,0.0011609211,0.9160062,0.0008519984,0.00036918846,0.00028585648,0.00032216794,0.005121484,0.014605788],"genre_scores_gemma":[0.18925136,0.0013587628,0.79659504,0.0003158768,0.0001707227,0.00019979826,0.0011163307,0.0006587759,0.010333258],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9991773,0.00029724362,0.000083888845,0.00016785234,0.0002118989,0.000061796025],"domain_scores_gemma":[0.9987381,0.00055043254,0.00009598944,0.00019603108,0.00039756444,0.0000219502],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011966428,0.0009139081,0.0006313533,0.0008230433,0.00071323884,0.001368776,0.0005000932,0.000591923,0.005385929],"category_scores_gemma":[0.0030596068,0.0002817014,0.0008546216,0.0011855221,0.0003731231,0.0019310546,0.0007222163,0.00092416105,0.0038353954],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017235213,0.000278255,0.0013743424,0.0009968202,0.00010083976,0.00042244824,0.0009122802,0.0069445693,0.157221,0.018440777,0.008273937,0.80486244],"study_design_scores_gemma":[0.00019788262,0.0009263203,0.0070025134,0.00023061624,0.00063438184,0.0021503423,0.0013559369,0.33397728,0.46358076,0.044118416,0.14563748,0.00018806521],"about_ca_topic_score_codex":0.0011085548,"about_ca_topic_score_gemma":0.0023171525,"teacher_disagreement_score":0.005385929,"about_ca_system_score_codex":0.00034642217,"about_ca_system_score_gemma":0.0007920842,"threshold_uncertainty_score":0.018017769},"labels":[],"label_agreement":null},{"id":"W4231138872","doi":"10.32920/14636088.v1","title":"Identifying emerging priorities in Knowledge Translation from the perspective of trainees","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; Dalhousie University","funders":"","keywords":"Knowledge translation; Perspective (graphical); Field (mathematics); Engineering ethics; Knowledge management; Medical education; Political science; Medicine; Computer science; Engineering","score_opus":0.06160024869133392,"score_gpt":0.34735914235194026,"score_spread":0.28575889366060636,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4231138872","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.46343714,0.014999411,0.10604549,0.2985907,0.0016365902,0.00039472192,0.00026159882,0.00016094858,0.11447338],"genre_scores_gemma":[0.95357966,0.0042992514,0.02949914,0.0056067323,0.00032097913,0.00020724899,0.00014589299,0.000086201166,0.0062548663],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9204648,0.0589722,0.0031315591,0.0029981225,0.0071687703,0.0072645713],"domain_scores_gemma":[0.8978422,0.063214414,0.007981301,0.002812371,0.014516716,0.013632909],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.07064019,0.00070205797,0.0007600163,0.0043857503,0.0105288625,0.025478452,0.0019144461,0.006490781,0.005138882],"category_scores_gemma":[0.07584296,0.0010324134,0.00059967133,0.003974451,0.014767144,0.025304869,0.016926033,0.010163007,0.0013753702],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00040399752,0.00020545087,0.020211652,0.0017064237,0.00009251295,0.0023240205,0.579889,0.00071543275,0.005123767,0.27710158,0.01015333,0.10207281],"study_design_scores_gemma":[0.000042810887,0.00014458987,0.0061220783,0.001365216,0.00004983172,0.0010877821,0.6452575,0.0023075722,0.002497086,0.2612208,0.0797983,0.000106341126],"about_ca_topic_score_codex":0.0031897482,"about_ca_topic_score_gemma":0.004336321,"teacher_disagreement_score":0.9293598,"about_ca_system_score_codex":0.010220157,"about_ca_system_score_gemma":0.021446716,"threshold_uncertainty_score":0.37358552},"labels":[],"label_agreement":null},{"id":"W4231360391","doi":"10.18653/v1/2021.naacl-tutorials","title":"Proceedings of the 2021 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies: Tutorials","year":2021,"lang":"en","type":"paratext","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada; Canada First Research Excellence Fund","keywords":"Computer science; Computational linguistics; Association (psychology); Language technology; Linguistics; Artificial intelligence; Natural language; Philosophy; Epistemology","score_opus":0.01656639919389856,"score_gpt":0.2864483349353311,"score_spread":0.26988193574143254,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4231360391","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0044967914,0.28977168,0.12685461,0.041487105,0.0983043,0.00044716627,0.005488117,0.0067127957,0.4264375],"genre_scores_gemma":[0.00848099,0.13574232,0.041772842,0.0046228147,0.018922213,0.000613175,0.009236098,0.0033305413,0.77727896],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9990201,0.00026882987,0.000077580284,0.00017655834,0.00034908848,0.00010797418],"domain_scores_gemma":[0.99627197,0.0014520048,0.00013663483,0.00029936575,0.0012048441,0.0006351856],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033887557,0.0015395193,0.0014924813,0.0035813404,0.0008926968,0.0043355036,0.0013461502,0.0016358537,0.15122916],"category_scores_gemma":[0.00468264,0.00066426717,0.0006173398,0.0037651397,0.0010702019,0.0056586284,0.0027121606,0.0028702277,0.1028306],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000031301966,0.000053270625,0.00012470047,0.00028443654,0.000014061405,0.00003819801,0.000058540038,0.0001973476,0.0005617682,0.0046850974,0.79325336,0.20069796],"study_design_scores_gemma":[0.0000037447269,0.000014221278,0.00030130017,0.00025046378,0.000007808664,0.00009129963,0.000044131895,0.00056439743,0.00015082308,0.00320289,0.99535996,0.000009009025],"about_ca_topic_score_codex":0.002384407,"about_ca_topic_score_gemma":0.0063268202,"teacher_disagreement_score":0.15122916,"about_ca_system_score_codex":0.0014181932,"about_ca_system_score_gemma":0.002541018,"threshold_uncertainty_score":0.50591195},"labels":[],"label_agreement":null},{"id":"W4231496788","doi":"10.1016/s1571-0661(05)82622-2","title":"Preface","year":2003,"lang":"en","type":"article","venue":"Electronic Notes in Theoretical Computer Science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Parsing; Gratitude; Programming language; Compiler; Software engineering","score_opus":0.005166389861819994,"score_gpt":0.2582236165892906,"score_spread":0.2530572267274706,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4231496788","genre_codex":"other","genre_gemma":"editorial","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"editorial","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.001325523,0.016947435,0.006973106,0.030048648,0.1272195,0.0005066626,0.014149016,0.0024729788,0.80035704],"genre_scores_gemma":[0.0057716127,0.009297148,0.0032951005,0.006260921,0.019231603,0.00029957574,0.015298642,0.0014936641,0.9390518],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9982955,0.00021868519,0.00013831515,0.00034691242,0.00086291216,0.00013767714],"domain_scores_gemma":[0.99549687,0.00081256893,0.00017822767,0.00053635327,0.0022824588,0.00069357664],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015173922,0.0011318463,0.0011254477,0.0031546855,0.0023417037,0.006894658,0.0018794056,0.0017472061,0.5294501],"category_scores_gemma":[0.009914666,0.00037771047,0.00085469044,0.0031794778,0.00079162617,0.0045679286,0.0030120597,0.002655117,0.377716],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000027984139,0.000017025033,0.00009945775,0.00016139969,0.0000033990143,0.000039697992,0.000109368055,0.000057867364,0.00012666412,0.004824889,0.94255465,0.051977627],"study_design_scores_gemma":[0.000002638131,0.000009633158,0.0001498816,0.00009460959,0.0000016165112,0.000048472313,0.00007082248,0.00002031503,0.00004747769,0.0012706816,0.9982803,0.000003661177],"about_ca_topic_score_codex":0.0020964413,"about_ca_topic_score_gemma":0.0022932452,"teacher_disagreement_score":0.5294501,"about_ca_system_score_codex":0.0021832066,"about_ca_system_score_gemma":0.0025083083,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W4232154163","doi":"10.3115/1610243.1610250","title":"TransType","year":2000,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Translation (biology); Embedding; Machine translation; Translation system; Natural language processing; Artificial intelligence; Chemistry","score_opus":0.007148780858881537,"score_gpt":0.24483461589041255,"score_spread":0.237685835031531,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4232154163","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011979947,0.0010159964,0.6094858,0.0027299302,0.0033743654,0.00056501955,0.020134293,0.13333884,0.21737579],"genre_scores_gemma":[0.14079383,0.0023159755,0.3432589,0.006788435,0.001518341,0.0009855984,0.05427809,0.12690794,0.323153],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99643886,0.0008136194,0.00036055953,0.00088134175,0.001210527,0.00029517617],"domain_scores_gemma":[0.9941636,0.0015351798,0.0002451434,0.0022234656,0.0016175561,0.00021513017],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027250242,0.0011845846,0.0009638384,0.0015721521,0.0014364135,0.003834177,0.002273067,0.0013804744,0.08748687],"category_scores_gemma":[0.006251557,0.0010074808,0.0013413336,0.0013616885,0.0010857034,0.007119595,0.0053799674,0.0025531093,0.06316903],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010271227,0.00020930044,0.004032739,0.0013744906,0.000126128,0.0015279219,0.002801042,0.0014415793,0.019627806,0.25665694,0.39797175,0.3132032],"study_design_scores_gemma":[0.000027596574,0.00005038396,0.00035748247,0.00007637119,0.000024104571,0.0007134341,0.00020966685,0.0018049548,0.011373389,0.01952223,0.96578896,0.000051384603],"about_ca_topic_score_codex":0.0017821963,"about_ca_topic_score_gemma":0.0031830342,"teacher_disagreement_score":0.08748687,"about_ca_system_score_codex":0.0011250885,"about_ca_system_score_gemma":0.0014308986,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W4232169493","doi":"10.22215/etd/2005-07595","title":"Word sense disambiguation and context","year":2005,"lang":"en","type":"dissertation","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University; Canadian Heritage","funders":"","keywords":"Word-sense disambiguation; Word (group theory); Context (archaeology); Linguistics; Computer science; Humanities; Natural language processing; Artificial intelligence; Philosophy; History","score_opus":0.009518472308773772,"score_gpt":0.2820678186368202,"score_spread":0.2725493463280464,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4232169493","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10739435,0.028447604,0.7335939,0.007274415,0.005584417,0.0009816732,0.01007092,0.01657692,0.09007583],"genre_scores_gemma":[0.5070529,0.0054608043,0.44838,0.0012760953,0.0011824152,0.0003828746,0.012587248,0.0011842168,0.022493483],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9962752,0.0012192475,0.0004347487,0.0010886187,0.00073760503,0.00024449266],"domain_scores_gemma":[0.99683607,0.0015096338,0.00030593429,0.00061295176,0.0005853352,0.0001499748],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020782899,0.0012728809,0.0012202436,0.0077457638,0.0021320414,0.0039301454,0.0013022288,0.001205182,0.01286405],"category_scores_gemma":[0.007912927,0.00082529016,0.00119719,0.006312154,0.0015733556,0.008180208,0.003953163,0.0014601543,0.0068160677],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007334561,0.00022499658,0.0058159847,0.0012842827,0.00027712897,0.0013467807,0.0014633981,0.0067919083,0.016396997,0.14094818,0.06989018,0.7548268],"study_design_scores_gemma":[0.00019705418,0.00017966934,0.007166604,0.0007889973,0.00033477525,0.0026865937,0.0024679117,0.08719627,0.036826506,0.6272336,0.23470193,0.00022012959],"about_ca_topic_score_codex":0.0031833823,"about_ca_topic_score_gemma":0.006755694,"teacher_disagreement_score":0.01286405,"about_ca_system_score_codex":0.00082607695,"about_ca_system_score_gemma":0.0021474191,"threshold_uncertainty_score":0.043034554},"labels":[],"label_agreement":null},{"id":"W4232285071","doi":"10.1007/978-0-387-39940-9_2509","title":"Document Segmentation","year":2009,"lang":"en","type":"book-chapter","venue":"Encyclopedia of Database Systems","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Artificial intelligence","score_opus":0.011352733433689489,"score_gpt":0.2582528176788796,"score_spread":0.2469000842451901,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4232285071","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011695073,0.009775627,0.5350114,0.0010019965,0.0014040255,0.00079064834,0.014304986,0.046843596,0.37917262],"genre_scores_gemma":[0.06264219,0.00635465,0.4969962,0.00077365583,0.0004765133,0.00034458548,0.040192872,0.005862041,0.3863573],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9996426,0.000019810755,0.000027411654,0.00015757313,0.000110650326,0.000041901072],"domain_scores_gemma":[0.9994343,0.000093295836,0.00002769691,0.0001575101,0.00023960618,0.000047671852],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000330467,0.001132494,0.0008158537,0.003049271,0.00094916654,0.0028845912,0.0012107828,0.00089130295,0.088919595],"category_scores_gemma":[0.00091560587,0.0005722138,0.00082049443,0.0035974476,0.00041201635,0.0020880438,0.0010721397,0.00083946117,0.09248455],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013179277,0.000049474067,0.00026612362,0.0004600186,0.000022036758,0.00012593708,0.000103185994,0.00067593215,0.04840949,0.0098334,0.107462406,0.83246017],"study_design_scores_gemma":[0.000028880871,0.00010749085,0.0018541996,0.00017142393,0.00008807761,0.0013488148,0.00016095866,0.010566029,0.11569522,0.014957717,0.8549691,0.000052109714],"about_ca_topic_score_codex":0.0017311547,"about_ca_topic_score_gemma":0.0026058943,"teacher_disagreement_score":0.088919595,"about_ca_system_score_codex":0.0006255804,"about_ca_system_score_gemma":0.00096769334,"threshold_uncertainty_score":0.29746568},"labels":[],"label_agreement":null},{"id":"W4232503947","doi":"10.18653/v1/w17-45","title":"Proceedings of the Workshop on New Frontiers in Summarization","year":2017,"lang":"en","type":"paratext","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Nvidia","keywords":"Automatic summarization; Computer science; Data science; Information retrieval","score_opus":0.015623304089776238,"score_gpt":0.2748374213215636,"score_spread":0.25921411723178733,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4232503947","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015262156,0.0635407,0.7783359,0.040812384,0.019519042,0.00080699555,0.011171316,0.016783595,0.053767998],"genre_scores_gemma":[0.115156986,0.031134186,0.5887847,0.0055094827,0.016577594,0.0011103501,0.058520947,0.006687264,0.17651853],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9956696,0.0014966403,0.00046483288,0.001049927,0.0010663067,0.0002526997],"domain_scores_gemma":[0.9909882,0.003789571,0.00026788056,0.0019313906,0.0024281582,0.0005947221],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00617442,0.0019873562,0.001683598,0.0033144935,0.00134602,0.007379788,0.0033270482,0.00212835,0.0505962],"category_scores_gemma":[0.015668685,0.00065796776,0.0020624981,0.0035650306,0.0013750743,0.010772031,0.003148265,0.0035147285,0.022547586],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029913252,0.00017357711,0.00046640786,0.0009127991,0.00013902332,0.00024476534,0.00084152445,0.003343472,0.0049928864,0.019268159,0.39103755,0.5782807],"study_design_scores_gemma":[0.00007153764,0.00014885087,0.0015644492,0.000453262,0.0001097635,0.00043836742,0.00089357264,0.04313589,0.0065815314,0.06588479,0.88063836,0.000079603196],"about_ca_topic_score_codex":0.003603562,"about_ca_topic_score_gemma":0.00448294,"teacher_disagreement_score":0.0505962,"about_ca_system_score_codex":0.0018184391,"about_ca_system_score_gemma":0.0017695978,"threshold_uncertainty_score":0.16926116},"labels":[],"label_agreement":null},{"id":"W4232882609","doi":"10.18653/v1/2020.nlp4convai-1","title":"Proceedings of the 2nd Workshop on Natural Language Processing for Conversational AI","year":2020,"lang":"en","type":"paratext","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Natural language processing; Natural (archaeology); Natural language; Artificial intelligence; History; Archaeology","score_opus":0.014343493025170994,"score_gpt":0.29315921341339535,"score_spread":0.27881572038822433,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4232882609","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0127548585,0.04932871,0.74834514,0.036320385,0.025610404,0.00090083294,0.0039052225,0.009405503,0.1134289],"genre_scores_gemma":[0.100951225,0.033553597,0.5476885,0.0066965236,0.0107112,0.002027262,0.022221757,0.00487712,0.2712729],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9968478,0.0015127768,0.0002805608,0.00059363485,0.0005814952,0.00018376426],"domain_scores_gemma":[0.99402535,0.0031989464,0.000118742755,0.001123009,0.0011030777,0.00043081495],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.006765828,0.0016566687,0.0014824917,0.0014807858,0.0011811593,0.007064993,0.0030622636,0.0024378095,0.046369202],"category_scores_gemma":[0.009329344,0.0006806351,0.0015378578,0.0012874998,0.002243074,0.008145813,0.0033928468,0.00475381,0.01766686],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00054595,0.00037708887,0.00061018096,0.0009393313,0.0001851695,0.0005253461,0.0011648046,0.0037123472,0.0070444746,0.03311444,0.46811995,0.48366085],"study_design_scores_gemma":[0.000076981116,0.00014129723,0.0012195605,0.0005354481,0.00008823126,0.00049877795,0.0005877985,0.037172876,0.0044841515,0.063394,0.8917372,0.00006365407],"about_ca_topic_score_codex":0.004045942,"about_ca_topic_score_gemma":0.0053080623,"teacher_disagreement_score":0.9536308,"about_ca_system_score_codex":0.0017363507,"about_ca_system_score_gemma":0.0021082598,"threshold_uncertainty_score":0.15512043},"labels":[],"label_agreement":null},{"id":"W4233012245","doi":"10.1007/978-0-387-39940-9_2039","title":"Anchor Text Surrogate","year":2009,"lang":"en","type":"book-chapter","venue":"Encyclopedia of Database Systems","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Surrogate endpoint; Artificial intelligence; Natural language processing; Medicine; Internal medicine","score_opus":0.013395470133698326,"score_gpt":0.24779828594421904,"score_spread":0.23440281581052072,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4233012245","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0034003933,0.001556204,0.25784844,0.0020625922,0.004000884,0.0005681121,0.030335285,0.030764865,0.6694633],"genre_scores_gemma":[0.044934493,0.0016630088,0.11815232,0.0010561476,0.0006100342,0.00039981102,0.049482495,0.01043874,0.773263],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9990464,0.00014345712,0.00008580491,0.00021385404,0.00043637064,0.00007422242],"domain_scores_gemma":[0.99832696,0.00025774835,0.00007516752,0.0006919469,0.0005541573,0.000093995426],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.00064792327,0.0012699141,0.0009842723,0.002125559,0.00087757065,0.003464487,0.0018914955,0.001552824,0.35035965],"category_scores_gemma":[0.0051530236,0.00048622317,0.0006598627,0.0024141977,0.0005351699,0.004567707,0.0034012934,0.001610993,0.33735225],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037601643,0.000096880416,0.00016225733,0.00062290777,0.000017056856,0.0003576011,0.00012957717,0.0014420797,0.010200815,0.070803374,0.5181983,0.39759308],"study_design_scores_gemma":[0.000036378126,0.00003608842,0.00010194152,0.00012237743,0.0000145001395,0.00033285088,0.000055874265,0.0043053497,0.009826775,0.029607762,0.9555412,0.000018868906],"about_ca_topic_score_codex":0.0007397943,"about_ca_topic_score_gemma":0.00090841093,"teacher_disagreement_score":0.35035965,"about_ca_system_score_codex":0.00057591475,"about_ca_system_score_gemma":0.00089773146,"threshold_uncertainty_score":0.9266331},"labels":[],"label_agreement":null},{"id":"W4233783659","doi":"10.26686/wgtn.12552221.v1","title":"How large a vocabulary is needed for reading and listening?","year":2020,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":54,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"Victoria University; Victoria University of Wellington","keywords":"Vocabulary; Linguistics; Reading comprehension; Computer science; Active listening; Word (group theory); Comprehension; Reading (process); Natural language processing; Artificial intelligence; Psychology; Communication","score_opus":0.022204957687457658,"score_gpt":0.28395266465346125,"score_spread":0.2617477069660036,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4233783659","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.60085535,0.0488521,0.096353225,0.039210513,0.001270941,0.00034018894,0.0024514354,0.0017202081,0.208946],"genre_scores_gemma":[0.9524684,0.010255635,0.026673634,0.0013379862,0.0005124694,0.00024673922,0.001580818,0.0006277029,0.006296728],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99701464,0.0011982261,0.00028608795,0.0005191338,0.0006908004,0.00029118184],"domain_scores_gemma":[0.9856455,0.009140452,0.001017882,0.0013838864,0.0022143512,0.0005979704],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027394018,0.00041333126,0.00079285714,0.0018481603,0.0011147392,0.003258405,0.0009797818,0.0011868265,0.009189421],"category_scores_gemma":[0.03135994,0.00043223434,0.00035995277,0.0014536157,0.0033319453,0.013133107,0.0015882901,0.0012760594,0.0039542895],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006980042,0.0001103625,0.022005223,0.0024379126,0.0001755759,0.0011143906,0.019801348,0.0011819083,0.05311068,0.096775696,0.017362516,0.78522646],"study_design_scores_gemma":[0.00025267235,0.0010018587,0.14897947,0.0020425,0.000481703,0.0062860795,0.045510672,0.0041973894,0.030778926,0.48444998,0.27564287,0.00037591136],"about_ca_topic_score_codex":0.007721997,"about_ca_topic_score_gemma":0.0070709763,"teacher_disagreement_score":0.009189421,"about_ca_system_score_codex":0.0011807358,"about_ca_system_score_gemma":0.0018990528,"threshold_uncertainty_score":0.030741692},"labels":[],"label_agreement":null},{"id":"W4233966082","doi":"10.1162/coli_x_00181","title":"Publications Received","year":2014,"lang":"en","type":"article","venue":"Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science","score_opus":0.017556002457512956,"score_gpt":0.29004651036624157,"score_spread":0.27249050790872864,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4233966082","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00045888816,0.008215106,0.0013064935,0.0067630257,0.02256957,0.00021713944,0.008018347,0.002477618,0.9499738],"genre_scores_gemma":[0.0013914177,0.005057682,0.000822869,0.0025378538,0.002687374,0.000082669525,0.005900228,0.001009369,0.98051065],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99638027,0.00029511208,0.00030640894,0.00069769943,0.0020298732,0.00029063935],"domain_scores_gemma":[0.9932318,0.00053080294,0.00032220932,0.0009297899,0.003795922,0.0011895744],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.0016194746,0.0017194542,0.0018546953,0.0045630177,0.0022040284,0.016587857,0.0024291754,0.0033611518,0.80939347],"category_scores_gemma":[0.009485399,0.00078633183,0.0012426996,0.0060713496,0.00090726354,0.007937112,0.004443865,0.003446402,0.84654695],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000021526443,0.000025216066,0.00010540396,0.00034318995,0.0000062476734,0.00006714858,0.000046873687,0.000040097395,0.0001880506,0.0039404603,0.9077531,0.08746271],"study_design_scores_gemma":[0.0000033479844,0.000007955638,0.000109886314,0.00011400713,0.000002045973,0.00006638616,0.00002754459,0.000011823868,0.00004473978,0.0006740009,0.9989334,0.0000049127307],"about_ca_topic_score_codex":0.0015038176,"about_ca_topic_score_gemma":0.0025223177,"teacher_disagreement_score":0.19060653,"about_ca_system_score_codex":0.0021643415,"about_ca_system_score_gemma":0.0037996338,"threshold_uncertainty_score":0.2718771},"labels":[],"label_agreement":null},{"id":"W4234452932","doi":"10.1007/978-0-387-39940-9_3321","title":"Prefix Tree","year":2009,"lang":"en","type":"book-chapter","venue":"Encyclopedia of Database Systems","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Prefix; Computer science; Tree (set theory); Mathematics; Combinatorics; Philosophy; Linguistics","score_opus":0.013870868320175846,"score_gpt":0.24566577972515605,"score_spread":0.2317949114049802,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4234452932","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008151136,0.0032053657,0.537091,0.0019030117,0.00095265853,0.00037892946,0.01815396,0.013151176,0.41701272],"genre_scores_gemma":[0.08587208,0.006114597,0.55129266,0.0013170681,0.0005491754,0.00043952078,0.05136645,0.0047975276,0.29825094],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9993825,0.00006613266,0.000057432564,0.00017302824,0.0002583996,0.000062484425],"domain_scores_gemma":[0.99915814,0.00019514898,0.00004073828,0.00030979357,0.0002391599,0.000056991033],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00041205602,0.0006356033,0.00083428225,0.0017685866,0.001259933,0.0032858849,0.0012859807,0.0009471749,0.08775509],"category_scores_gemma":[0.0023565218,0.00056514883,0.0006976846,0.0043877806,0.00072059385,0.006630523,0.0020404274,0.0013361031,0.05741277],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010772103,0.00007431798,0.00032672758,0.00039305977,0.000021744801,0.00014565178,0.00021030077,0.0013777073,0.0039734123,0.36767915,0.16986063,0.4558296],"study_design_scores_gemma":[0.00001896283,0.00003529113,0.00018217959,0.00011286777,0.00002695835,0.0004550845,0.00007984594,0.00406164,0.0036003783,0.24639168,0.74501234,0.000022784667],"about_ca_topic_score_codex":0.0011273178,"about_ca_topic_score_gemma":0.0020075429,"teacher_disagreement_score":0.08775509,"about_ca_system_score_codex":0.00074135995,"about_ca_system_score_gemma":0.0014558232,"threshold_uncertainty_score":0.29356998},"labels":[],"label_agreement":null},{"id":"W4234619290","doi":"10.18653/v1/w18-16","title":"Proceedings of the Second Workshop on Stylistic Variation","year":2018,"lang":"en","type":"paratext","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Thomson Reuters (Canada)","funders":"","keywords":"Variation (astronomy); Computer science; Natural language processing; Physics; Astronomy","score_opus":0.013931209323150538,"score_gpt":0.271418391439397,"score_spread":0.25748718211624644,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4234619290","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.031621926,0.045319173,0.6280839,0.055934366,0.029668765,0.0006885568,0.010256164,0.0077406033,0.1906866],"genre_scores_gemma":[0.2727172,0.026845263,0.30829665,0.007358747,0.012107473,0.0011705814,0.03855867,0.008352124,0.3245932],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.995833,0.0014174326,0.00030092633,0.0010541197,0.0011092273,0.00028533168],"domain_scores_gemma":[0.9897126,0.0042493814,0.00020935813,0.0029674347,0.0021997646,0.0006615135],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006495842,0.0010987091,0.0013310341,0.0032285983,0.0019674234,0.007519273,0.0025753544,0.0019840307,0.04088889],"category_scores_gemma":[0.015799785,0.00067274633,0.0016344673,0.0036395176,0.0021097572,0.007677135,0.00551132,0.0034157627,0.014489689],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003591143,0.00015295141,0.0019386063,0.00062110304,0.00011015501,0.00046363854,0.0021462669,0.004133692,0.005405524,0.07888957,0.36650616,0.5392732],"study_design_scores_gemma":[0.00003528195,0.000042432526,0.0020570874,0.00029519136,0.0000549026,0.00039472463,0.0005372695,0.008475186,0.0032545615,0.08244764,0.90235114,0.000054593405],"about_ca_topic_score_codex":0.003334589,"about_ca_topic_score_gemma":0.005448673,"teacher_disagreement_score":0.04088889,"about_ca_system_score_codex":0.0020294327,"about_ca_system_score_gemma":0.002120619,"threshold_uncertainty_score":0.13678694},"labels":[],"label_agreement":null},{"id":"W4235954735","doi":"10.1145/1871840","title":"Proceedings of the fourth workshop on Analytics for noisy unstructured text data","year":2010,"lang":"en","type":"paratext","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Listing (finance); Analytics; Key (lock); Information retrieval; Data science; Library science; World Wide Web","score_opus":0.033355835646512524,"score_gpt":0.3122525185803998,"score_spread":0.2788966829338873,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4235954735","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014919176,0.04694668,0.7875031,0.05286435,0.029727735,0.001424769,0.020124558,0.014091864,0.032397628],"genre_scores_gemma":[0.07568079,0.032917485,0.6135943,0.011079605,0.013465781,0.0018465733,0.09065432,0.0071310564,0.15363015],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.98944026,0.0038519227,0.00084827846,0.0015915177,0.003786361,0.0004817152],"domain_scores_gemma":[0.97410756,0.0119284205,0.0008028619,0.0050081625,0.0060969857,0.0020559507],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.014213865,0.0018197248,0.0027142793,0.003986183,0.001308354,0.011630669,0.0034036087,0.002313707,0.023250794],"category_scores_gemma":[0.033752445,0.0011549493,0.0019555376,0.004872351,0.0019739894,0.010310922,0.004706786,0.0055725803,0.013954936],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003347341,0.00022072456,0.0014623359,0.000673934,0.00021726814,0.00037607434,0.00077167095,0.0039477404,0.002795071,0.008430549,0.70244706,0.27832285],"study_design_scores_gemma":[0.0000867078,0.00015093452,0.004843814,0.0008727805,0.00013012944,0.00078655605,0.0014851276,0.037059445,0.004946124,0.058443867,0.8910312,0.00016322434],"about_ca_topic_score_codex":0.004673817,"about_ca_topic_score_gemma":0.007308844,"teacher_disagreement_score":0.023250794,"about_ca_system_score_codex":0.0018005866,"about_ca_system_score_gemma":0.0035809302,"threshold_uncertainty_score":0.07778162},"labels":[],"label_agreement":null},{"id":"W4236158887","doi":"10.31234/osf.io/hdftz","title":"Estonian case inflection made simple. A case study in Word and Paradigm morphology with Linear Discriminative Learning.","year":2019,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Inflection; Computer science; Morpheme; Comprehension; Linguistics; Artificial intelligence; Natural language processing; Discriminative model; Estonian; Analogy; Noun","score_opus":0.024964181472817037,"score_gpt":0.3216952408774313,"score_spread":0.29673105940461425,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4236158887","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9491919,0.0008682337,0.0053848606,0.0011723491,0.00012259836,0.000050913863,0.0011322412,0.00012519573,0.041951697],"genre_scores_gemma":[0.98708576,0.00024137482,0.0025623692,0.00021598004,0.00002838737,0.0000239795,0.00069841254,0.000063375686,0.009080469],"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99911255,0.000262947,0.00013923227,0.00017884941,0.00020571722,0.0001007323],"domain_scores_gemma":[0.99683565,0.0015929423,0.00038422985,0.0007673436,0.00032773995,0.000092216826],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012448402,0.00073744473,0.00039204455,0.0010340706,0.0017654123,0.0016714846,0.00087984896,0.0012927805,0.0048597874],"category_scores_gemma":[0.0066355094,0.00036598774,0.0007278572,0.0011302135,0.002532649,0.0018709751,0.0016613695,0.0013487886,0.0013268491],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014640555,0.0008950646,0.22963127,0.0008316803,0.00021090994,0.22271639,0.18789126,0.008140716,0.011376057,0.11493203,0.032660812,0.1892497],"study_design_scores_gemma":[0.00019591679,0.0008166478,0.42019436,0.00071503705,0.00024961564,0.24791534,0.05056313,0.023690859,0.019462608,0.048149884,0.18772918,0.00031743015],"about_ca_topic_score_codex":0.017466731,"about_ca_topic_score_gemma":0.026989993,"teacher_disagreement_score":0.017466731,"about_ca_system_score_codex":0.0023334357,"about_ca_system_score_gemma":0.0007063895,"threshold_uncertainty_score":0.034730136},"labels":[],"label_agreement":null},{"id":"W4236758004","doi":"10.1145/2766462.2767827","title":"Using Term Location Information to Enhance Probabilistic Information Retrieval","year":2015,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Probabilistic logic; Term (time); Divergence-from-randomness model; Term Discrimination; Information retrieval; Kernel (algebra); Artificial intelligence; Vector space model; Search engine; Concept search; Web search query; Mathematics","score_opus":0.024342774849127155,"score_gpt":0.3131374087954922,"score_spread":0.288794633946365,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4236758004","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.31411746,0.0075259637,0.6647122,0.0013053629,0.00029585036,0.00034766598,0.00062740385,0.005209365,0.0058587175],"genre_scores_gemma":[0.86265117,0.0012225395,0.13153246,0.00024527812,0.0002112942,0.00014584883,0.000619228,0.00016758591,0.00320461],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99829906,0.0006279896,0.00016826359,0.000333073,0.00044703673,0.00012466806],"domain_scores_gemma":[0.9945374,0.003063288,0.00055731507,0.0007142042,0.00097361853,0.0001541561],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032510601,0.00096053135,0.0013560756,0.0022766623,0.0005899306,0.0012399649,0.0013792817,0.0011771667,0.0016865652],"category_scores_gemma":[0.014565109,0.0003329415,0.0008361625,0.0023837755,0.0006737004,0.004292965,0.0011990971,0.0010823089,0.0014852895],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0024169108,0.0009653262,0.012341114,0.000999173,0.000336908,0.0002920112,0.00048486766,0.143709,0.09127513,0.00887259,0.0075565684,0.73075044],"study_design_scores_gemma":[0.00020562869,0.0013961942,0.009356327,0.000058142617,0.00041372486,0.00053585897,0.000116877665,0.9326001,0.03817743,0.010512672,0.006456559,0.00017039789],"about_ca_topic_score_codex":0.0046100537,"about_ca_topic_score_gemma":0.0074044405,"teacher_disagreement_score":0.0046100537,"about_ca_system_score_codex":0.00124351,"about_ca_system_score_gemma":0.0013022033,"threshold_uncertainty_score":0.017193437},"labels":[],"label_agreement":null},{"id":"W4236789232","doi":"10.1007/978-3-319-41337-2_10","title":"Relatedness","year":2016,"lang":"en","type":"book-chapter","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Computer Research Institute of Montréal","funders":"","keywords":"Natural language processing; Word (group theory); Similarity (geometry); Artificial intelligence; Computer science; Distributional semantics; Semantics (computer science); Context (archaeology); Semantic similarity; Field (mathematics); Linguistics; Mathematics; Geography","score_opus":0.012965842339000831,"score_gpt":0.241289666180245,"score_spread":0.22832382384124417,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4236789232","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0031324066,0.0017424978,0.11138289,0.0016244716,0.00082526426,0.00022394555,0.0018203355,0.0024052055,0.87684304],"genre_scores_gemma":[0.06690489,0.003387037,0.07579205,0.0013922262,0.0008511612,0.00033861728,0.007760942,0.0016824328,0.8418907],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9991523,0.00014207549,0.000036173085,0.00028694177,0.00031278806,0.000069675465],"domain_scores_gemma":[0.9992539,0.00014192077,0.00003940757,0.00029588307,0.00020030975,0.00006841309],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007210242,0.0008581369,0.00048413765,0.0022794574,0.001674326,0.0027207085,0.0014270112,0.0010977377,0.10927084],"category_scores_gemma":[0.0032483803,0.00040982856,0.0006083345,0.0018952631,0.0010485197,0.005553379,0.0028536466,0.0017358036,0.09314223],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00004029832,0.00005914454,0.00031283376,0.00024178991,0.000016239503,0.00009090898,0.0003716715,0.00058879284,0.0050548525,0.45442948,0.17052464,0.3682693],"study_design_scores_gemma":[0.000006401587,0.000023745099,0.00041162333,0.00007978245,0.000016862623,0.00038874592,0.00011575786,0.0009375225,0.0027199509,0.15825622,0.8370249,0.000018469313],"about_ca_topic_score_codex":0.0011591851,"about_ca_topic_score_gemma":0.0015296225,"teacher_disagreement_score":0.10927084,"about_ca_system_score_codex":0.0008770211,"about_ca_system_score_gemma":0.0011007977,"threshold_uncertainty_score":0.36554736},"labels":[],"label_agreement":null},{"id":"W4237813545","doi":"10.18653/v1/w18-37","title":"Proceedings of the 5th Workshop on Natural Language Processing Techniques for Educational Applications","year":2018,"lang":"en","type":"paratext","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Office of the Chief Medical Examiner","funders":"Advanced Distributed Learning Initiative; University of Memphis; U.S. Department of Defense","keywords":"Computer science; Natural language processing; Artificial intelligence; Cognitive science; Psychology","score_opus":0.013936412024985592,"score_gpt":0.33213827302408017,"score_spread":0.3182018609990946,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4237813545","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016173482,0.037710756,0.8055303,0.016276488,0.014706027,0.0011187119,0.002574408,0.010621809,0.0952881],"genre_scores_gemma":[0.06464975,0.027970532,0.5969694,0.0031437138,0.0064689885,0.001771896,0.016076853,0.005346735,0.27760205],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99589765,0.0018530459,0.00040887707,0.0006561856,0.00093849597,0.00024575513],"domain_scores_gemma":[0.9922725,0.0037565306,0.00020221321,0.0016711636,0.0015073284,0.00059037737],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0075390944,0.0016913586,0.0014932163,0.002511153,0.000981084,0.0061341194,0.0029852632,0.0025767377,0.063156635],"category_scores_gemma":[0.010571201,0.0007537338,0.0017751771,0.0017997632,0.0016374447,0.00695126,0.0034388308,0.0041457713,0.02767614],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005517236,0.00067718927,0.0007054815,0.00093464146,0.00013754747,0.00065104564,0.0012419524,0.0019958736,0.012832502,0.02222898,0.2638681,0.694175],"study_design_scores_gemma":[0.0000856057,0.0001941263,0.0014498485,0.00053158985,0.000088063054,0.00070838747,0.0004154922,0.018721472,0.008666227,0.029895267,0.939193,0.000050918017],"about_ca_topic_score_codex":0.001841184,"about_ca_topic_score_gemma":0.0023204277,"teacher_disagreement_score":0.9368434,"about_ca_system_score_codex":0.0013269724,"about_ca_system_score_gemma":0.0018074636,"threshold_uncertainty_score":0.21127999},"labels":[],"label_agreement":null},{"id":"W4237884881","doi":"10.1353/cjl.2015.0024","title":"Evidence for a DP-projection in West Greenlandic Inuit","year":2015,"lang":"fr","type":"article","venue":"The Canadian Journal of Linguistics / La revue canadienne de linguistique","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Definiteness; Linguistics; Word order; Head (geology); Projection (relational algebra); Ergative case; Philosophy; Mathematics; Geology; Combinatorics","score_opus":0.08209641607453451,"score_gpt":0.32279989084798394,"score_spread":0.24070347477344944,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4237884881","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9863831,0.00012447852,0.00032518603,0.0002520333,0.000008533733,0.000011958241,0.00023056832,0.000009519233,0.012654649],"genre_scores_gemma":[0.9987394,0.000058661022,0.00016713791,0.000050406463,0.0000023409916,0.00000492639,0.0000928706,0.0000090188605,0.00087520276],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9993918,0.00009776801,0.00005948843,0.00013964545,0.00009728985,0.00021402899],"domain_scores_gemma":[0.9983381,0.00046182668,0.0003789905,0.00017120816,0.0004764262,0.00017346982],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00064174534,0.00031834535,0.0004984328,0.001843633,0.0026003763,0.0017968819,0.00089619437,0.0005037416,0.007275096],"category_scores_gemma":[0.001823034,0.0003556464,0.00014727314,0.002632952,0.0023019603,0.0012132446,0.0017883367,0.00065244833,0.00047939934],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011534952,0.00025834268,0.41164303,0.0009654643,0.0001898052,0.01637697,0.34997788,0.00041795615,0.09685651,0.04223966,0.0037104955,0.07621046],"study_design_scores_gemma":[0.000042169308,0.00012350522,0.8025197,0.0002118612,0.000100427125,0.0029452574,0.1577668,0.0008145531,0.005864615,0.0051396578,0.024387687,0.00008379006],"about_ca_topic_score_codex":0.38471437,"about_ca_topic_score_gemma":0.7024223,"teacher_disagreement_score":0.38471437,"about_ca_system_score_codex":0.0023695685,"about_ca_system_score_gemma":0.0017691692,"threshold_uncertainty_score":0.76495016},"labels":[],"label_agreement":null},{"id":"W4238206514","doi":"10.3115/1610243.1610249","title":"Pre-processing closed captions for machine translation","year":2000,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Pipeline (software); Machine translation; Natural language processing; Artificial intelligence; Speech recognition; Speech translation; Translation (biology); Machine translation system; Translation system; Segmentation; Programming language","score_opus":0.018399525557642404,"score_gpt":0.29350888927938096,"score_spread":0.27510936372173855,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4238206514","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0051858765,0.00056899467,0.972808,0.00034960167,0.00077297457,0.0003319151,0.0009975602,0.009288757,0.009696347],"genre_scores_gemma":[0.06591547,0.000699162,0.9145966,0.00031985465,0.0006918866,0.0007055948,0.0054196753,0.0025138443,0.0091378745],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9972556,0.0011299107,0.00024239814,0.0005536127,0.00067588815,0.00014265081],"domain_scores_gemma":[0.9923028,0.0026941947,0.0004172492,0.0014171746,0.0029945245,0.00017410419],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018198757,0.0020086295,0.0011350537,0.0016735673,0.0018098708,0.004034198,0.001259837,0.0017240067,0.027753327],"category_scores_gemma":[0.010979117,0.00078523107,0.0010509889,0.0018959144,0.0011951016,0.002837423,0.002013378,0.0025744115,0.021969596],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00064406835,0.0002560593,0.0008133398,0.0017958645,0.00012718442,0.0015889014,0.002125358,0.014884398,0.19615293,0.047375455,0.076935716,0.6573008],"study_design_scores_gemma":[0.0000963514,0.00068038807,0.0025915683,0.0004263558,0.00016941498,0.002810462,0.0011601807,0.18902344,0.2608942,0.09187619,0.4499599,0.00031153974],"about_ca_topic_score_codex":0.00079411146,"about_ca_topic_score_gemma":0.0014792003,"teacher_disagreement_score":0.027753327,"about_ca_system_score_codex":0.00063652475,"about_ca_system_score_gemma":0.0011583494,"threshold_uncertainty_score":0.09284413},"labels":[],"label_agreement":null},{"id":"W4238701560","doi":"10.1007/978-0-387-39940-9_2088","title":"Auto-Annotation","year":2009,"lang":"en","type":"book-chapter","venue":"Encyclopedia of Database Systems","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Annotation; Computer science; Artificial intelligence","score_opus":0.013196122375145095,"score_gpt":0.25174499810526063,"score_spread":0.23854887573011554,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4238701560","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0045488384,0.0016446426,0.4842415,0.0012052825,0.00209264,0.0003977277,0.023002591,0.12912785,0.35373896],"genre_scores_gemma":[0.04285278,0.0016464527,0.39218646,0.0016931271,0.00043787295,0.000479825,0.07947898,0.034992218,0.4462324],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99853003,0.0001875099,0.00009479033,0.00054344477,0.00053793105,0.00010636443],"domain_scores_gemma":[0.9975424,0.00042022092,0.000051133124,0.0013231928,0.00060357753,0.000059486032],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001265208,0.0015662719,0.0010837564,0.002556187,0.0017646371,0.0030247995,0.002164635,0.0012039526,0.14196283],"category_scores_gemma":[0.0031489204,0.0010166754,0.0011862427,0.0026202332,0.0007317659,0.0042689336,0.003637934,0.002054252,0.12801145],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014264681,0.00006884325,0.0005702826,0.0005655533,0.000030271118,0.00018498056,0.00039843435,0.0006618211,0.008464156,0.03792811,0.40716314,0.54382175],"study_design_scores_gemma":[0.000010220929,0.000013648942,0.0004635739,0.00011260668,0.000026611075,0.0004340619,0.00010027162,0.0034448428,0.016763289,0.022004763,0.95658886,0.000037294867],"about_ca_topic_score_codex":0.0034429026,"about_ca_topic_score_gemma":0.0051861024,"teacher_disagreement_score":0.14196283,"about_ca_system_score_codex":0.0007341551,"about_ca_system_score_gemma":0.0016477117,"threshold_uncertainty_score":0.47491294},"labels":[],"label_agreement":null},{"id":"W4238891094","doi":"10.1007/978-3-030-71363-8_14","title":"Get Control of Your Commas","year":2021,"lang":"en","type":"book-chapter","venue":"Innovation and change in professional education","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Sentence; Computer science; Meaning (existential); Linguistics; Natural language processing; Control (management); Contrast (vision); Artificial intelligence; Philosophy; Epistemology","score_opus":0.0504789636530071,"score_gpt":0.35537487311804367,"score_spread":0.3048959094650366,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4238891094","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0013203479,0.0011165069,0.025982616,0.005901662,0.003126323,0.000049425886,0.00017805345,0.0012057486,0.9611194],"genre_scores_gemma":[0.007319545,0.00043845977,0.0035090046,0.00072170555,0.00031036322,0.000029467143,0.000081444305,0.0005485375,0.9870415],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99951196,0.00010888177,0.000016029122,0.00009321468,0.00018851606,0.00008137846],"domain_scores_gemma":[0.9992041,0.00027451798,0.000038415066,0.00019661359,0.00018243902,0.000103783306],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005670573,0.00057727075,0.0003632632,0.00090014003,0.002685245,0.0043937946,0.00087751047,0.0015554707,0.17772608],"category_scores_gemma":[0.003486594,0.00030827927,0.0003496292,0.001176817,0.0016837567,0.0071012345,0.0020531877,0.002976303,0.11488151],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00004001187,0.00003294873,0.00014376915,0.00007069176,0.000003858993,0.00009802297,0.0016168163,0.0000996198,0.0010203692,0.26373246,0.48530963,0.24783182],"study_design_scores_gemma":[0.0000026782543,0.0000072351904,0.00008814267,0.000031812866,0.0000015000579,0.000072373616,0.00030994782,0.00012753528,0.00037153892,0.023118714,0.9758637,0.0000048594834],"about_ca_topic_score_codex":0.0024480694,"about_ca_topic_score_gemma":0.005090517,"teacher_disagreement_score":0.17772608,"about_ca_system_score_codex":0.0009805822,"about_ca_system_score_gemma":0.0011233723,"threshold_uncertainty_score":0.594553},"labels":[],"label_agreement":null},{"id":"W4239325927","doi":"10.1162/002438904322793347","title":"Lethal Ambiguity","year":2004,"lang":"en","type":"article","venue":"Linguistic Inquiry","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":62,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Ambiguity; Phrase; Index (typography); Linguistics; Mathematics; Computer science; Philosophy; Programming language","score_opus":0.028221229595017088,"score_gpt":0.31090812001188783,"score_spread":0.28268689041687073,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4239325927","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05146004,0.0011110292,0.40429625,0.0064563355,0.00060208054,0.00016264494,0.000556654,0.0010055301,0.53434944],"genre_scores_gemma":[0.8988741,0.00090936857,0.048310265,0.0023811997,0.00057894504,0.00016789028,0.0005572441,0.0011656402,0.047055203],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9940637,0.0015419809,0.00033345207,0.0011933991,0.0023958117,0.00047157798],"domain_scores_gemma":[0.991319,0.0043390645,0.0005398452,0.0021016123,0.0014361214,0.00026428892],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00377532,0.0010065386,0.0007344262,0.0018415519,0.0031741185,0.003965868,0.0023834612,0.0028250862,0.030601913],"category_scores_gemma":[0.014561625,0.0007241022,0.0012160306,0.0012181638,0.0059946785,0.012814613,0.009284144,0.006398083,0.005280685],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000124590115,0.0000120438635,0.00037195682,0.000025553043,0.000004756883,0.0003623339,0.0010800365,0.00019131384,0.0004172196,0.98650646,0.0016908266,0.009324969],"study_design_scores_gemma":[0.0000126504865,0.000016293443,0.00038056713,0.000049716673,0.000017739507,0.0011273982,0.000569525,0.001494541,0.0017434619,0.93644345,0.058119338,0.000025352383],"about_ca_topic_score_codex":0.0008188328,"about_ca_topic_score_gemma":0.0007550936,"teacher_disagreement_score":0.030601913,"about_ca_system_score_codex":0.002228021,"about_ca_system_score_gemma":0.0014938263,"threshold_uncertainty_score":0.1023736},"labels":[],"label_agreement":null},{"id":"W4239787115","doi":"10.1017/9781316339732.018","title":"Cognitive Grammar","year":2017,"lang":"en","type":"book-chapter","venue":"Cambridge University Press eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Cognitive grammar; Linguistics; Grammar; Cognition; Psychology; Philosophy; Neuroscience","score_opus":0.024367537609406322,"score_gpt":0.2324229950459018,"score_spread":0.20805545743649548,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4239787115","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0027275193,0.005927635,0.07150174,0.00515216,0.00047661216,0.00005976755,0.0018568092,0.0011005959,0.91119707],"genre_scores_gemma":[0.28258693,0.017021628,0.08384784,0.003421952,0.0009240266,0.0003605392,0.008664683,0.002487927,0.6006845],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9995289,0.00014696253,0.000030626627,0.00011105573,0.00014451999,0.00003800337],"domain_scores_gemma":[0.9993388,0.00027891528,0.000023024628,0.00017947484,0.00014103702,0.000038722133],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006861969,0.0006684892,0.0003730228,0.001349063,0.0010155955,0.0036998983,0.00095263723,0.00092635304,0.07070243],"category_scores_gemma":[0.0026605115,0.00033832178,0.00062718836,0.0012731453,0.0031415005,0.004407085,0.0012871032,0.0016302429,0.021007327],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000013137934,0.000014001112,0.00018386668,0.0001827564,0.000012188088,0.00006889696,0.00094716524,0.00056266366,0.00030677635,0.8196963,0.093161464,0.084850796],"study_design_scores_gemma":[0.000004822737,0.000005527618,0.00028115275,0.00012592204,0.000005533188,0.00014424235,0.00025642023,0.0004670786,0.0002183535,0.61479884,0.38368425,0.000007909541],"about_ca_topic_score_codex":0.0035685399,"about_ca_topic_score_gemma":0.003880919,"teacher_disagreement_score":0.07070243,"about_ca_system_score_codex":0.0024186221,"about_ca_system_score_gemma":0.0014782068,"threshold_uncertainty_score":0.23652321},"labels":[],"label_agreement":null},{"id":"W4239941245","doi":"10.22215/etd/2007-06663","title":"Discriminating anaphoric and non-anaphoric definite nouns: a unified memory-based model","year":2007,"lang":"en","type":"dissertation","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University; Canadian Heritage","funders":"","keywords":"Noun; Computer science; Mathematics; Artificial intelligence","score_opus":0.015637033806473206,"score_gpt":0.2986097667214161,"score_spread":0.2829727329149429,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4239941245","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2825696,0.0017966924,0.6961198,0.00200784,0.00031033377,0.00032342126,0.00089357584,0.0026057826,0.013372969],"genre_scores_gemma":[0.8899425,0.00061388395,0.09889067,0.0003281137,0.00012454884,0.00019018,0.00083485583,0.000120115685,0.008955158],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994438,0.00010560215,0.000038364415,0.00020587712,0.00010128858,0.000104987404],"domain_scores_gemma":[0.997781,0.0011530963,0.00018823089,0.00030100186,0.00044535025,0.00013130812],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016581363,0.00084504805,0.001807037,0.002391358,0.0010401866,0.0045191897,0.005242702,0.0022291883,0.0059675938],"category_scores_gemma":[0.004629951,0.0007641085,0.0018415774,0.0016648831,0.0012852217,0.006823047,0.001459929,0.0011508754,0.0017787208],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0029279164,0.0019278737,0.0156637,0.00048122666,0.0010190737,0.0007076934,0.00086775987,0.32620293,0.018944891,0.072311215,0.012424884,0.5465209],"study_design_scores_gemma":[0.000049006674,0.00011970844,0.0007147858,0.000016882392,0.00014020783,0.00009155949,0.00007119685,0.9709526,0.0019114182,0.025306506,0.0005936097,0.000032587872],"about_ca_topic_score_codex":0.015887115,"about_ca_topic_score_gemma":0.010433707,"teacher_disagreement_score":0.015887115,"about_ca_system_score_codex":0.0013975152,"about_ca_system_score_gemma":0.002091558,"threshold_uncertainty_score":0.03158927},"labels":[],"label_agreement":null},{"id":"W4240086160","doi":"10.1080/17470218.2016.1179425","title":"Applying an exemplar model to an implicit rule-learning task: Implicit learning of semantic structure","year":2016,"lang":"en","type":"article","venue":"Quarterly Journal of Experimental Psychology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University; University of Manitoba","funders":"","keywords":"Implicit learning; Task (project management); Implicit knowledge; Cognitive psychology; Computer science; Natural language processing; Psychology; Sequence learning; Artificial intelligence; Cognitive science; Cognition","score_opus":0.016967118787908388,"score_gpt":0.3453577670893784,"score_spread":0.32839064830146997,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4240086160","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.75812215,0.0001984794,0.23348173,0.00082205865,0.0002808147,0.00024592265,0.00037942702,0.0012591899,0.0052103526],"genre_scores_gemma":[0.9157117,0.00018963762,0.080233864,0.00020454716,0.000055959877,0.00011509038,0.00069271465,0.000119582546,0.002676802],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99911934,0.0002899347,0.00007217403,0.00032627035,0.00014404187,0.000048258386],"domain_scores_gemma":[0.9917964,0.005487357,0.00042898237,0.0015930195,0.0004439222,0.00025028738],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027106947,0.00075367413,0.00095912075,0.0002837927,0.00046040287,0.002257813,0.0024276334,0.0021970398,0.0041795042],"category_scores_gemma":[0.019030655,0.00086006883,0.00068891887,0.00042668858,0.000742012,0.005299763,0.0014298129,0.004028571,0.0011467077],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.005015836,0.0055193664,0.020474471,0.0012480365,0.00079540256,0.001109616,0.0023518458,0.14160094,0.17261894,0.027416473,0.00838397,0.61346513],"study_design_scores_gemma":[0.0002786202,0.0006283473,0.0031585572,0.00003774406,0.00017429123,0.00028546914,0.00015162973,0.92915106,0.025534667,0.038955905,0.0015546592,0.00008906817],"about_ca_topic_score_codex":0.0019972639,"about_ca_topic_score_gemma":0.0024555027,"teacher_disagreement_score":0.0041795042,"about_ca_system_score_codex":0.00051585934,"about_ca_system_score_gemma":0.0009946106,"threshold_uncertainty_score":0.0143357515},"labels":[],"label_agreement":null},{"id":"W4240525481","doi":"10.24124/2005/bpgub411","title":"A hybrid approach for FAQ retrieval tasks","year":2005,"lang":"en","type":"dissertation","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Northern British Columbia","funders":"","keywords":"Computer science; Information retrieval; Search engine indexing; Task (project management); Question answering; Automatic indexing; Frequently asked questions; Artificial intelligence","score_opus":0.01397038719359756,"score_gpt":0.2916027259003706,"score_spread":0.27763233870677306,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4240525481","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013750899,0.00035589034,0.9748868,0.00015460786,0.00005225323,0.0006122404,0.00036714075,0.0054906555,0.004329605],"genre_scores_gemma":[0.058103744,0.0002656886,0.930465,0.000121752164,0.00004780906,0.0008396479,0.0013381519,0.00029656247,0.008521724],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99723107,0.0008033661,0.00020966714,0.0006798397,0.00095054397,0.00012547288],"domain_scores_gemma":[0.9973908,0.0011027897,0.00008133247,0.00068872015,0.00063774065,0.00009861787],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025385234,0.0012490792,0.0012906496,0.0025626644,0.0008163401,0.0026244773,0.0031977545,0.0017660444,0.006354053],"category_scores_gemma":[0.00561229,0.0006243304,0.0015095551,0.0028292695,0.0006285659,0.0049031205,0.002001779,0.0014432259,0.0054502054],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00041088823,0.000596641,0.00097308576,0.0005269975,0.00016293378,0.00015169093,0.0008673938,0.009531842,0.053179443,0.017415375,0.0104422765,0.9057414],"study_design_scores_gemma":[0.00030361765,0.00091721927,0.0030382911,0.00007434172,0.0002761265,0.0015731165,0.0009347405,0.8151289,0.058125645,0.048360564,0.07101753,0.00024989486],"about_ca_topic_score_codex":0.0025451172,"about_ca_topic_score_gemma":0.0039543365,"teacher_disagreement_score":0.006354053,"about_ca_system_score_codex":0.00074938283,"about_ca_system_score_gemma":0.001185709,"threshold_uncertainty_score":0.021256447},"labels":[],"label_agreement":null},{"id":"W4240701158","doi":"10.32920/ryerson.14647515.v1","title":"Classification and generation of grammatical errors","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Grammaticality; Computer science; Natural language processing; Artificial intelligence; Syntax; Grammar; Sentence; Natural language; Feature (linguistics); Linguistics","score_opus":0.05923823747943864,"score_gpt":0.3130221755326479,"score_spread":0.25378393805320926,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4240701158","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.89333856,0.00040006478,0.09450934,0.00046426384,0.00018239453,0.00025357096,0.0035237006,0.002303281,0.005024878],"genre_scores_gemma":[0.89997023,0.00023178977,0.09092701,0.000081609265,0.0000684812,0.0001468891,0.0055343905,0.00030820785,0.0027314962],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99699473,0.00091708975,0.00027606194,0.0007841801,0.0008904032,0.0001374385],"domain_scores_gemma":[0.9751549,0.014216013,0.0035498554,0.0023909914,0.004406321,0.00028186],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002048548,0.00044047713,0.00055421254,0.002791671,0.00051512034,0.0012752704,0.0007671447,0.0007084352,0.0017607906],"category_scores_gemma":[0.021362258,0.00017362922,0.00044051243,0.0014373419,0.00046975634,0.00096751685,0.00064228196,0.0006549107,0.001151439],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006281903,0.0003700379,0.23890904,0.0005191249,0.00015512854,0.0012437351,0.0019736267,0.023694692,0.026436195,0.007218958,0.01098364,0.68786764],"study_design_scores_gemma":[0.000059190254,0.00043252987,0.19769266,0.00018992885,0.00019121522,0.0024943189,0.0015681345,0.6689339,0.09664673,0.014400362,0.017281381,0.000109574896],"about_ca_topic_score_codex":0.0011147796,"about_ca_topic_score_gemma":0.0012404645,"teacher_disagreement_score":0.002791671,"about_ca_system_score_codex":0.00065814017,"about_ca_system_score_gemma":0.000810616,"threshold_uncertainty_score":0.010833919},"labels":[],"label_agreement":null},{"id":"W4240848750","doi":"10.1145/572043.572056","title":"A character-level error analysis technique for evaluating text entry methods","year":2002,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Weighting; Character (mathematics); Computer science; Set (abstract data type); Character encoding; Sequence (biology); Error analysis; Data mining; Algorithm; Data set; Artificial intelligence; Natural language processing; Mathematics","score_opus":0.12161983069548121,"score_gpt":0.4292188861413062,"score_spread":0.307599055445825,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4240848750","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.032910224,0.00047739455,0.95330983,0.00011357924,0.00018982109,0.0012556906,0.0022301332,0.006524261,0.0029889704],"genre_scores_gemma":[0.0986497,0.00013971106,0.89362746,0.000075691154,0.00010593401,0.001631784,0.002655715,0.0012011116,0.0019128307],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9567963,0.012404703,0.006867466,0.003216711,0.019869711,0.00084502937],"domain_scores_gemma":[0.81196684,0.11048,0.015726428,0.015934706,0.04493612,0.00095591985],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.018163975,0.0022760676,0.0013046651,0.011845277,0.0014250353,0.0034612913,0.0016780205,0.0016336909,0.004798247],"category_scores_gemma":[0.120037414,0.00051941996,0.0011779767,0.007963912,0.0008814663,0.0033649139,0.0019088109,0.0025695919,0.002180461],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016827037,0.0005997869,0.023850698,0.0021334859,0.0006532823,0.0004320834,0.002115908,0.012904697,0.06803936,0.008675876,0.014427422,0.8644847],"study_design_scores_gemma":[0.0006273432,0.0046000895,0.07042322,0.0011262548,0.0011683236,0.0040106415,0.002121542,0.49735227,0.32232854,0.027741916,0.067661755,0.000838174],"about_ca_topic_score_codex":0.0011595599,"about_ca_topic_score_gemma":0.001552116,"teacher_disagreement_score":0.018163975,"about_ca_system_score_codex":0.00087547896,"about_ca_system_score_gemma":0.0012858681,"threshold_uncertainty_score":0.09606147},"labels":[],"label_agreement":null},{"id":"W4241644272","doi":"10.3115/1604683.1604688","title":"Comparing corpora and lexical ambiguity","year":2000,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Atomic Energy of Canada Limited; Fonds De La Recherche Scientifique - FNRS; Schweizerischer Nationalfonds zur Förderung der Wissenschaftlichen Forschung; National Science Foundation","keywords":"Ambiguity; Computer science; Domain (mathematical analysis); Newspaper; Natural language processing; Artificial intelligence; Information retrieval; Programming language; Mathematics","score_opus":0.025312341402633876,"score_gpt":0.2696550163850276,"score_spread":0.2443426749823937,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4241644272","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6865314,0.032378465,0.1536734,0.0023714579,0.0027023866,0.0015231603,0.00974698,0.002767274,0.10830548],"genre_scores_gemma":[0.8190464,0.00678685,0.14683536,0.0008459228,0.0008945464,0.0019029045,0.018193131,0.0011323709,0.004362522],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9631146,0.02192784,0.0034565912,0.002372505,0.008537349,0.0005911996],"domain_scores_gemma":[0.83415365,0.13242218,0.0054755807,0.013644709,0.013423691,0.00088012504],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.019983932,0.0008181366,0.0010969421,0.013554605,0.0022546838,0.0069627212,0.00166538,0.0015912092,0.00502201],"category_scores_gemma":[0.14972399,0.00063047203,0.0008605804,0.016866567,0.0021289212,0.006782485,0.0040329453,0.0009898334,0.0014013668],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0033636617,0.0011175425,0.1472882,0.0061608944,0.0023661025,0.0019643637,0.009941094,0.028695898,0.01942514,0.09064774,0.02982092,0.6592084],"study_design_scores_gemma":[0.0010345214,0.003020745,0.23886253,0.0037347497,0.0028308833,0.00838761,0.02085181,0.07627233,0.053382423,0.17150514,0.41918984,0.0009274247],"about_ca_topic_score_codex":0.0021770275,"about_ca_topic_score_gemma":0.003181021,"teacher_disagreement_score":0.019983932,"about_ca_system_score_codex":0.00138334,"about_ca_system_score_gemma":0.0009560502,"threshold_uncertainty_score":0.105686426},"labels":[],"label_agreement":null},{"id":"W4241903662","doi":"10.1108/9781787567214","title":"Machine Translation and Global Research: Towards Improved Machine Translation Literacy in the Scholarly Community","year":2019,"lang":"en","type":"book","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":189,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Translation (biology); Machine translation; Literacy; Computer science; Translation studies; Artificial intelligence; Linguistics; Sociology; Pedagogy; Philosophy","score_opus":0.08274042959644197,"score_gpt":0.3703568733226899,"score_spread":0.28761644372624795,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4241903662","genre_codex":"other","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0053841895,0.05715736,0.11363161,0.0543527,0.0070839236,0.00014054825,0.0002362006,0.0024590497,0.7595544],"genre_scores_gemma":[0.073483214,0.067776285,0.13366014,0.0132617215,0.005883397,0.00031441727,0.001134341,0.0031260466,0.7013605],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99839574,0.0005636818,0.00007847295,0.00016089938,0.00069541525,0.00010577338],"domain_scores_gemma":[0.9953317,0.002809871,0.00017198408,0.0005376772,0.000843523,0.00030530282],"candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.0025026335,0.0005153268,0.00057682116,0.0023235227,0.0019697356,0.012082017,0.0007321977,0.0016311894,0.022066051],"category_scores_gemma":[0.0067295586,0.0003915312,0.00043031728,0.003850497,0.0033592405,0.015139237,0.0050632665,0.0033379842,0.008744447],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000019730403,0.000029409237,0.00034444188,0.0004276687,0.000009929836,0.0001074691,0.0046126926,0.00025240445,0.0012117632,0.34576663,0.24222097,0.40499693],"study_design_scores_gemma":[0.0000052577284,0.00001568327,0.0003619499,0.0003400182,0.00000721245,0.0002927031,0.0015591907,0.0006998524,0.0006221385,0.07381982,0.9222645,0.000011638401],"about_ca_topic_score_codex":0.00090558804,"about_ca_topic_score_gemma":0.0028199996,"teacher_disagreement_score":0.98791796,"about_ca_system_score_codex":0.0015570668,"about_ca_system_score_gemma":0.0029333197,"threshold_uncertainty_score":0.073818326},"labels":[],"label_agreement":null},{"id":"W4242963595","doi":"10.18653/v1/w19-43","title":"Proceedings of the 4th Workshop on Representation Learning for NLP (RepL4NLP-2019)","year":2019,"lang":"en","type":"paratext","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Vector Institute; Canadian Institute for Advanced Research","keywords":"Artificial intelligence; Computer science; Natural language processing; Representation (politics); Machine learning","score_opus":0.02380679412803097,"score_gpt":0.3188423832957672,"score_spread":0.2950355891677362,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4242963595","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010491145,0.03277065,0.82029885,0.034270883,0.021255514,0.00082226656,0.020378396,0.02748201,0.0322304],"genre_scores_gemma":[0.053333372,0.02232042,0.6324038,0.010552447,0.008281278,0.0025283643,0.1399206,0.011779529,0.11888021],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99262613,0.0033110322,0.0004927076,0.001695212,0.001373343,0.00050153426],"domain_scores_gemma":[0.9888868,0.0050846627,0.000215127,0.003239169,0.0016781318,0.0008961493],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.010375645,0.002744633,0.003292122,0.0027083857,0.0011831411,0.0067528533,0.004948531,0.004078932,0.06280251],"category_scores_gemma":[0.02113399,0.0011668779,0.002719564,0.0029083318,0.0016731556,0.011887837,0.00771166,0.0081269,0.040063694],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003672374,0.0003912168,0.00035615292,0.00064648053,0.00021569768,0.00024060703,0.000298917,0.00500253,0.0021775968,0.015716534,0.5975682,0.37701878],"study_design_scores_gemma":[0.0001800003,0.00022731745,0.0012387644,0.0005869444,0.00012165523,0.0003798705,0.00029411635,0.07957648,0.0040210546,0.09018409,0.82308525,0.0001044635],"about_ca_topic_score_codex":0.0055794516,"about_ca_topic_score_gemma":0.007095699,"teacher_disagreement_score":0.9371975,"about_ca_system_score_codex":0.0025544248,"about_ca_system_score_gemma":0.0032242292,"threshold_uncertainty_score":0.21009535},"labels":[],"label_agreement":null},{"id":"W4242981620","doi":"10.31234/osf.io/fnqzt","title":"Beyond plain and extra-grammatical morphology: echo-pairs in Hungarian","year":2020,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Context (archaeology); Similarity (geometry); Echo (communications protocol); Computer science; Grammar; Contrast (vision); Lexicon; Natural language processing; Set (abstract data type); Artificial intelligence; Linguistics; Metric (unit); Variation (astronomy); Morphology (biology); Physics; Geography; Geology; Astrophysics","score_opus":0.018970662219236514,"score_gpt":0.27617891597841715,"score_spread":0.25720825375918066,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4242981620","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9933942,0.00016766075,0.0030547227,0.00005383808,0.000007043529,0.000014373048,0.00015723,0.000026271737,0.0031246757],"genre_scores_gemma":[0.99762493,0.00007322428,0.0015249762,0.000027024573,0.000006072182,0.000010042498,0.00026501602,0.000038021415,0.00043076204],"study_design_codex":"observational","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9991417,0.00033471276,0.000101604164,0.00021191884,0.00013884266,0.00007120339],"domain_scores_gemma":[0.99316674,0.004952852,0.0005843223,0.0007412754,0.00038164298,0.00017330682],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009963034,0.00027593816,0.00064826204,0.0010576718,0.00075452,0.0012583997,0.00044885417,0.00062936015,0.00420506],"category_scores_gemma":[0.0077530276,0.00026849232,0.00021559681,0.0014306654,0.0019434876,0.0026581013,0.0018000964,0.00070881,0.0007918374],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0022705588,0.00048432508,0.33461776,0.002521811,0.0002805313,0.008777374,0.24359404,0.0025209347,0.10149521,0.030179646,0.0028820196,0.2703758],"study_design_scores_gemma":[0.00007865361,0.00033223827,0.86397463,0.00013953878,0.00011575686,0.0072238934,0.06302695,0.005775718,0.018255554,0.017911147,0.022985617,0.00018022321],"about_ca_topic_score_codex":0.0012662774,"about_ca_topic_score_gemma":0.0020159215,"teacher_disagreement_score":0.00420506,"about_ca_system_score_codex":0.00033801034,"about_ca_system_score_gemma":0.00021222889,"threshold_uncertainty_score":0.014067292},"labels":[],"label_agreement":null},{"id":"W4243041937","doi":"10.1017/cbo9780511801686","title":"Analyzing Linguistic Data","year":2008,"lang":"en","type":"book","venue":"Cambridge University Press eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3194,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Variety (cybernetics); Computer science; Construct (python library); Variation (astronomy); Visualization; Natural language processing; Range (aeronautics); Artificial intelligence; Measure (data warehouse); Statistical model; Corpus linguistics; Linguistics; Data science; Data mining; Engineering; Programming language","score_opus":0.033527306789434844,"score_gpt":0.24245010166766912,"score_spread":0.20892279487823429,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4243041937","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010722587,0.007347953,0.6669684,0.0127617465,0.002731959,0.0013837437,0.031147934,0.010141751,0.256794],"genre_scores_gemma":[0.049713887,0.010521294,0.74702275,0.004213885,0.0013565258,0.0018585926,0.035466194,0.004877033,0.14496987],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9949432,0.0017306502,0.00045727415,0.0006536764,0.002130448,0.000084872125],"domain_scores_gemma":[0.9877977,0.007480753,0.00046413825,0.0018373469,0.0022694764,0.0001505522],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005411941,0.00093731197,0.00081523776,0.005406387,0.0014339042,0.0072097024,0.0014761707,0.0009199222,0.05099769],"category_scores_gemma":[0.026920753,0.00049773086,0.00082359446,0.005818343,0.0016727464,0.00523484,0.002571823,0.002213044,0.037673693],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000026964011,0.000041572534,0.0016483634,0.0012352691,0.00006349267,0.0002513525,0.004097162,0.0010484783,0.002271621,0.10015137,0.23501292,0.65415144],"study_design_scores_gemma":[0.000007934844,0.000032345415,0.001965866,0.0008052386,0.00002612755,0.00040723511,0.0024055624,0.0024377247,0.0014611207,0.085368514,0.9050379,0.00004441099],"about_ca_topic_score_codex":0.0013652418,"about_ca_topic_score_gemma":0.0019863604,"teacher_disagreement_score":0.05099769,"about_ca_system_score_codex":0.0014001449,"about_ca_system_score_gemma":0.0022115062,"threshold_uncertainty_score":0.17060429},"labels":[],"label_agreement":null},{"id":"W4244519801","doi":"10.32920/14636088","title":"Identifying emerging priorities in Knowledge Translation from the perspective of trainees","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; Dalhousie University","funders":"","keywords":"Knowledge translation; Field (mathematics); Perspective (graphical); Engineering ethics; Clinical Practice; Medical education; Knowledge management; Medicine; Political science; Psychology; Computer science; Engineering; Nursing","score_opus":0.06160024869133392,"score_gpt":0.34735914235194026,"score_spread":0.28575889366060636,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4244519801","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.46343714,0.014999411,0.10604549,0.2985907,0.0016365902,0.00039472192,0.00026159882,0.00016094858,0.11447338],"genre_scores_gemma":[0.95357966,0.0042992514,0.02949914,0.0056067323,0.00032097913,0.00020724899,0.00014589299,0.000086201166,0.0062548663],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9204648,0.0589722,0.0031315591,0.0029981225,0.0071687703,0.0072645713],"domain_scores_gemma":[0.8978422,0.063214414,0.007981301,0.002812371,0.014516716,0.013632909],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.07064019,0.00070205797,0.0007600163,0.0043857503,0.0105288625,0.025478452,0.0019144461,0.006490781,0.005138882],"category_scores_gemma":[0.07584296,0.0010324134,0.00059967133,0.003974451,0.014767144,0.025304869,0.016926033,0.010163007,0.0013753702],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00040399752,0.00020545087,0.020211652,0.0017064237,0.00009251295,0.0023240205,0.579889,0.00071543275,0.005123767,0.27710158,0.01015333,0.10207281],"study_design_scores_gemma":[0.000042810887,0.00014458987,0.0061220783,0.001365216,0.00004983172,0.0010877821,0.6452575,0.0023075722,0.002497086,0.2612208,0.0797983,0.000106341126],"about_ca_topic_score_codex":0.0031897482,"about_ca_topic_score_gemma":0.004336321,"teacher_disagreement_score":0.9293598,"about_ca_system_score_codex":0.010220157,"about_ca_system_score_gemma":0.021446716,"threshold_uncertainty_score":0.37358552},"labels":[],"label_agreement":null},{"id":"W4244580177","doi":"10.18653/v1/w18-07","title":"Proceedings of the First Workshop on Computational Models of Reference, Anaphora and Coreference","year":2018,"lang":"en","type":"paratext","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Deutsche Forschungsgemeinschaft","keywords":"Coreference; Anaphora (linguistics); Computer science; Natural language processing; Artificial intelligence; Resolution (logic)","score_opus":0.03548965268301026,"score_gpt":0.2833075706707328,"score_spread":0.24781791798772254,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4244580177","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014471984,0.03134959,0.7979677,0.051945668,0.010901984,0.0003119038,0.004649665,0.0023403731,0.086061224],"genre_scores_gemma":[0.26293612,0.02883223,0.53283304,0.008805489,0.008422288,0.0014217345,0.02213108,0.0024632858,0.13215478],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99724895,0.0014743435,0.00018500497,0.0005833186,0.000390607,0.00011771637],"domain_scores_gemma":[0.9901578,0.0071193106,0.00017127555,0.0013989857,0.000730475,0.0004221289],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005829389,0.0010886027,0.001605699,0.0013085783,0.0017444097,0.007482355,0.0031791318,0.0027465601,0.023219971],"category_scores_gemma":[0.014285439,0.0010438234,0.0024813912,0.002235978,0.0028876993,0.011804045,0.0037095535,0.0065848106,0.004271117],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000350195,0.00022900953,0.00097462203,0.0007215474,0.0002826181,0.00040724318,0.0015319685,0.012095766,0.0009995421,0.58640456,0.24648756,0.14951535],"study_design_scores_gemma":[0.00006412714,0.00003065291,0.00060269143,0.00023539536,0.00008312086,0.00022841447,0.0002285814,0.03444899,0.0007146559,0.67066514,0.29265466,0.000043579406],"about_ca_topic_score_codex":0.0051917615,"about_ca_topic_score_gemma":0.007544056,"teacher_disagreement_score":0.023219971,"about_ca_system_score_codex":0.003092503,"about_ca_system_score_gemma":0.002388994,"threshold_uncertainty_score":0.07767856},"labels":[],"label_agreement":null},{"id":"W4245257647","doi":"10.26686/wgtn.12552221","title":"How large a vocabulary is needed for reading and listening?","year":2020,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"Victoria University; Victoria University of Wellington","keywords":"Vocabulary; Linguistics; Computer science; Reading comprehension; Active listening; Comprehension; Word (group theory); Natural language processing; Reading (process); Artificial intelligence; Psychology; Communication","score_opus":0.022204957687457658,"score_gpt":0.28395266465346125,"score_spread":0.2617477069660036,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4245257647","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.60085535,0.0488521,0.096353225,0.039210513,0.001270941,0.00034018894,0.0024514354,0.0017202081,0.208946],"genre_scores_gemma":[0.9524684,0.010255635,0.026673634,0.0013379862,0.0005124694,0.00024673922,0.001580818,0.0006277029,0.006296728],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99701464,0.0011982261,0.00028608795,0.0005191338,0.0006908004,0.00029118184],"domain_scores_gemma":[0.9856455,0.009140452,0.001017882,0.0013838864,0.0022143512,0.0005979704],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027394018,0.00041333126,0.00079285714,0.0018481603,0.0011147392,0.003258405,0.0009797818,0.0011868265,0.009189421],"category_scores_gemma":[0.03135994,0.00043223434,0.00035995277,0.0014536157,0.0033319453,0.013133107,0.0015882901,0.0012760594,0.0039542895],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006980042,0.0001103625,0.022005223,0.0024379126,0.0001755759,0.0011143906,0.019801348,0.0011819083,0.05311068,0.096775696,0.017362516,0.78522646],"study_design_scores_gemma":[0.00025267235,0.0010018587,0.14897947,0.0020425,0.000481703,0.0062860795,0.045510672,0.0041973894,0.030778926,0.48444998,0.27564287,0.00037591136],"about_ca_topic_score_codex":0.007721997,"about_ca_topic_score_gemma":0.0070709763,"teacher_disagreement_score":0.009189421,"about_ca_system_score_codex":0.0011807358,"about_ca_system_score_gemma":0.0018990528,"threshold_uncertainty_score":0.030741692},"labels":[],"label_agreement":null},{"id":"W4245446776","doi":"10.1093/acrefore/9780199384655.013.9","title":"Eskimo-Aleut","year":2016,"lang":"en","type":"reference-entry","venue":"Oxford Research Encyclopedia of Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":31,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Ergative case; Linguistics; Austronesian languages; Language family; History; North Germanic languages; Pidgin; Consonant; Germanic languages; Languages of Africa; Variety (cybernetics); Grammar; Language contact; Vowel; Geography; Computer science; Mathematics; German; Creole language","score_opus":0.04809420827594272,"score_gpt":0.3692342398757151,"score_spread":0.32114003159977234,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4245446776","genre_codex":"empirical","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6271675,0.003579038,0.0004975197,0.0009831518,0.0001997497,0.00010450772,0.0044564446,0.000106412146,0.36290565],"genre_scores_gemma":[0.876361,0.002644854,0.00071079284,0.0005515341,0.00005531953,0.00007486652,0.0038419347,0.000034180815,0.115725465],"study_design_codex":"observational","study_design_gemma":"not_applicable","domain_scores_codex":[0.99989295,0.000010915168,0.00000521927,0.000019134786,0.000024948948,0.000046849014],"domain_scores_gemma":[0.99992037,0.0000096774875,0.000010378557,0.000006520354,0.00003136663,0.00002172351],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00011345887,0.00037719967,0.00015369775,0.0008445044,0.001963209,0.0009821618,0.00025018174,0.00018442425,0.019916596],"category_scores_gemma":[0.00025057254,0.000060645565,0.000120280674,0.0011455242,0.00030360377,0.00035898158,0.0012647996,0.00023409631,0.0038921505],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00089725514,0.00045625403,0.4331949,0.000740578,0.00010686446,0.010961181,0.021697272,0.00050280214,0.0095202895,0.048441157,0.052160937,0.4213206],"study_design_scores_gemma":[0.000066312896,0.00009741092,0.40469715,0.00036427463,0.00005972252,0.003651459,0.023128735,0.000507068,0.0025549412,0.0025486893,0.5622952,0.000029057534],"about_ca_topic_score_codex":0.061960083,"about_ca_topic_score_gemma":0.089820854,"teacher_disagreement_score":0.061960083,"about_ca_system_score_codex":0.0009396708,"about_ca_system_score_gemma":0.0010123695,"threshold_uncertainty_score":0.12319887},"labels":[],"label_agreement":null},{"id":"W4245510738","doi":"10.1111/bjd.12753","title":"Plain Language Summaries","year":2014,"lang":"en","type":"article","venue":"British Journal of Dermatology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Medicine","score_opus":0.004832617028212291,"score_gpt":0.23732686511124645,"score_spread":0.23249424808303415,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4245510738","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":"reporting","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":"reporting","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0010182639,0.009176437,0.008702595,0.040639903,0.052911445,0.0018310093,0.3182039,0.009513507,0.5580029],"genre_scores_gemma":[0.014041753,0.013178054,0.0068679717,0.027413787,0.018932428,0.0026872596,0.22943076,0.005670072,0.68177795],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99695134,0.0008663541,0.0008233959,0.000313302,0.00086697814,0.00017866687],"domain_scores_gemma":[0.9670903,0.0162412,0.002402497,0.0025505868,0.010701283,0.0010140968],"candidate_categories":["metaresearch","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.00263316,0.0013140748,0.001323018,0.004982147,0.0013176943,0.00474264,0.002316771,0.0023348904,0.7718204],"category_scores_gemma":[0.0524194,0.00051835284,0.0008116598,0.0053044152,0.0008233991,0.004411554,0.0020077254,0.0023077005,0.64778084],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000037033093,0.00000809528,0.000033296856,0.00047002258,0.0000020770342,0.0000546588,0.00004679725,0.000030099076,0.0000569958,0.0010528428,0.98346657,0.014741435],"study_design_scores_gemma":[0.000012973818,0.000011471542,0.00014423137,0.00028016168,0.0000018342076,0.00009531194,0.000060937655,0.000032336473,0.000053591495,0.0013014181,0.99799806,0.000007759606],"about_ca_topic_score_codex":0.0025026654,"about_ca_topic_score_gemma":0.002496838,"teacher_disagreement_score":0.99736685,"about_ca_system_score_codex":0.0017163432,"about_ca_system_score_gemma":0.0023180908,"threshold_uncertainty_score":0.32547045},"labels":[],"label_agreement":null},{"id":"W4246001087","doi":"10.18653/v1/w18-10","title":"Proceedings of the Workshop on Generalization in the Age of Deep Learning","year":2018,"lang":"en","type":"paratext","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; McGill University","funders":"Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs; Canadian Institute for Advanced Research","keywords":"Generalization; Computer science; Artificial intelligence; Epistemology; Philosophy","score_opus":0.01637968248543957,"score_gpt":0.2818999357589256,"score_spread":0.265520253273486,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4246001087","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.043663368,0.05761291,0.7421241,0.0802418,0.008659031,0.0002909803,0.0027781082,0.00420574,0.060423978],"genre_scores_gemma":[0.48125717,0.030522866,0.3633463,0.014545989,0.009481283,0.00081408245,0.009068574,0.0020491586,0.088914454],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9974075,0.0011117257,0.00017541909,0.00067651627,0.00041893535,0.00021003548],"domain_scores_gemma":[0.98765266,0.007837729,0.00023906287,0.0025148392,0.0012067205,0.0005490955],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0088332705,0.0013480482,0.0014106956,0.00092912203,0.0008832891,0.00450811,0.0031094016,0.002760841,0.018411016],"category_scores_gemma":[0.0243189,0.00072700175,0.001200982,0.0011212406,0.001995761,0.0132608535,0.0042909025,0.0065039154,0.0043002167],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007166992,0.00042124698,0.002463252,0.00072851067,0.00036542927,0.0003349179,0.00088952476,0.031002278,0.0027709682,0.18068346,0.201983,0.57764065],"study_design_scores_gemma":[0.00010236525,0.00016058321,0.0017163164,0.00039887344,0.00012393948,0.00023690848,0.0002260591,0.19986196,0.0037692008,0.60478914,0.1885596,0.000055037264],"about_ca_topic_score_codex":0.004769782,"about_ca_topic_score_gemma":0.0056194956,"teacher_disagreement_score":0.98158896,"about_ca_system_score_codex":0.0023387647,"about_ca_system_score_gemma":0.001587398,"threshold_uncertainty_score":0.06159097},"labels":[],"label_agreement":null},{"id":"W4247857759","doi":"10.18653/v1/p19-4","title":"Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics: Tutorial Abstracts","year":2019,"lang":"en","type":"paratext","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of the Fraser Valley; University of British Columbia","funders":"Deutsche Forschungsgemeinschaft","keywords":"Computer science; Computational linguistics; Association (psychology); Library science; Natural language processing; Philosophy; Epistemology","score_opus":0.008674975200917489,"score_gpt":0.2695221782550612,"score_spread":0.2608472030541437,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4247857759","genre_codex":"review","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011890039,0.22799401,0.17220894,0.13227808,0.21726239,0.0014539246,0.016567938,0.010849287,0.20949547],"genre_scores_gemma":[0.020864293,0.082391076,0.077561036,0.011938199,0.042298395,0.002237882,0.036478337,0.0058049224,0.72042584],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99774444,0.0006563344,0.00026161037,0.00046572203,0.00069382513,0.00017808202],"domain_scores_gemma":[0.98705184,0.004459872,0.0005000524,0.0010162812,0.0052232044,0.0017487092],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007985777,0.0016256712,0.0025964885,0.004769703,0.0017365325,0.0060223173,0.0025122843,0.0025717313,0.14953046],"category_scores_gemma":[0.011897687,0.0006943492,0.0009315863,0.003402999,0.0019313396,0.006595162,0.0037400476,0.003948391,0.10733991],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008890666,0.000056363795,0.00025709096,0.0003157927,0.000032880223,0.00007838992,0.00009966792,0.00009560522,0.00076774706,0.001366401,0.87839967,0.11844145],"study_design_scores_gemma":[0.000031496773,0.000035528294,0.00082161976,0.0005103074,0.00003974058,0.00017788187,0.00016248126,0.0012359531,0.00036661804,0.0042134,0.992376,0.000029015442],"about_ca_topic_score_codex":0.0039501283,"about_ca_topic_score_gemma":0.006589513,"teacher_disagreement_score":0.14953046,"about_ca_system_score_codex":0.0017975124,"about_ca_system_score_gemma":0.0038339456,"threshold_uncertainty_score":0.50022924},"labels":[],"label_agreement":null},{"id":"W4248692001","doi":"10.1007/978-1-4939-7131-2_100319","title":"Election Prediction","year":2018,"lang":"en","type":"book-chapter","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Computer science","score_opus":0.011023245364575586,"score_gpt":0.24013405989530326,"score_spread":0.22911081453072768,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4248692001","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.032513186,0.0039554657,0.56563205,0.008023664,0.0042286357,0.0010275551,0.02252917,0.01991041,0.34217992],"genre_scores_gemma":[0.46957085,0.0023849397,0.17636615,0.0014166624,0.0018791754,0.00062797114,0.03725451,0.0015911423,0.3089086],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99929714,0.0001339262,0.000026704993,0.00022446775,0.00021452818,0.00010322567],"domain_scores_gemma":[0.99864584,0.00054172636,0.000072886396,0.00040860756,0.00025055048,0.000080404825],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011672148,0.0008823108,0.00090115174,0.0015796007,0.0009964226,0.0020269705,0.0016098783,0.00068674155,0.07085214],"category_scores_gemma":[0.005875178,0.00042511677,0.00075910357,0.0016166614,0.0003541805,0.0024721387,0.0008656416,0.0018731679,0.03358793],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022074765,0.0002063842,0.0040446157,0.00018566828,0.00006765906,0.000056665875,0.000053404074,0.031127235,0.00091708073,0.048787262,0.3312263,0.58310694],"study_design_scores_gemma":[0.00006969213,0.00010699979,0.003795786,0.00013161449,0.00008954792,0.00023974804,0.000117842406,0.63171023,0.0053115305,0.17844829,0.17992249,0.000056294924],"about_ca_topic_score_codex":0.0053204508,"about_ca_topic_score_gemma":0.010838138,"teacher_disagreement_score":0.07085214,"about_ca_system_score_codex":0.00097038684,"about_ca_system_score_gemma":0.0016576129,"threshold_uncertainty_score":0.23702401},"labels":[],"label_agreement":null},{"id":"W4248700681","doi":"10.1007/978-1-4614-6170-8_100012","title":"Election Prediction","year":2014,"lang":"en","type":"book-chapter","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Political science; Computer science","score_opus":0.00873331151419013,"score_gpt":0.22763018089857998,"score_spread":0.21889686938438985,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4248700681","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.031952344,0.0037581124,0.5682886,0.007448429,0.0036109134,0.0009590848,0.01987481,0.017870873,0.34623688],"genre_scores_gemma":[0.46415365,0.0023420076,0.17934512,0.0013088895,0.0016227803,0.0005871299,0.032691568,0.0014666278,0.31648225],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9993394,0.0001299296,0.000025006433,0.00020314187,0.0002067648,0.00009565472],"domain_scores_gemma":[0.99875236,0.0005152182,0.00006936268,0.00036523078,0.00022662195,0.000071148024],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011310137,0.0008384538,0.00086901913,0.0015563057,0.0009785463,0.0019001124,0.0015702074,0.0006510414,0.06780004],"category_scores_gemma":[0.005418477,0.00042078717,0.00072299066,0.0016013798,0.000356226,0.002343625,0.0008285559,0.0017414985,0.030529438],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020159697,0.00018551042,0.0039912644,0.0001699011,0.000063552594,0.000055467193,0.0000564869,0.031547073,0.0008288365,0.04972138,0.30763996,0.60553896],"study_design_scores_gemma":[0.00006584987,0.0001036036,0.0039621443,0.00013457646,0.00008517373,0.00025481026,0.00013023474,0.6215675,0.005210746,0.18600982,0.1824178,0.000057839392],"about_ca_topic_score_codex":0.0055843056,"about_ca_topic_score_gemma":0.011577554,"teacher_disagreement_score":0.06780004,"about_ca_system_score_codex":0.0009617795,"about_ca_system_score_gemma":0.0016022116,"threshold_uncertainty_score":0.22681373},"labels":[],"label_agreement":null},{"id":"W4248814784","doi":"10.22215/etd/2013-06719","title":"The music criticism of Jacob Siskind : a case study of corpus linguistic analysis","year":2013,"lang":"en","type":"dissertation","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University; Library and Archives Canada","funders":"","keywords":"Linguistics; Criticism; Humanities; Art; Literature; Philosophy","score_opus":0.015449675693186376,"score_gpt":0.3097651840650042,"score_spread":0.2943155083718178,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4248814784","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.539838,0.011106439,0.049091082,0.13329166,0.0017861605,0.00019991423,0.00047234862,0.00092986925,0.26328447],"genre_scores_gemma":[0.9062789,0.0031459266,0.02048153,0.009177136,0.00045070943,0.00010251967,0.0003203708,0.001434238,0.05860867],"study_design_codex":"qualitative","study_design_gemma":"observational","domain_scores_codex":[0.99205196,0.0040157586,0.00034621038,0.00059031905,0.0027114889,0.00028433482],"domain_scores_gemma":[0.9482024,0.041735124,0.0021038952,0.0021970205,0.0049947426,0.0007667547],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011332496,0.00030533847,0.0003605562,0.004043274,0.008887205,0.0058484995,0.0011146963,0.0024933324,0.0030019358],"category_scores_gemma":[0.045774512,0.00032661634,0.00017787179,0.00539578,0.008136087,0.0041718013,0.0023545823,0.0037791117,0.000609289],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023442804,0.000089036694,0.01233236,0.000603771,0.000047143167,0.006296573,0.48611328,0.0007730144,0.0062837447,0.1875571,0.09756118,0.20210844],"study_design_scores_gemma":[0.000023085368,0.00004857726,0.014181607,0.0010541792,0.00004428311,0.0031044532,0.17349602,0.0038162651,0.0061353184,0.033447113,0.7645565,0.00009258158],"about_ca_topic_score_codex":0.017846297,"about_ca_topic_score_gemma":0.048272215,"teacher_disagreement_score":0.017846297,"about_ca_system_score_codex":0.00499037,"about_ca_system_score_gemma":0.0047874344,"threshold_uncertainty_score":0.05993271},"labels":[],"label_agreement":null},{"id":"W4249221689","doi":"10.3115/1567564.1567565","title":"Concept identification and presentation in the context of technical text summarization","year":2000,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"McGill University","keywords":"Automatic summarization; Computer science; Identification (biology); Context (archaeology); Presentation (obstetrics); Information retrieval; Text graph; Process (computing); Natural language processing; Multi-document summarization; Artificial intelligence","score_opus":0.010317124412362677,"score_gpt":0.2832430542500074,"score_spread":0.27292592983764474,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4249221689","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019706523,0.00273513,0.96868503,0.0011301537,0.00029972263,0.00030230306,0.00032156217,0.0037010186,0.0031185176],"genre_scores_gemma":[0.13435268,0.0018893192,0.8575449,0.00027509136,0.0006705353,0.00036180913,0.0013611629,0.00045238782,0.0030921747],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.993166,0.0039637135,0.000595916,0.00091254344,0.0011657687,0.00019613477],"domain_scores_gemma":[0.98094934,0.011271216,0.002236116,0.001634987,0.003472317,0.00043601272],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0069064572,0.0013261809,0.0009929906,0.0039173663,0.0011955895,0.0045546195,0.0016397592,0.0014713962,0.0046555544],"category_scores_gemma":[0.028594987,0.00049937697,0.0009785183,0.0031490282,0.0011430341,0.0061119646,0.0018431048,0.0020741778,0.0028456445],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006411805,0.00023370383,0.0014240874,0.0042147343,0.00016129861,0.0011519756,0.007705448,0.008762093,0.1095736,0.032652244,0.02009991,0.8133797],"study_design_scores_gemma":[0.000342362,0.0019319387,0.006814941,0.0010874288,0.00094884593,0.0049334816,0.0065592704,0.15773378,0.34025726,0.16503191,0.31379277,0.0005659605],"about_ca_topic_score_codex":0.0003070409,"about_ca_topic_score_gemma":0.00032827773,"teacher_disagreement_score":0.0069064572,"about_ca_system_score_codex":0.0005442869,"about_ca_system_score_gemma":0.0010336654,"threshold_uncertainty_score":0.03652531},"labels":[],"label_agreement":null},{"id":"W4249507038","doi":"10.24124/2006/bpgub433","title":"A basis for pronominal anaphora resolution using a model of working memory and long-term memory","year":2006,"lang":"en","type":"dissertation","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Northern British Columbia","funders":"University of Northern British Columbia","keywords":"Computer science; Natural language processing; Anaphora (linguistics); Artificial intelligence; Term (time); Semantic memory; Sentence; Representation (politics); Working memory; Memory model; Modular design; Basis (linear algebra); Cognitive model; Resolution (logic); Cognitive science; Cognition; Psychology; Mathematics; Programming language","score_opus":0.03872365006935224,"score_gpt":0.2996913671874682,"score_spread":0.26096771711811595,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4249507038","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.047438066,0.0004190411,0.9349573,0.0013278456,0.00009843391,0.000066740424,0.000111709385,0.00034101948,0.015239841],"genre_scores_gemma":[0.6541725,0.0006086672,0.33247867,0.00029933816,0.000116825424,0.00036267206,0.00036036537,0.00016357264,0.01143738],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9995437,0.00013127655,0.000037331694,0.000099803,0.00014036825,0.00004754866],"domain_scores_gemma":[0.9986498,0.0007461234,0.00009202378,0.00025428165,0.00018631323,0.00007149889],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012980662,0.0003369207,0.00066788646,0.000835333,0.0009420656,0.002882661,0.0024208745,0.0012534759,0.004846932],"category_scores_gemma":[0.0046765637,0.00047244556,0.0015821574,0.00072901754,0.001575984,0.0065741264,0.0010289238,0.0013384484,0.0011559895],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009337276,0.000062786465,0.000799608,0.000084423824,0.00004679588,0.00025212136,0.00067622366,0.037239596,0.003218144,0.92066246,0.0014206837,0.035443813],"study_design_scores_gemma":[0.00003420409,0.00003244796,0.00028931227,0.00003219508,0.00002728887,0.00021254408,0.00010991459,0.34369385,0.0016143862,0.6495627,0.004366217,0.000024901216],"about_ca_topic_score_codex":0.0019625009,"about_ca_topic_score_gemma":0.0018230538,"teacher_disagreement_score":0.004846932,"about_ca_system_score_codex":0.0010411924,"about_ca_system_score_gemma":0.0011164885,"threshold_uncertainty_score":0.01621461},"labels":[],"label_agreement":null},{"id":"W4249939617","doi":"10.18653/v1/2021.spnlp-1","title":"Proceedings of the 5th Workshop on Structured Prediction for NLP (SPNLP 2021)","year":2021,"lang":"en","type":"paratext","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Google (Canada); University of Alberta","funders":"","keywords":"Computer science; Artificial intelligence; Natural language processing; Machine learning","score_opus":0.015326543218306482,"score_gpt":0.2868517970105378,"score_spread":0.27152525379223136,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4249939617","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021365853,0.035928156,0.76983386,0.029686345,0.020954944,0.0018943939,0.03428021,0.046032384,0.040023856],"genre_scores_gemma":[0.06303886,0.015993146,0.60738033,0.005958611,0.0056346953,0.0022845569,0.18732184,0.008886186,0.10350186],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99306524,0.0032238122,0.0004585169,0.001620458,0.0012752722,0.00035672894],"domain_scores_gemma":[0.9863651,0.006587651,0.00032912556,0.0031027375,0.0026476528,0.00096778286],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.009915223,0.003242505,0.0031312585,0.0033714622,0.0014195254,0.006179633,0.0047963937,0.00411932,0.050937038],"category_scores_gemma":[0.02107221,0.0010085717,0.002368827,0.0032573147,0.0015182225,0.011068444,0.0052503496,0.005979647,0.037391193],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005143298,0.00042203686,0.0004918359,0.00092414493,0.0002586124,0.0003794051,0.00039949358,0.006907883,0.0045917435,0.005764268,0.5943221,0.38502407],"study_design_scores_gemma":[0.00025914126,0.00041217162,0.0021771155,0.0005750197,0.00020881128,0.00059840875,0.00052210863,0.14547646,0.008128876,0.048336945,0.79316413,0.00014085168],"about_ca_topic_score_codex":0.0074360655,"about_ca_topic_score_gemma":0.011548599,"teacher_disagreement_score":0.94906294,"about_ca_system_score_codex":0.0024700721,"about_ca_system_score_gemma":0.0033408692,"threshold_uncertainty_score":0.1704014},"labels":[],"label_agreement":null},{"id":"W4249988467","doi":"10.1007/978-0-387-39940-9_3881","title":"Translingual Information Retrieval","year":2009,"lang":"en","type":"book-chapter","venue":"Encyclopedia of Database Systems","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Information retrieval; Computer science","score_opus":0.010775527234006975,"score_gpt":0.24163296135660892,"score_spread":0.23085743412260193,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4249988467","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008678808,0.049030814,0.2951438,0.0045765303,0.0046350583,0.00039777491,0.0059338277,0.022816699,0.6087867],"genre_scores_gemma":[0.05904626,0.032858457,0.1456624,0.0030325688,0.0021126713,0.00021587069,0.019591084,0.0053028557,0.7321778],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9992797,0.000118924545,0.00007338624,0.00015044371,0.00030015619,0.0000774203],"domain_scores_gemma":[0.999091,0.0001744339,0.000042229716,0.0002764865,0.00035175207,0.00006405728],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007460208,0.0009436691,0.0011355008,0.004250236,0.0011153478,0.0042562056,0.0012025086,0.0008129416,0.08838467],"category_scores_gemma":[0.0020192608,0.0005262292,0.0006823956,0.0055119838,0.00073063356,0.005320642,0.0023682027,0.0010795431,0.06913431],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007235949,0.00006836865,0.00014952695,0.00078388676,0.00003406296,0.00018339515,0.00031626344,0.00032486586,0.007318324,0.026496865,0.24637446,0.71787775],"study_design_scores_gemma":[0.000013045568,0.000044755558,0.00036507423,0.00018705484,0.00003893744,0.0011994595,0.00015848743,0.0019154935,0.008066564,0.02114576,0.9668355,0.00002985255],"about_ca_topic_score_codex":0.0015962624,"about_ca_topic_score_gemma":0.0025232034,"teacher_disagreement_score":0.08838467,"about_ca_system_score_codex":0.00087574014,"about_ca_system_score_gemma":0.0012262184,"threshold_uncertainty_score":0.29567617},"labels":[],"label_agreement":null},{"id":"W4250270644","doi":"10.3765/salt.v0i0.2807","title":"Micro-Events in Two Serial Verb Constructions","year":2015,"lang":"en","type":"article","venue":"Proceedings from Semantics and Linguistic Theory","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Verb; Linguistics; Computer science; Natural language processing; Psychology; Artificial intelligence; Mathematics; Philosophy","score_opus":0.013053310898498312,"score_gpt":0.27263337688169975,"score_spread":0.25958006598320144,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4250270644","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4363306,0.0012903614,0.15591872,0.004902141,0.0017033127,0.00019389064,0.0008501896,0.0013784505,0.39743236],"genre_scores_gemma":[0.9701071,0.00014527934,0.011677911,0.0001448256,0.0000972558,0.00004908575,0.00022713908,0.00026047687,0.017291],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.999246,0.00028563107,0.00004547598,0.00015852647,0.00017676271,0.00008763116],"domain_scores_gemma":[0.99800926,0.0013521683,0.0001384296,0.0002559393,0.0001633964,0.00008082756],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00088877487,0.0004272578,0.00035355386,0.00058910885,0.0017763366,0.0026847802,0.0006876391,0.0015677081,0.018939536],"category_scores_gemma":[0.0036335483,0.00043871233,0.00051654916,0.0007786289,0.002572633,0.003933978,0.0018721296,0.0021972123,0.0014300211],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025534045,0.000049846818,0.0005292518,0.00009586168,0.000011720362,0.0010593908,0.0042716507,0.00029231238,0.0067010825,0.9743363,0.0023377766,0.01005942],"study_design_scores_gemma":[0.00017344212,0.00014443065,0.0055416487,0.00017231458,0.00006665086,0.002473492,0.0048022475,0.008591366,0.023073649,0.8686548,0.086224414,0.0000815344],"about_ca_topic_score_codex":0.0010904092,"about_ca_topic_score_gemma":0.001278464,"teacher_disagreement_score":0.018939536,"about_ca_system_score_codex":0.001168675,"about_ca_system_score_gemma":0.00045666317,"threshold_uncertainty_score":0.06335902},"labels":[],"label_agreement":null},{"id":"W4250821834","doi":"10.1007/978-0-387-39940-9_3720","title":"Suffix Stripping","year":2009,"lang":"en","type":"book-chapter","venue":"Encyclopedia of Database Systems","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Suffix; Computer science; Stripping (fiber); Materials science; Linguistics; Philosophy; Composite material","score_opus":0.01463490014768878,"score_gpt":0.2480926576420964,"score_spread":0.23345775749440761,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4250821834","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007020413,0.0076851067,0.48328868,0.0010638243,0.0024744552,0.0004216321,0.010790727,0.03545026,0.45180497],"genre_scores_gemma":[0.043266356,0.00787119,0.41295028,0.0010841999,0.0006611237,0.00026840283,0.034013428,0.01008188,0.4898031],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99932766,0.0000536995,0.00007043731,0.00017973215,0.00030608944,0.00006243054],"domain_scores_gemma":[0.9989096,0.0001648386,0.00003687976,0.0004701425,0.00038084883,0.00003771896],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005303184,0.0012382322,0.0010725096,0.0024673492,0.0013321146,0.0032168296,0.0018880051,0.00093448785,0.12824184],"category_scores_gemma":[0.0019515463,0.00074361963,0.0009628247,0.0042530843,0.0006247504,0.003148124,0.0022995423,0.0017596303,0.15282191],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000098187054,0.00006592955,0.0001952694,0.0005548178,0.00003305586,0.0002153446,0.0001425358,0.0006197328,0.021874273,0.028357096,0.1248246,0.82301915],"study_design_scores_gemma":[0.000017648377,0.00003816873,0.00028810735,0.00012866122,0.000042629217,0.0010568703,0.000091816655,0.0024269796,0.040220913,0.029237946,0.92641705,0.000033211003],"about_ca_topic_score_codex":0.000734827,"about_ca_topic_score_gemma":0.0013069105,"teacher_disagreement_score":0.12824184,"about_ca_system_score_codex":0.00041601795,"about_ca_system_score_gemma":0.0011384132,"threshold_uncertainty_score":0.4290117},"labels":[],"label_agreement":null},{"id":"W4251616496","doi":"10.18653/v1/w17-48","title":"Proceedings of the Third Workshop on Discourse in Machine Translation","year":2017,"lang":"en","type":"paratext","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"Vetenskapsrådet; European Commission; Schweizerischer Nationalfonds zur Förderung der Wissenschaftlichen Forschung; National Science Foundation","keywords":"Machine translation; Computer science; Translation (biology); Linguistics; Natural language processing; Artificial intelligence; Chemistry; Philosophy","score_opus":0.023737979639620096,"score_gpt":0.3219455282126653,"score_spread":0.2982075485730452,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4251616496","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.030431964,0.07974577,0.5604381,0.059950843,0.0504944,0.0012753321,0.010698262,0.015452555,0.19151275],"genre_scores_gemma":[0.20277841,0.037324417,0.35899922,0.010228934,0.016485222,0.0027201304,0.058832135,0.009919231,0.30271232],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99328786,0.004037353,0.0004617937,0.00094566867,0.00085216213,0.00041514356],"domain_scores_gemma":[0.9899504,0.004631185,0.00022827722,0.0024457532,0.0018253472,0.000919026],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.010467668,0.0020261894,0.0022170395,0.002687547,0.001907603,0.009650304,0.0032251473,0.002778037,0.059182946],"category_scores_gemma":[0.0155269625,0.00069933746,0.0013327644,0.0027056101,0.0022458439,0.01035996,0.0065857302,0.004986929,0.027586779],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000686984,0.0005373979,0.0007080663,0.0010166141,0.0001762749,0.0004911229,0.0019623858,0.0020952916,0.0055942237,0.0388285,0.5116118,0.43629128],"study_design_scores_gemma":[0.00013071798,0.00013815764,0.0010789064,0.00056853367,0.00008738567,0.00046638722,0.0010885077,0.013865025,0.0067539397,0.057651687,0.91810715,0.000063528365],"about_ca_topic_score_codex":0.0028591484,"about_ca_topic_score_gemma":0.0035320392,"teacher_disagreement_score":0.94081706,"about_ca_system_score_codex":0.0023365894,"about_ca_system_score_gemma":0.0029116895,"threshold_uncertainty_score":0.19798666},"labels":[],"label_agreement":null},{"id":"W4251741293","doi":"10.1145/860500.860534","title":"Passage retrieval vs. document retrieval for factoid question answering","year":2003,"lang":"en","type":"article","venue":"Proceedings of the 26th annual international ACM SIGIR conference on Research and development in informaion retrieval - SIGIR '03","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Question answering; Information retrieval; Computer science; Document retrieval; Natural language processing; Artificial intelligence","score_opus":0.047081778044312704,"score_gpt":0.3471654451101543,"score_spread":0.3000836670658416,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4251741293","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2028463,0.11916502,0.5156373,0.00704567,0.0070248647,0.005667789,0.012930188,0.019178648,0.11050429],"genre_scores_gemma":[0.70155823,0.016459879,0.24197957,0.0011929531,0.0022677002,0.0013930936,0.010188456,0.0011354528,0.023824772],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9967487,0.001901898,0.00021610798,0.0003816319,0.0004939633,0.00025770903],"domain_scores_gemma":[0.98609364,0.011898041,0.00023421358,0.00084688445,0.000621725,0.00030546892],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0049565765,0.00072324194,0.0017954577,0.0037945455,0.0009461742,0.0024667305,0.0017493629,0.0019730416,0.055301834],"category_scores_gemma":[0.024235131,0.00029500888,0.0011239118,0.0026693959,0.000829558,0.00486436,0.0014797914,0.001099375,0.011079007],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.016664585,0.00066286343,0.0022307155,0.0066249124,0.0005159101,0.00028630128,0.0004600103,0.004737021,0.047883675,0.016555747,0.04931959,0.8540588],"study_design_scores_gemma":[0.0059119673,0.01739271,0.027379738,0.0034324725,0.007180952,0.00502516,0.0023202081,0.44477597,0.16959563,0.119036615,0.19705173,0.0008968467],"about_ca_topic_score_codex":0.0037654603,"about_ca_topic_score_gemma":0.0028666721,"teacher_disagreement_score":0.055301834,"about_ca_system_score_codex":0.0007385901,"about_ca_system_score_gemma":0.0007204085,"threshold_uncertainty_score":0.18500304},"labels":[],"label_agreement":null},{"id":"W4252110927","doi":"10.3115/1119176","title":"Proceedings of the seventh conference on Natural language learning at HLT-NAACL 2003 -","year":2003,"lang":"en","type":"paratext","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":122,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Presentation (obstetrics); Computer science; Conjunction (astronomy); Natural language processing; Natural language; Natural (archaeology); Artificial intelligence; Library science; History; Medicine; Archaeology","score_opus":0.01095413753061084,"score_gpt":0.265145987527732,"score_spread":0.25419184999712113,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4252110927","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011949205,0.03223612,0.44998452,0.06980003,0.06988665,0.004170705,0.06517971,0.070954144,0.22583887],"genre_scores_gemma":[0.024677064,0.017100004,0.17035513,0.015280906,0.0070854505,0.0040032016,0.24742667,0.013264269,0.5008072],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9898689,0.0047230055,0.0009377811,0.0013047172,0.002715303,0.0004503666],"domain_scores_gemma":[0.9845414,0.005235121,0.0004632572,0.002762987,0.005093257,0.0019040135],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.012763291,0.001930688,0.0033767794,0.0025958675,0.0020611952,0.012801663,0.005081146,0.0033435884,0.15172261],"category_scores_gemma":[0.02540551,0.0010205106,0.0015322684,0.002500142,0.0021716405,0.011453794,0.006472585,0.0067532687,0.12753943],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017509455,0.0002225544,0.00041839966,0.0003434137,0.00005422087,0.000095614705,0.0002009187,0.00036047035,0.0011868967,0.0036568649,0.904306,0.08897963],"study_design_scores_gemma":[0.00011116629,0.00008760629,0.00063820544,0.0002953863,0.000062651736,0.00025902953,0.00033629138,0.004376642,0.001978596,0.012037219,0.979764,0.000053296986],"about_ca_topic_score_codex":0.00595808,"about_ca_topic_score_gemma":0.011772133,"teacher_disagreement_score":0.8482774,"about_ca_system_score_codex":0.0024491374,"about_ca_system_score_gemma":0.0061349166,"threshold_uncertainty_score":0.5075627},"labels":[],"label_agreement":null},{"id":"W4253348229","doi":"10.1007/978-0-387-39940-9_3009","title":"Machine-Readable Dictionary (MRD)","year":2009,"lang":"en","type":"book-chapter","venue":"Encyclopedia of Database Systems","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence","score_opus":0.010977528200181871,"score_gpt":0.23769799272409986,"score_spread":0.22672046452391797,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4253348229","genre_codex":"dataset","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0033405202,0.007052494,0.26962057,0.0028644835,0.0047773886,0.0005242013,0.4124243,0.044531073,0.25486493],"genre_scores_gemma":[0.024616413,0.00749419,0.26526323,0.0023688253,0.0009360843,0.00060414,0.5171682,0.013506544,0.1680424],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9992483,0.00008723714,0.00020026868,0.00017696891,0.00024514066,0.00004202202],"domain_scores_gemma":[0.9971506,0.00066470227,0.00016536449,0.0008420316,0.0010223292,0.00015506345],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00062322343,0.0009840041,0.0012526577,0.00492124,0.00075844344,0.003545521,0.002128737,0.0009581503,0.15996538],"category_scores_gemma":[0.004718797,0.00046349276,0.00052533066,0.007678962,0.00057251845,0.004852183,0.0021696226,0.0015399306,0.15704621],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000108734326,0.000039622664,0.00021416726,0.0015459533,0.000017486547,0.00017031642,0.00018789075,0.00042784563,0.0049265474,0.029368378,0.6105508,0.35244223],"study_design_scores_gemma":[0.000009014414,0.000010463537,0.00014299048,0.00011008313,0.00000722377,0.00024814074,0.000042617623,0.00041189155,0.0015485211,0.0041869823,0.9932682,0.0000137580755],"about_ca_topic_score_codex":0.00167315,"about_ca_topic_score_gemma":0.0026026827,"teacher_disagreement_score":0.15996538,"about_ca_system_score_codex":0.0005594147,"about_ca_system_score_gemma":0.0019276675,"threshold_uncertainty_score":0.53513753},"labels":[],"label_agreement":null},{"id":"W4253378051","doi":"10.1017/9781108674553","title":"An Advanced Introduction to Semantics","year":2020,"lang":"en","type":"book","venue":"Cambridge University Press eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":42,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University; Université de Montréal","funders":"","keywords":"Computer science; Semantics (computer science); Representation (politics); Natural language processing; Computational semantics; Meaning (existential); Artificial intelligence; Natural language; Sentence; Set (abstract data type); Linguistics; Natural language understanding; Knowledge representation and reasoning; Programming language; Operational semantics; Psychology","score_opus":0.010172036708698365,"score_gpt":0.21919219912859828,"score_spread":0.2090201624198999,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4253378051","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00083735585,0.0552451,0.10165425,0.013242136,0.0060841604,0.00007394497,0.00122194,0.0015680289,0.82007307],"genre_scores_gemma":[0.03050636,0.07707155,0.10579266,0.00663114,0.007491199,0.0002766619,0.0027214522,0.0020343752,0.7674746],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99924755,0.00015253626,0.00005961159,0.00012768361,0.00035299824,0.000059725964],"domain_scores_gemma":[0.99941504,0.00033016424,0.000018557159,0.00006996056,0.00012088942,0.00004538198],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00068288966,0.0011445361,0.0008278649,0.0022127177,0.0013777897,0.004233213,0.0010635478,0.0012129477,0.06365109],"category_scores_gemma":[0.0016409234,0.0006071077,0.0010877516,0.00331891,0.0028299154,0.007877683,0.0018739882,0.0034631207,0.036585975],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000009916199,0.00001614693,0.000062039995,0.00021937917,0.000008090333,0.00007233227,0.0003426873,0.00041629106,0.00035376553,0.66750866,0.24070515,0.090285555],"study_design_scores_gemma":[0.0000013685179,0.0000036990955,0.000041932377,0.00007061161,0.0000014137324,0.000109735236,0.000045804914,0.00021425982,0.00005360824,0.1852043,0.81424916,0.0000041214557],"about_ca_topic_score_codex":0.0014279288,"about_ca_topic_score_gemma":0.0019765706,"teacher_disagreement_score":0.06365109,"about_ca_system_score_codex":0.002028891,"about_ca_system_score_gemma":0.0015847221,"threshold_uncertainty_score":0.21293408},"labels":[],"label_agreement":null},{"id":"W4253990414","doi":"10.1504/ijmso.2018.096454","title":"Ontology of folktales in the Greater Mekong Subregion","year":2018,"lang":"en","type":"article","venue":"International Journal of Metadata Semantics and Ontologies","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Humanities Research Group, University of Windsor; Khon Kaen University","keywords":"Ontology; Upper ontology; Documentation; Scope (computer science); Computer science; Domain (mathematical analysis); Ontology-based data integration; Process ontology; Digital library; Field (mathematics); Suggested Upper Merged Ontology; Information retrieval; Data science; Knowledge management; Linguistics; Semantic Web; Epistemology; Philosophy","score_opus":0.03199908701069654,"score_gpt":0.3192780661648376,"score_spread":0.2872789791541411,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4253990414","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.77128375,0.0030306908,0.049957637,0.0027224983,0.00010076064,0.00037434066,0.0026271224,0.0002784899,0.16962473],"genre_scores_gemma":[0.9640385,0.0011930806,0.022053862,0.00013258701,0.000006080632,0.000082220235,0.0011668069,0.0000310152,0.011295887],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99960047,0.00014765556,0.000046397105,0.000075470045,0.000060954582,0.00006910828],"domain_scores_gemma":[0.99943227,0.00013975317,0.000107123036,0.00009416046,0.00015941387,0.00006723468],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00059475424,0.000197051,0.00016535987,0.0024860294,0.0024449436,0.0028090726,0.0004982105,0.00027021728,0.0024659329],"category_scores_gemma":[0.00086637674,0.00015315325,0.00022493479,0.004700665,0.0026332575,0.0036045364,0.001631492,0.0004727571,0.00022501693],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019376045,0.0001293028,0.060099877,0.001599442,0.00006615046,0.0032943536,0.16827412,0.0051564043,0.0097132,0.57491934,0.0072734915,0.16928048],"study_design_scores_gemma":[0.00002039191,0.000046008736,0.09916113,0.00089497474,0.000104315055,0.0020934765,0.33870292,0.0051447013,0.0044920854,0.07162342,0.47762114,0.00009541873],"about_ca_topic_score_codex":0.1540311,"about_ca_topic_score_gemma":0.25254783,"teacher_disagreement_score":0.1540311,"about_ca_system_score_codex":0.0064819157,"about_ca_system_score_gemma":0.008013635,"threshold_uncertainty_score":0.30626905},"labels":[],"label_agreement":null},{"id":"W4254153873","doi":"10.1075/hts.1.tra9","title":"Translation tools","year":2010,"lang":"en","type":"book-chapter","venue":"Handbook of translation studies","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Translation (biology); Computer science; Linguistics; Philosophy; Biology; Genetics","score_opus":0.1074508003532897,"score_gpt":0.3257478369640071,"score_spread":0.21829703661071742,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4254153873","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0017526518,0.012101933,0.5313352,0.0016017876,0.0019388342,0.00051425217,0.0125340745,0.04641451,0.3918068],"genre_scores_gemma":[0.021988174,0.015626172,0.51974237,0.0013291775,0.0007147428,0.0008998477,0.050717644,0.01691416,0.3720677],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99846965,0.00034182888,0.00020314204,0.00029506555,0.0005815094,0.0001087029],"domain_scores_gemma":[0.9982085,0.0006096269,0.00005729304,0.00059926324,0.00046400086,0.00006134045],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012872822,0.0019325524,0.0014367488,0.0052658604,0.0016029396,0.006378603,0.0028921952,0.0015286789,0.17950635],"category_scores_gemma":[0.0049922266,0.0013327536,0.0013344517,0.007705977,0.0010849098,0.006869563,0.003407501,0.002507992,0.1704262],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006072539,0.00006922997,0.00011647644,0.0009759687,0.000030726387,0.00016367796,0.0004242992,0.0005321595,0.0030680157,0.08390809,0.2734936,0.6371571],"study_design_scores_gemma":[0.000019158297,0.000018814799,0.0001496672,0.00030701663,0.000024207548,0.00053733564,0.0001523982,0.0014533584,0.004069415,0.042415094,0.95082855,0.000024961191],"about_ca_topic_score_codex":0.0012396165,"about_ca_topic_score_gemma":0.0016995142,"teacher_disagreement_score":0.17950635,"about_ca_system_score_codex":0.0009309678,"about_ca_system_score_gemma":0.0018138455,"threshold_uncertainty_score":0.6005086},"labels":[],"label_agreement":null},{"id":"W4254861393","doi":"10.32920/ryerson.14647515","title":"Classification and generation of grammatical errors","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Grammaticality; Computer science; Natural language processing; Artificial intelligence; Syntax; Grammar; Sentence; Natural language; Feature (linguistics); Linguistics","score_opus":0.05923823747943864,"score_gpt":0.3130221755326479,"score_spread":0.25378393805320926,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4254861393","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.89333856,0.00040006478,0.09450934,0.00046426384,0.00018239453,0.00025357096,0.0035237006,0.002303281,0.005024878],"genre_scores_gemma":[0.89997023,0.00023178977,0.09092701,0.000081609265,0.0000684812,0.0001468891,0.0055343905,0.00030820785,0.0027314962],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99699473,0.00091708975,0.00027606194,0.0007841801,0.0008904032,0.0001374385],"domain_scores_gemma":[0.9751549,0.014216013,0.0035498554,0.0023909914,0.004406321,0.00028186],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002048548,0.00044047713,0.00055421254,0.002791671,0.00051512034,0.0012752704,0.0007671447,0.0007084352,0.0017607906],"category_scores_gemma":[0.021362258,0.00017362922,0.00044051243,0.0014373419,0.00046975634,0.00096751685,0.00064228196,0.0006549107,0.001151439],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006281903,0.0003700379,0.23890904,0.0005191249,0.00015512854,0.0012437351,0.0019736267,0.023694692,0.026436195,0.007218958,0.01098364,0.68786764],"study_design_scores_gemma":[0.000059190254,0.00043252987,0.19769266,0.00018992885,0.00019121522,0.0024943189,0.0015681345,0.6689339,0.09664673,0.014400362,0.017281381,0.000109574896],"about_ca_topic_score_codex":0.0011147796,"about_ca_topic_score_gemma":0.0012404645,"teacher_disagreement_score":0.002791671,"about_ca_system_score_codex":0.00065814017,"about_ca_system_score_gemma":0.000810616,"threshold_uncertainty_score":0.010833919},"labels":[],"label_agreement":null},{"id":"W4255209682","doi":"10.5220/0007382000960104","title":"Umple as a Template Language (Umple-TL)","year":2019,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Programming language; Natural language processing; Linguistics; Philosophy","score_opus":0.006237508612002995,"score_gpt":0.27117939059230467,"score_spread":0.26494188198030166,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4255209682","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0017546518,0.00012929928,0.9280454,0.00037688203,0.00034693617,0.00015708068,0.003355445,0.048231684,0.017602604],"genre_scores_gemma":[0.09915461,0.00049284386,0.79234546,0.0016646169,0.00037366935,0.0007757039,0.016787535,0.04075082,0.04765476],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99698573,0.00092692015,0.00042072244,0.0006711972,0.0006980059,0.00029742302],"domain_scores_gemma":[0.9964043,0.0009417759,0.00018877977,0.001921734,0.00041327142,0.00013029137],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022188313,0.0012405331,0.00091799983,0.0013621899,0.0009825124,0.004817542,0.0027828298,0.0017786582,0.045441303],"category_scores_gemma":[0.0070911082,0.0012697702,0.0021168503,0.0014898643,0.0014365796,0.009333558,0.00523414,0.003470717,0.027143331],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00044187487,0.00014606875,0.0007555883,0.0010894152,0.00009230345,0.00057416543,0.0012036545,0.0029063718,0.01457954,0.5643609,0.14301886,0.27083126],"study_design_scores_gemma":[0.00007009005,0.00009730111,0.00022924202,0.00027543696,0.000084726824,0.0008876709,0.00021641079,0.03510062,0.05370166,0.2520387,0.65717804,0.00012017467],"about_ca_topic_score_codex":0.00094681943,"about_ca_topic_score_gemma":0.0014759887,"teacher_disagreement_score":0.045441303,"about_ca_system_score_codex":0.00076683407,"about_ca_system_score_gemma":0.001414783,"threshold_uncertainty_score":0.15201628},"labels":[],"label_agreement":null},{"id":"W4255514414","doi":"10.5771/0943-7444-2020-4-320","title":"Facet","year":2020,"lang":"en","type":"article","venue":"KNOWLEDGE ORGANIZATION","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Facet (psychology); Product (mathematics); Object (grammar); Term (time); Popularity; Field (mathematics); Subject (documents); Function (biology); Computer science; Sociology; Knowledge management; Psychology; Library science; Artificial intelligence; Social psychology","score_opus":0.01350486682738577,"score_gpt":0.2450513619245166,"score_spread":0.23154649509713082,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4255514414","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0105262995,0.020358251,0.34690908,0.03132084,0.008977352,0.00031983625,0.002359265,0.005432113,0.57379705],"genre_scores_gemma":[0.20461327,0.028254095,0.26689756,0.014153785,0.003906788,0.00050580356,0.0076245726,0.0032996123,0.47074446],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9973061,0.00045469165,0.00020356684,0.0007800125,0.001029351,0.00022632293],"domain_scores_gemma":[0.9958768,0.0011018778,0.00026421202,0.00067408837,0.0018346529,0.0002483271],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017239716,0.0008646671,0.00073615735,0.0020974851,0.0028173593,0.0054999213,0.001427381,0.0018108401,0.039237957],"category_scores_gemma":[0.0074425195,0.00049262284,0.0010609986,0.003206314,0.0022971982,0.008732542,0.0038070055,0.0028382568,0.029601043],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001430712,0.000030759587,0.00095179945,0.0004950432,0.00002194945,0.00031644048,0.0016091492,0.0004712187,0.0024466983,0.4835855,0.21189462,0.29803374],"study_design_scores_gemma":[0.000006732618,0.000028995528,0.00025771567,0.00012656406,0.000011095396,0.00085017824,0.00037055832,0.0009821998,0.0016580558,0.06830618,0.9273776,0.000024130193],"about_ca_topic_score_codex":0.0040502925,"about_ca_topic_score_gemma":0.0048801256,"teacher_disagreement_score":0.039237957,"about_ca_system_score_codex":0.002444283,"about_ca_system_score_gemma":0.003365829,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W4255710711","doi":"10.1007/978-3-319-41337-2_7","title":"Bilingual Corpora","year":2016,"lang":"en","type":"book-chapter","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Computer Research Institute of Montréal","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Parallel corpora; Character (mathematics); Term (time); Identification (biology); Point (geometry); Corpus linguistics; Linguistics; Machine translation; Mathematics","score_opus":0.01868149459437175,"score_gpt":0.25505652725125055,"score_spread":0.23637503265687881,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4255710711","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0037932151,0.005718627,0.10272601,0.0022516435,0.0027151096,0.0004560122,0.028254345,0.008557061,0.84552807],"genre_scores_gemma":[0.053657502,0.010169303,0.18205951,0.0017336801,0.0014126234,0.0010028753,0.15513082,0.009700023,0.5851336],"study_design_codex":"not_applicable","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9989557,0.00026259,0.00009293395,0.00028230698,0.00033014064,0.000076422715],"domain_scores_gemma":[0.99829525,0.00037195304,0.000046560443,0.00065103965,0.0005218674,0.00011341968],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015797218,0.0012652503,0.00091206824,0.004157675,0.001851751,0.0031596692,0.0017009103,0.0009071667,0.2264946],"category_scores_gemma":[0.003973668,0.00090559584,0.0005564884,0.0047680596,0.00086700363,0.0053044655,0.0032860648,0.0018988043,0.11267836],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015064064,0.000103503204,0.00018254622,0.0009657726,0.000026924827,0.00017970052,0.00035377117,0.0005666347,0.0072891205,0.10733541,0.45885584,0.4239902],"study_design_scores_gemma":[0.000019093437,0.000015228612,0.0002061281,0.0001372465,0.000014897291,0.00032162797,0.0000816994,0.0005637793,0.005299738,0.022012927,0.9713127,0.000014974365],"about_ca_topic_score_codex":0.0021001445,"about_ca_topic_score_gemma":0.004705781,"teacher_disagreement_score":0.2264946,"about_ca_system_score_codex":0.0010961655,"about_ca_system_score_gemma":0.0027471713,"threshold_uncertainty_score":0.7576999},"labels":[],"label_agreement":null},{"id":"W4280510965","doi":"10.18653/v1/2022.naacl-main.249","title":"On the Use of Bert for Automated Essay Scoring: Joint Learning of Multi-Scale Essay Representation","year":2022,"lang":"en","type":"article","venue":"Proceedings of the 2022 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":97,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Readability; Computer science; Representation (politics); Artificial intelligence; Scale (ratio); Set (abstract data type); Natural language processing; Transfer of learning; Task (project management); Feature learning; Joint (building); Deep learning; Domain (mathematical analysis); Machine learning; Multi-task learning; Programming language; Mathematics; Engineering","score_opus":0.045477286995067465,"score_gpt":0.2898841241251151,"score_spread":0.24440683713004763,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4280510965","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19636257,0.001382658,0.786039,0.0012514216,0.00029234172,0.0001824196,0.0006674331,0.0070291413,0.0067929593],"genre_scores_gemma":[0.87495077,0.00040060622,0.115860134,0.00019533523,0.00014856424,0.00010616415,0.0012975805,0.0001329834,0.0069079436],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99915373,0.00038876515,0.00004946074,0.00019849025,0.00014765546,0.00006190292],"domain_scores_gemma":[0.9977602,0.0009326489,0.00027228938,0.0004562018,0.00041666767,0.000162038],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018997686,0.001187017,0.00069700903,0.0010434411,0.00035170038,0.0012000878,0.00102615,0.0010512252,0.002070007],"category_scores_gemma":[0.0066336934,0.0002450959,0.0003048739,0.001107598,0.0005045083,0.0032042023,0.0015726637,0.0017914507,0.001134397],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005871914,0.00035633674,0.0057101813,0.00011624453,0.00007644324,0.00012673314,0.00018754396,0.09183167,0.01068141,0.0070910305,0.008812601,0.8744226],"study_design_scores_gemma":[0.000021413585,0.00013807893,0.0015192522,0.000014959802,0.000012804293,0.000056114906,0.00004667038,0.985759,0.0035881104,0.0070157824,0.0018077004,0.000020162412],"about_ca_topic_score_codex":0.0015088494,"about_ca_topic_score_gemma":0.0025356058,"teacher_disagreement_score":0.002070007,"about_ca_system_score_codex":0.00058424915,"about_ca_system_score_gemma":0.0005885134,"threshold_uncertainty_score":0.0100470185},"labels":[],"label_agreement":null},{"id":"W4280583959","doi":"10.31234/osf.io/wjavp","title":"LSTMs Can Learn Basic Wh- and Relative Clause Dependencies in Norwegian","year":2022,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Focus (optics); Computer science; Norwegian; Dependency (UML); Sentence; Verb; Artificial intelligence; Natural language processing; Recurrent neural network; Generality; Linguistics; Artificial neural network; Psychology","score_opus":0.01792078873601291,"score_gpt":0.271579650121968,"score_spread":0.25365886138595506,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4280583959","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.81022257,0.0007307359,0.16507679,0.0011948776,0.00031685177,0.00006218586,0.0026664971,0.003184425,0.016545005],"genre_scores_gemma":[0.97182626,0.00020122535,0.021355914,0.00019290755,0.00002079618,0.00004503827,0.002396506,0.00022136922,0.003739846],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996897,0.00007780379,0.000023743982,0.000122030266,0.000037395086,0.00004940141],"domain_scores_gemma":[0.9989518,0.00072060525,0.00009476464,0.00008759684,0.00010899815,0.000036188878],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00092845026,0.0011256754,0.00038657515,0.00026371185,0.00032860524,0.0006747012,0.00072037725,0.0006748237,0.0025670242],"category_scores_gemma":[0.0048046517,0.0006169678,0.0007971962,0.00027050727,0.0007031092,0.002242661,0.00086534634,0.0014412572,0.00079648395],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00082257576,0.00018972825,0.013409159,0.0006080285,0.00020923937,0.0013898525,0.0028593931,0.7320469,0.08503251,0.016260462,0.0073320167,0.13984016],"study_design_scores_gemma":[0.000044830656,0.00013590076,0.0049924846,0.00006411189,0.00006306527,0.00009594687,0.00039456866,0.96070135,0.00908085,0.021323094,0.0030492246,0.000054667526],"about_ca_topic_score_codex":0.0334517,"about_ca_topic_score_gemma":0.040966503,"teacher_disagreement_score":0.0334517,"about_ca_system_score_codex":0.0012055099,"about_ca_system_score_gemma":0.00085757626,"threshold_uncertainty_score":0.066514015},"labels":[],"label_agreement":null},{"id":"W4280588123","doi":"10.18172/jes.5324","title":"Automatic Lemmatization of Old English Class III Strong Verbs (L-Y) with ALOEV3","year":2022,"lang":"en","type":"article","venue":"Journal of English Studies","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Lemmatisation; Lemma (botany); Computer science; Artificial intelligence; Natural language processing; Linguistics; Parsing; Class (philosophy); Alternation (linguistics); Verb; Philosophy","score_opus":0.013690521355006745,"score_gpt":0.2646892288346338,"score_spread":0.25099870747962705,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4280588123","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19062857,0.00075215753,0.76387745,0.0002105407,0.00024844264,0.0005244437,0.005230851,0.02085243,0.017675066],"genre_scores_gemma":[0.4668457,0.0003886221,0.505201,0.00013633103,0.0000658457,0.0003508519,0.0123093715,0.0050677517,0.009634483],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99921143,0.0001984262,0.00011494897,0.00020976519,0.00018401773,0.00008143108],"domain_scores_gemma":[0.9978994,0.0011023685,0.00019396617,0.00040099738,0.000364169,0.00003901979],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010607146,0.0005313382,0.0005546195,0.0017081106,0.0006018594,0.0017149795,0.0006322577,0.00040962212,0.00884048],"category_scores_gemma":[0.0032142883,0.000621572,0.0005812459,0.0012116763,0.00074850465,0.0019701086,0.0015124377,0.0007605673,0.0035706921],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012896308,0.0001624784,0.013733142,0.0014695992,0.00013088818,0.0016248794,0.00560278,0.003078241,0.18468373,0.04961904,0.028539227,0.71006626],"study_design_scores_gemma":[0.00042054098,0.00048350246,0.034506213,0.00039825152,0.00019864352,0.0038606934,0.0044796015,0.098733105,0.39864308,0.056684773,0.40130165,0.0002900366],"about_ca_topic_score_codex":0.0011679194,"about_ca_topic_score_gemma":0.002154489,"teacher_disagreement_score":0.00884048,"about_ca_system_score_codex":0.0004893438,"about_ca_system_score_gemma":0.0007307293,"threshold_uncertainty_score":0.029574335},"labels":[],"label_agreement":null},{"id":"W4280632712","doi":"10.1162/coli_a_00447","title":"Noun2Verb: Probabilistic Frame Semantics for Word Class Conversion","year":2022,"lang":"en","type":"article","venue":"Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute; University of Toronto","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Linguistics; Verb; Principle of compositionality; Semantics (computer science); Programming language","score_opus":0.016793507886842647,"score_gpt":0.27922083496854416,"score_spread":0.2624273270817015,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4280632712","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.028870443,0.00006687344,0.9668657,0.00029019953,0.000041153235,0.000050556097,0.00024212325,0.0015608214,0.0020121853],"genre_scores_gemma":[0.75242186,0.00008956412,0.24486539,0.00014037141,0.00003687329,0.0002004268,0.0006064351,0.00034712598,0.0012919719],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987594,0.00060870056,0.00005231245,0.00029488106,0.00021419016,0.0000706176],"domain_scores_gemma":[0.996763,0.0020529612,0.0002555358,0.00058079226,0.0002486709,0.0000989387],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021152415,0.0005150759,0.0005936728,0.00066505297,0.0007021105,0.001746831,0.0022771172,0.00096492446,0.005434425],"category_scores_gemma":[0.00824252,0.0005531334,0.001310401,0.00050091685,0.0018795271,0.003441574,0.0017818833,0.0017617716,0.0006006217],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035142316,0.00021777264,0.0033558707,0.00023173357,0.000113160015,0.0003975972,0.0015436512,0.36813393,0.012723621,0.48363134,0.0061385483,0.12316136],"study_design_scores_gemma":[0.000018148177,0.000024527964,0.00027725802,0.000008795659,0.0000069689295,0.000049634396,0.000042983265,0.87788206,0.0013000536,0.11868253,0.0016926249,0.000014460242],"about_ca_topic_score_codex":0.0075375643,"about_ca_topic_score_gemma":0.0077747884,"teacher_disagreement_score":0.0075375643,"about_ca_system_score_codex":0.0012881339,"about_ca_system_score_gemma":0.0010994215,"threshold_uncertainty_score":0.018179953},"labels":[],"label_agreement":null},{"id":"W4281391185","doi":"10.3389/frai.2022.796741","title":"Beyond the Benchmarks: Toward Human-Like Lexical Representations","year":2022,"lang":"en","type":"review","venue":"Frontiers in Artificial Intelligence","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada; National Science Foundation","keywords":"Conceptualization; Lexicon; Lexical semantics; Computer science; Natural language processing; Semantics (computer science); Artificial intelligence; Computational semantics; Human language; Lexical item; Process (computing); Key (lock); Word (group theory); Linguistics; Cognitive science; Psychology; Programming language; Operational semantics","score_opus":0.10248626005897338,"score_gpt":0.390488722566902,"score_spread":0.2880024625079286,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4281391185","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0018335892,0.9257188,0.04464007,0.006097266,0.0004943175,0.00004958314,0.00017537491,0.00022483691,0.020766199],"genre_scores_gemma":[0.029014649,0.9197568,0.039937954,0.0033395768,0.000599361,0.00013742497,0.0005190448,0.00010572491,0.0065894825],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99937254,0.00022349288,0.000059238413,0.000100281795,0.0002160572,0.00002833251],"domain_scores_gemma":[0.9987256,0.00079706055,0.0000857103,0.0001317503,0.00021842895,0.000041357718],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002063363,0.0005905987,0.0010272498,0.0021320404,0.00030260216,0.0036372964,0.001677463,0.0015634611,0.0046263],"category_scores_gemma":[0.0051153013,0.0003253378,0.00036390807,0.002731529,0.00243567,0.008800105,0.0015586881,0.0021680272,0.0038016108],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00004168548,0.000031318254,0.00013890362,0.004629737,0.000050393322,0.000057819714,0.0002703795,0.0008172973,0.0017334505,0.16038145,0.013348637,0.818499],"study_design_scores_gemma":[0.000022374863,0.000056824407,0.0005232891,0.0041166786,0.000047748475,0.00038638277,0.0002753409,0.0012007544,0.001167881,0.17794834,0.8142192,0.00003516222],"about_ca_topic_score_codex":0.0013765837,"about_ca_topic_score_gemma":0.0013162695,"teacher_disagreement_score":0.0046263,"about_ca_system_score_codex":0.00087436684,"about_ca_system_score_gemma":0.0021317676,"threshold_uncertainty_score":0.015476525},"labels":[],"label_agreement":null},{"id":"W4281673866","doi":"10.1145/3529372.3533280","title":"Integration of text and geospatial search for hydrographic datasets using the lucene search library","year":2022,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Geospatial analysis; Computer science; Information retrieval; Hydrography; Index (typography); Metadata; World Wide Web; Geography; Cartography","score_opus":0.031095303217114966,"score_gpt":0.31011445904433577,"score_spread":0.2790191558272208,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4281673866","genre_codex":"software","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024439173,0.0018448879,0.31819546,0.0015153557,0.00022504867,0.0008336607,0.13175863,0.46936017,0.05182761],"genre_scores_gemma":[0.12855285,0.0019288692,0.45222524,0.0015147357,0.00023876123,0.0014147228,0.3378042,0.038779166,0.037541464],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99819,0.00023393535,0.00026229993,0.00040825809,0.0007855153,0.00012012471],"domain_scores_gemma":[0.997274,0.0014257128,0.00019546966,0.00041530275,0.00051035563,0.00017918584],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014706503,0.0014977269,0.0013283183,0.009083354,0.0011558402,0.0033399358,0.0018301769,0.0009393051,0.028806163],"category_scores_gemma":[0.0066642924,0.0006390148,0.0015074206,0.0058705234,0.0006207654,0.0068605994,0.0040526,0.0011001623,0.020132478],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012123596,0.00048064577,0.006797211,0.005186234,0.000721617,0.0023176451,0.0027841611,0.008707933,0.031956874,0.02524499,0.55222166,0.3623687],"study_design_scores_gemma":[0.00030797478,0.00023528375,0.012445461,0.00070969964,0.00023378718,0.001768213,0.0015650172,0.1242409,0.062465716,0.030374547,0.76500213,0.00065124215],"about_ca_topic_score_codex":0.013193152,"about_ca_topic_score_gemma":0.023835821,"teacher_disagreement_score":0.028806163,"about_ca_system_score_codex":0.0012798413,"about_ca_system_score_gemma":0.0014591529,"threshold_uncertainty_score":0.09636617},"labels":[],"label_agreement":null},{"id":"W4281688623","doi":"10.1075/tlrp.23.13mar","title":"Knowledge patterns in corpora","year":2022,"lang":"en","type":"book-chapter","venue":"Terminology and lexicography research and practice","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Paralanguage; Identification (biology); Computer science; Relation (database); Natural language processing; Artificial intelligence; Linguistics; Data mining","score_opus":0.12286483438369375,"score_gpt":0.4049428103849939,"score_spread":0.28207797600130013,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4281688623","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.059076607,0.016606031,0.48789948,0.014249963,0.0023975947,0.0013906524,0.050384942,0.008584704,0.35941008],"genre_scores_gemma":[0.20474109,0.011522747,0.6411926,0.0015620361,0.0006348868,0.0015335482,0.07660954,0.0024542261,0.059749335],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99452555,0.0019894794,0.00058530545,0.0010650053,0.0017053261,0.00012935257],"domain_scores_gemma":[0.9820494,0.011043572,0.0008251215,0.0034225357,0.0024030397,0.00025632203],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0041463105,0.0004848447,0.0006350235,0.010779207,0.0019439065,0.005937232,0.0016623448,0.0013196442,0.01729297],"category_scores_gemma":[0.021336634,0.00062606466,0.00040001865,0.023071837,0.0018530249,0.0072030365,0.0026245734,0.0014351952,0.005150616],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000084330335,0.000078019366,0.0039159525,0.0015201132,0.00005544629,0.0010014722,0.0033599185,0.0042278315,0.0028747441,0.2694606,0.11243056,0.60099095],"study_design_scores_gemma":[0.000022841823,0.00002244281,0.0032831836,0.0009432133,0.000029798171,0.0008581072,0.0020434684,0.013829377,0.004073372,0.11443707,0.86041534,0.000041722982],"about_ca_topic_score_codex":0.004941222,"about_ca_topic_score_gemma":0.00709078,"teacher_disagreement_score":0.01729297,"about_ca_system_score_codex":0.0023136067,"about_ca_system_score_gemma":0.0022268477,"threshold_uncertainty_score":0.057850778},"labels":[],"label_agreement":null},{"id":"W4281701033","doi":"10.63317/3t84wkkcstmg","title":"What a Creole Wants, What a Creole Needs","year":2022,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Creole language; Linguistics; Computer science; Resource (disambiguation); English-based creole languages; Artificial intelligence; Natural language processing; Foreign language; Modern language","score_opus":0.018655344221841282,"score_gpt":0.28920034605063405,"score_spread":0.27054500182879276,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4281701033","genre_codex":"empirical","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4055622,0.005613373,0.03842735,0.20829955,0.0014655486,0.00017438777,0.00123823,0.00077942887,0.33843994],"genre_scores_gemma":[0.87691724,0.002333707,0.019612221,0.021855565,0.00027643397,0.00010954598,0.00080290483,0.0003955088,0.07769694],"study_design_codex":"qualitative","study_design_gemma":"not_applicable","domain_scores_codex":[0.99688584,0.0017167921,0.0000976553,0.0004064976,0.00040880262,0.00048437857],"domain_scores_gemma":[0.992931,0.0025793503,0.00065913517,0.00047214454,0.0014514984,0.0019067726],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036302353,0.00040442307,0.00036863287,0.00092312077,0.005440614,0.0071500195,0.0009836588,0.0032283668,0.0103157405],"category_scores_gemma":[0.015361064,0.0002901011,0.00036213236,0.000814879,0.004869552,0.011141019,0.003359721,0.0035761506,0.0052639246],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002563409,0.00017052828,0.05165995,0.0007223841,0.000053892243,0.0026523366,0.46736655,0.0003234605,0.008517746,0.18318376,0.14365783,0.14143528],"study_design_scores_gemma":[0.000008658244,0.000055477718,0.009054408,0.00046833084,0.000025110054,0.0020397555,0.36010024,0.0006018154,0.0013580103,0.034012675,0.5921534,0.0001221831],"about_ca_topic_score_codex":0.0052959635,"about_ca_topic_score_gemma":0.007222352,"teacher_disagreement_score":0.0103157405,"about_ca_system_score_codex":0.0018804023,"about_ca_system_score_gemma":0.0018084147,"threshold_uncertainty_score":0.0345096},"labels":[],"label_agreement":null},{"id":"W4281737355","doi":"10.21203/rs.3.rs-1689966/v1","title":"Sentence Continuation Inference of Urdu Text by BERT Technique","year":2022,"lang":"en","type":"preprint","venue":"Research Square","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Urdu; Computer science; Natural language processing; Transformer; Artificial intelligence; Inference; Sentence; Cursive; Arabic; Linguistics; Engineering","score_opus":0.04341727310126934,"score_gpt":0.4113188378765017,"score_spread":0.36790156477523234,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4281737355","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15000728,0.0022705744,0.776069,0.0010802392,0.00058796426,0.0002989372,0.006334962,0.048424777,0.014926329],"genre_scores_gemma":[0.7874103,0.00049534213,0.18978679,0.00019976411,0.00026596076,0.00007936449,0.010199513,0.0010818205,0.010481115],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994349,0.00013564184,0.000035814675,0.00021802279,0.00011334468,0.000062225736],"domain_scores_gemma":[0.9985654,0.00063265924,0.000107319065,0.00022757353,0.00040345432,0.00006358183],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00063674775,0.0011220248,0.0005958791,0.0016652372,0.0004186312,0.0009260759,0.0010892614,0.00087888114,0.009617515],"category_scores_gemma":[0.0028125192,0.000342169,0.0007565471,0.0007869206,0.00038172086,0.0021909382,0.0006951115,0.0012446239,0.0057656425],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010133544,0.0001927243,0.005308797,0.00067188725,0.00010642871,0.0014039283,0.0006349924,0.064054176,0.050854106,0.012312237,0.035301637,0.8281457],"study_design_scores_gemma":[0.000019817078,0.00006768523,0.0014019064,0.00005635876,0.00002828526,0.00017434696,0.00014881276,0.96252805,0.016957778,0.0086904,0.009900793,0.000025814981],"about_ca_topic_score_codex":0.007838737,"about_ca_topic_score_gemma":0.009499141,"teacher_disagreement_score":0.009617515,"about_ca_system_score_codex":0.00080361246,"about_ca_system_score_gemma":0.0007605691,"threshold_uncertainty_score":0.032173753},"labels":[],"label_agreement":null},{"id":"W4281775169","doi":"10.14393/ufu.di.2022.5018","title":"Aplicativos móveis de interpretação automática: expectativas e percepções dos usuários brasileiros","year":2022,"lang":"pt","type":"dissertation","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Alberta-Pacific Forest Industries","keywords":"Interpreter; Perception; Computer science; Point (geometry); Service (business); Quality (philosophy); Interpretation (philosophy); Identification (biology); World Wide Web; Psychology; Business","score_opus":0.01631124316680539,"score_gpt":0.32775174158196996,"score_spread":0.3114404984151646,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4281775169","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9662824,0.00082902523,0.007869091,0.0018375608,0.00003626075,0.000075242795,0.00006637709,0.00016595089,0.02283817],"genre_scores_gemma":[0.995833,0.0005690151,0.0021138513,0.00017006895,0.000011944027,0.000019447865,0.000027426933,0.00002938218,0.0012258084],"study_design_codex":"qualitative","study_design_gemma":"observational","domain_scores_codex":[0.99453956,0.0023619805,0.00044103494,0.0006393289,0.0016419629,0.00037623188],"domain_scores_gemma":[0.9716969,0.018276552,0.002977136,0.001384951,0.004971625,0.00069280114],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008436236,0.00051307003,0.00040918836,0.0014930747,0.0017778736,0.0042409995,0.0005593288,0.0010953841,0.0015953265],"category_scores_gemma":[0.043867175,0.00061745895,0.00050952815,0.00096655986,0.004584591,0.0025232346,0.0019160623,0.0014579784,0.00044472024],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00042491077,0.00017568638,0.17770128,0.0006563603,0.00008211278,0.0014841199,0.6873277,0.00066214066,0.018322298,0.00669043,0.0017137572,0.10475924],"study_design_scores_gemma":[0.000059914208,0.0006396315,0.22866707,0.001018305,0.00030407117,0.0021049161,0.68787175,0.011418884,0.0114592705,0.009245398,0.046891697,0.00031900516],"about_ca_topic_score_codex":0.04638513,"about_ca_topic_score_gemma":0.04804489,"teacher_disagreement_score":0.04638513,"about_ca_system_score_codex":0.0017391425,"about_ca_system_score_gemma":0.0027557677,"threshold_uncertainty_score":0.09223032},"labels":[],"label_agreement":null},{"id":"W4283790576","doi":"10.1609/aaai.v36i10.21340","title":"From Fully Trained to Fully Random Embeddings: Improving Neural Machine Translation with Compact Word Embedding Tables","year":2022,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; Thomson Reuters (Canada)","funders":"","keywords":"Computer science; Embedding; Machine translation; Word (group theory); Word embedding; Natural language processing; Artificial intelligence; Context (archaeology); Artificial neural network; Task (project management); Key (lock); Language model; Speech recognition; Mathematics","score_opus":0.03792903244196088,"score_gpt":0.29245507295840056,"score_spread":0.2545260405164397,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4283790576","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1300249,0.002354999,0.8464744,0.0006742985,0.00043589674,0.00012101975,0.00091318495,0.014710803,0.004290365],"genre_scores_gemma":[0.59752625,0.001036682,0.3869179,0.0005391847,0.00023161381,0.00028040883,0.0052529573,0.0010086083,0.0072063454],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99918896,0.00031428895,0.00009158217,0.00018563237,0.00016307428,0.000056475812],"domain_scores_gemma":[0.99772996,0.0011174551,0.00015055173,0.0005784644,0.00037406894,0.00004954599],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001331958,0.0012973568,0.0011103975,0.0006773446,0.00031266978,0.0012776094,0.0012841919,0.0010317115,0.0038294645],"category_scores_gemma":[0.007882078,0.0005212794,0.00063184835,0.0011737149,0.0004968648,0.0048467754,0.0014303412,0.0014053214,0.003698566],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005528746,0.00027292673,0.0010769105,0.00029342485,0.00013243099,0.00028219927,0.00017543788,0.32911032,0.016793348,0.009971699,0.012049641,0.6292888],"study_design_scores_gemma":[0.000038729824,0.00013021336,0.00018118,0.000019670008,0.000024208417,0.00007566519,0.000028248898,0.9816976,0.0059475964,0.009344322,0.00249516,0.000017439295],"about_ca_topic_score_codex":0.0021340197,"about_ca_topic_score_gemma":0.0033528209,"teacher_disagreement_score":0.0038294645,"about_ca_system_score_codex":0.00045843743,"about_ca_system_score_gemma":0.0006869913,"threshold_uncertainty_score":0.012810886},"labels":[],"label_agreement":null},{"id":"W4284891255","doi":"10.1017/s1351324922000298","title":"Neural automated writing evaluation for Korean L2 writing","year":2022,"lang":"en","type":"article","venue":"Natural Language Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"National Research Foundation of Korea; National Research Foundation","keywords":"Computer science; Fluency; Parsing; Artificial intelligence; Natural language processing; Complement (music); Artificial neural network; Reliability (semiconductor); Linguistics","score_opus":0.009429505732520903,"score_gpt":0.2826202360865886,"score_spread":0.2731907303540677,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4284891255","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.85569,0.0004251754,0.13131659,0.00021559361,0.00012646602,0.00021881271,0.00076836755,0.006594815,0.0046441066],"genre_scores_gemma":[0.95234835,0.00008083415,0.04262829,0.000064550186,0.000013855467,0.00009429042,0.001001994,0.00007833339,0.0036895254],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9989942,0.0003936054,0.000086775726,0.00024162781,0.00021965493,0.000064189095],"domain_scores_gemma":[0.99698323,0.0012903925,0.00019924615,0.00035344635,0.0010623903,0.000111369445],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017273132,0.0005999577,0.00047576678,0.00075606746,0.00029553022,0.0008095186,0.00059963844,0.0005294825,0.0030766595],"category_scores_gemma":[0.0063184286,0.00016785125,0.0002603517,0.00041797885,0.00021497361,0.0012159623,0.00087762426,0.0006184797,0.0009598836],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00085651595,0.00076151994,0.014976845,0.00028290268,0.00013653561,0.00036187185,0.00036682075,0.039574616,0.05595846,0.0008169355,0.0050155595,0.8808915],"study_design_scores_gemma":[0.000046885503,0.0003766963,0.013788265,0.00002541867,0.00003811423,0.0001306478,0.00022279918,0.936131,0.046965178,0.00057712325,0.0016618242,0.000036049092],"about_ca_topic_score_codex":0.0032810208,"about_ca_topic_score_gemma":0.0046982537,"teacher_disagreement_score":0.0032810208,"about_ca_system_score_codex":0.00052044314,"about_ca_system_score_gemma":0.00042480358,"threshold_uncertainty_score":0.01029247},"labels":[],"label_agreement":null},{"id":"W4285084440","doi":"10.1038/s41587-022-01369-0","title":"Standardized annotation of translated open reading frames","year":2022,"lang":"en","type":"letter","venue":"Nature Biotechnology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":250,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Sherbrooke","funders":"National Cancer Institute; National Human Genome Research Institute; National Institute on Aging; RIKEN; Agència de Gestió d'Ajuts Universitaris i de Recerca; Medical Research Council; Canadian Institutes of Health Research; Directorate for Biological Sciences; National Institutes of Health; European Molecular Biology Laboratory; Russian Science Foundation; Institució Catalana de Recerca i Estudis Avançats; National Institute of General Medical Sciences; University College Cork; University of Leeds; Universitat Pompeu Fabra; Fondation Leducq; Stowers Institute for Medical Research; Australian Government; Searle Scholars Program; National Science Foundation; Science Foundation Ireland; Biotechnology and Biological Sciences Research Council; Université de Sherbrooke; Wellcome Trust; European Commission; University of California, Irvine; Broad Institute; University of Pittsburgh; Howard Hughes Medical Institute; Staatssekretariat für Bildung, Forschung und Innovation; Musella Foundation For Brain Tumor Research and Information; Agencia Estatal de Investigación; Computer Science and Artificial Intelligence Laboratory, Massachusetts Institute of Technology; Alex's Lemonade Stand Foundation for Childhood Cancer; Yale University","keywords":"Annotation; Open reading frame; Reading (process); Computer science; Information retrieval; Computational biology; Natural language processing; Biology; Artificial intelligence; Linguistics; Genetics; Philosophy; Peptide sequence","score_opus":0.009975389999421257,"score_gpt":0.2940358937373858,"score_spread":0.28406050373796454,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4285084440","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0322214,0.05416159,0.3994699,0.32901978,0.0579285,0.0014551827,0.025825562,0.0076364935,0.092281595],"genre_scores_gemma":[0.088313214,0.034077074,0.59390277,0.14061701,0.022445256,0.0018040172,0.069702365,0.0026676296,0.046470724],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9929289,0.0022184073,0.0011803523,0.000928584,0.0023678632,0.00037583185],"domain_scores_gemma":[0.9837516,0.0047782427,0.0018507787,0.0033738278,0.0056420537,0.0006035283],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0071532996,0.00072797615,0.0008087646,0.0028587992,0.0011620723,0.0021568153,0.0015672604,0.0025742622,0.0032965026],"category_scores_gemma":[0.014136574,0.00042993348,0.00058759144,0.0025803985,0.0014469485,0.0016510095,0.0013463649,0.0050328984,0.009486219],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00034850952,0.000109369794,0.0032198527,0.0011909825,0.000039702332,0.0019746234,0.00051210774,0.00073462253,0.08339601,0.06544748,0.48373282,0.35929394],"study_design_scores_gemma":[0.000024642857,0.000105050196,0.002354124,0.00030619188,0.00002028027,0.0012425111,0.00017090679,0.0016268578,0.02074423,0.0144909555,0.9588595,0.000054808457],"about_ca_topic_score_codex":0.0014565401,"about_ca_topic_score_gemma":0.0036998405,"teacher_disagreement_score":0.0071532996,"about_ca_system_score_codex":0.0024770552,"about_ca_system_score_gemma":0.0031819297,"threshold_uncertainty_score":0.03783071},"labels":[],"label_agreement":null},{"id":"W4285105102","doi":"10.18653/v1/2022.acl-long.17","title":"CipherDAug: Ciphertext based Data Augmentation for Neural Machine Translation","year":2022,"lang":"en","type":"article","venue":"Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Natural Sciences and Engineering Research Council of Canada; Ministère de la Défense Nationale","keywords":"Computer science; Ciphertext; Machine translation; Plaintext; Artificial neural network; Leverage (statistics); Artificial intelligence; Source code; Transformer; Algorithm; Encryption; Programming language","score_opus":0.019254073827921773,"score_gpt":0.2733220621644697,"score_spread":0.2540679883365479,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4285105102","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018418422,0.00040523792,0.9714588,0.00033587858,0.00024277816,0.00016510134,0.0006552745,0.0054908567,0.0028276613],"genre_scores_gemma":[0.29863688,0.00042408129,0.6883707,0.00044641332,0.00019876308,0.00068152,0.003710118,0.00080832926,0.0067231483],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987312,0.0004768834,0.00010130201,0.0002869475,0.0003131901,0.00009042932],"domain_scores_gemma":[0.9972916,0.00084002747,0.0001709095,0.0011914232,0.00042029202,0.00008562953],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016617917,0.0012026975,0.0009885291,0.0008206535,0.0007107859,0.0010652939,0.0019853248,0.0010134426,0.0046160044],"category_scores_gemma":[0.0060901167,0.00047067026,0.0008990005,0.001415835,0.0013026772,0.0029639264,0.0032678281,0.002611245,0.0033960475],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011459211,0.0005712493,0.0018695066,0.000540751,0.00015938401,0.00037813198,0.00046579566,0.11923973,0.080787376,0.045204554,0.022437764,0.72719985],"study_design_scores_gemma":[0.00010217065,0.00032191348,0.00046068194,0.000052015024,0.000039235674,0.00030469405,0.000078912286,0.85250854,0.081979305,0.042583074,0.021510229,0.000059217098],"about_ca_topic_score_codex":0.0011876192,"about_ca_topic_score_gemma":0.0018796417,"teacher_disagreement_score":0.0046160044,"about_ca_system_score_codex":0.0006290544,"about_ca_system_score_gemma":0.001298687,"threshold_uncertainty_score":0.015442014},"labels":[],"label_agreement":null},{"id":"W4285116834","doi":"10.18653/v1/2022.ltedi-1.6","title":"Detoxifying Language Models with a Toxic Corpus","year":2022,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute; University of Toronto","funders":"Vector Institute; Canadian Institute for Advanced Research","keywords":"Debiasing; Computer science; Resource (disambiguation); Natural language processing; Process (computing); Artificial intelligence; Programming language; Psychology","score_opus":0.014246429890326302,"score_gpt":0.24809820249637224,"score_spread":0.23385177260604595,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4285116834","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1900945,0.00051239715,0.8019194,0.0007285209,0.0000975016,0.00012413594,0.00026139556,0.0032251645,0.0030369936],"genre_scores_gemma":[0.68607116,0.00047945313,0.30624908,0.0004796713,0.000095986776,0.0002882696,0.0012356364,0.00052720535,0.004573471],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988644,0.00056232885,0.00007190255,0.00021834766,0.00022028382,0.00006276418],"domain_scores_gemma":[0.99315417,0.0044964855,0.00038036463,0.0011255202,0.00074903347,0.00009442316],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029315378,0.0009329448,0.00059660076,0.00093369756,0.0006469767,0.0013661679,0.000959,0.00094685005,0.0025651983],"category_scores_gemma":[0.017510748,0.00056177774,0.00068638864,0.0006699561,0.0008905584,0.0025896635,0.0023462432,0.0020176664,0.0010525545],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00071694073,0.00030558385,0.0041177645,0.00042042191,0.00015752987,0.00077684806,0.0009009735,0.5851732,0.0698877,0.034804728,0.0031176656,0.2996207],"study_design_scores_gemma":[0.00003061643,0.00009738217,0.00026390734,0.000020491621,0.00003088491,0.000078698926,0.000096086675,0.9625748,0.019264203,0.015264618,0.0022574677,0.00002094494],"about_ca_topic_score_codex":0.0019515881,"about_ca_topic_score_gemma":0.003453722,"teacher_disagreement_score":0.0029315378,"about_ca_system_score_codex":0.0005934343,"about_ca_system_score_gemma":0.0011457065,"threshold_uncertainty_score":0.015503645},"labels":[],"label_agreement":null},{"id":"W4285128742","doi":"10.18653/v1/2022.computel-1","title":"Proceedings of the Fifth Workshop on the Use of Computational Methods in the Study of Endangered Languages","year":2022,"lang":"en","type":"paratext","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University; First Nations University of Canada; University of Saskatchewan; University of Alberta; National Research Council Canada; University of British Columbia","funders":"Consejo Nacional de Ciencia, Tecnología e Innovación Tecnológica","keywords":"Endangered species; Computer science; Programming language; Ecology; Biology","score_opus":0.07713893589199727,"score_gpt":0.38449391717603243,"score_spread":0.30735498128403516,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4285128742","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016527427,0.103704505,0.7154856,0.03893371,0.013106406,0.0003629223,0.00151683,0.0011831169,0.109179474],"genre_scores_gemma":[0.11145306,0.07040739,0.6803495,0.005493994,0.008109451,0.00112253,0.005863536,0.002620458,0.11458002],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9941189,0.0041012797,0.0003053455,0.0006709617,0.0006109936,0.00019238605],"domain_scores_gemma":[0.97979224,0.016055178,0.0002946419,0.0018354675,0.0012166604,0.00080566487],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.014680897,0.0008991795,0.0014394994,0.0025594146,0.0011030455,0.0077800094,0.0025006211,0.0020120123,0.021499889],"category_scores_gemma":[0.01661896,0.0005909035,0.001455178,0.0023417862,0.0033539522,0.0053780265,0.004617076,0.003520937,0.0037484074],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00040061923,0.00037490795,0.0023559367,0.0019299987,0.00040642158,0.0006738931,0.003493072,0.0047970824,0.0039017957,0.25489765,0.17389615,0.55287254],"study_design_scores_gemma":[0.00006243662,0.00008181078,0.002795767,0.001431967,0.00013132482,0.00044079413,0.0010098692,0.009026607,0.0021933303,0.17930596,0.80345434,0.00006569238],"about_ca_topic_score_codex":0.0022211412,"about_ca_topic_score_gemma":0.002935464,"teacher_disagreement_score":0.021499889,"about_ca_system_score_codex":0.0017592877,"about_ca_system_score_gemma":0.0021999655,"threshold_uncertainty_score":0.07764089},"labels":[],"label_agreement":null},{"id":"W4285130250","doi":"10.1007/978-3-031-08473-7_41","title":"A BERT-Based Approach for Multilingual Discourse Connective Detection","year":2022,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Linguistics; Natural language processing; Philosophy","score_opus":0.01765158086032472,"score_gpt":0.2870702121837653,"score_spread":0.2694186313234406,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4285130250","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0108862715,0.0005494667,0.9598691,0.00031095094,0.00021993712,0.00029537635,0.0019383148,0.01633271,0.00959781],"genre_scores_gemma":[0.12779447,0.00031548107,0.84565365,0.00019243965,0.00018836725,0.00029617097,0.0047955266,0.0011577705,0.019606134],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99834883,0.0003089637,0.00009882393,0.00039958215,0.0006745782,0.00016927732],"domain_scores_gemma":[0.9973571,0.00088506023,0.00013364374,0.00036801325,0.0010779372,0.00017833064],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012579287,0.0017662462,0.001417632,0.0053465464,0.0018645631,0.002626805,0.002540072,0.0017249286,0.021227255],"category_scores_gemma":[0.0035535204,0.0008609355,0.001062029,0.003729661,0.0006779575,0.0033046033,0.0032914006,0.0017063292,0.013266233],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001044174,0.00030396975,0.0020570667,0.0003801719,0.00012521367,0.00056425337,0.0003560378,0.009216516,0.07012103,0.021117626,0.028811403,0.8659026],"study_design_scores_gemma":[0.000059781716,0.00021221218,0.0023188945,0.00008494289,0.00011308358,0.00080775283,0.00046202805,0.88308376,0.042081397,0.032797854,0.03786533,0.000112985814],"about_ca_topic_score_codex":0.009122146,"about_ca_topic_score_gemma":0.018373434,"teacher_disagreement_score":0.021227255,"about_ca_system_score_codex":0.00094162056,"about_ca_system_score_gemma":0.0018066535,"threshold_uncertainty_score":0.07101232},"labels":[],"label_agreement":null},{"id":"W4285160596","doi":"10.18653/v1/2022.acl-tutorials","title":"Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics: Tutorial Abstracts","year":2022,"lang":"en","type":"paratext","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"NIH Office of the Director; National Science Foundation; Office of the Director of National Intelligence; Advanced Research Projects Agency; Defense Advanced Research Projects Agency; Intelligence Advanced Research Projects Activity","keywords":"Association (psychology); Computational linguistics; Computer science; Cognitive science; Natural language processing; Linguistics; Psychology; Philosophy; Epistemology","score_opus":0.009125582685065676,"score_gpt":0.27013480236840987,"score_spread":0.2610092196833442,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4285160596","genre_codex":"editorial","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011289707,0.12907971,0.16093245,0.18884134,0.27031204,0.0020098162,0.021439869,0.01438139,0.20171362],"genre_scores_gemma":[0.022322936,0.056959555,0.08398637,0.019040193,0.055133864,0.0032145502,0.046430547,0.009948155,0.7029639],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9960057,0.00140089,0.000448201,0.0008189434,0.0010223753,0.00030389737],"domain_scores_gemma":[0.9761702,0.007141683,0.00068577775,0.0016898633,0.010075317,0.0042372574],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01355398,0.001906495,0.0026490288,0.004545952,0.0019124637,0.0073365658,0.0028920604,0.0028580462,0.16248919],"category_scores_gemma":[0.020836141,0.0008487461,0.0010708715,0.002667788,0.0018481753,0.0076095373,0.004266901,0.004368048,0.13632032],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005680636,0.000043128177,0.00018397083,0.00025769853,0.000016046059,0.000049557308,0.00010057732,0.00006736117,0.0004770193,0.0007867884,0.9274318,0.07052916],"study_design_scores_gemma":[0.000020576947,0.000028727674,0.00059073867,0.0004497398,0.000018937162,0.00014428626,0.00017358299,0.0008464887,0.00024891543,0.0023367077,0.9951184,0.00002298076],"about_ca_topic_score_codex":0.0035969592,"about_ca_topic_score_gemma":0.0061446438,"teacher_disagreement_score":0.16248919,"about_ca_system_score_codex":0.0025137162,"about_ca_system_score_gemma":0.0075223804,"threshold_uncertainty_score":0.54358053},"labels":[],"label_agreement":null},{"id":"W4285179082","doi":"10.18653/v1/2022.acl-tutorials.8","title":"Natural Language Processing for Multilingual Task-Oriented Dialogue","year":2022,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; Mila - Quebec Artificial Intelligence Institute","funders":"","keywords":"Computer science; Task (project management); German; Dialog box; Natural language processing; Focus (optics); Multilingual Education; Artificial intelligence; Multilingualism; World Wide Web; Linguistics","score_opus":0.00979021322596528,"score_gpt":0.28704003873474704,"score_spread":0.27724982550878174,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4285179082","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008942506,0.0035910057,0.9612552,0.0019297675,0.0004022219,0.0003621941,0.0032014954,0.009276711,0.01103891],"genre_scores_gemma":[0.15425007,0.0030848153,0.8169981,0.00073870143,0.00029055326,0.0010804638,0.013711297,0.0007355465,0.009110557],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9975068,0.0012557815,0.00021423023,0.0005184738,0.0003682904,0.00013636898],"domain_scores_gemma":[0.99745387,0.0014922,0.00013419671,0.0004290308,0.00040393116,0.0000867888],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030465587,0.0011612748,0.00066166435,0.0014089093,0.0010471396,0.0030221115,0.001325307,0.0013346233,0.009953715],"category_scores_gemma":[0.0081229955,0.00045187824,0.001374705,0.0012405738,0.00097018114,0.004947185,0.0028595864,0.0024308295,0.006049254],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003525279,0.00024275422,0.001118925,0.0023817215,0.00019453741,0.00041172022,0.0022682783,0.030727854,0.022444647,0.08682547,0.071304426,0.7817271],"study_design_scores_gemma":[0.00006139527,0.00015388297,0.002086666,0.0004775852,0.00007801459,0.00051390927,0.0014022412,0.43519047,0.018393682,0.30906415,0.23244524,0.0001326988],"about_ca_topic_score_codex":0.0037768558,"about_ca_topic_score_gemma":0.0069900765,"teacher_disagreement_score":0.009953715,"about_ca_system_score_codex":0.0012189896,"about_ca_system_score_gemma":0.0016484492,"threshold_uncertainty_score":0.033298492},"labels":[],"label_agreement":null},{"id":"W4285182272","doi":"10.18653/v1/2022.acl-long.507","title":"Requirements and Motivations of Low-Resource Speech Synthesis for Language Revitalization","year":2022,"lang":"en","type":"article","venue":"Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"University of Edinburgh; UK Research and Innovation","keywords":"Naturalness; Computer science; Speech synthesis; Speech technology; Field (mathematics); Resource (disambiguation); Natural language processing; Indigenous; Artificial intelligence; Speech recognition","score_opus":0.008099032325082663,"score_gpt":0.24823653900384493,"score_spread":0.24013750667876227,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4285182272","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.24370739,0.0026049905,0.7008902,0.013151184,0.00019601498,0.0006928747,0.0006409205,0.0033408955,0.034775518],"genre_scores_gemma":[0.7221824,0.0010827172,0.26560214,0.00072468905,0.00016830872,0.0007387871,0.00089342025,0.0007408724,0.007866682],"study_design_codex":"design_other","study_design_gemma":"qualitative","domain_scores_codex":[0.99676955,0.001359247,0.00020708292,0.00064964,0.0007253194,0.000289116],"domain_scores_gemma":[0.9823829,0.010439006,0.0006256711,0.002186932,0.0036339257,0.0007315687],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006722916,0.0008219084,0.00080889487,0.0006087764,0.0010002995,0.002553062,0.0023076083,0.0015480923,0.00402193],"category_scores_gemma":[0.013715417,0.00066666387,0.00048342563,0.00048856216,0.0022419558,0.0044440273,0.0029560148,0.0028952893,0.002642525],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001958206,0.000720873,0.0067322417,0.0022937302,0.00008013443,0.0016069603,0.00658735,0.079509184,0.3139031,0.14545508,0.00569699,0.43545625],"study_design_scores_gemma":[0.00027571237,0.0018918329,0.00734536,0.0005246767,0.0001287066,0.002141968,0.007188396,0.46442994,0.24179585,0.11898113,0.15501565,0.00028077624],"about_ca_topic_score_codex":0.0038208647,"about_ca_topic_score_gemma":0.0027695587,"teacher_disagreement_score":0.006722916,"about_ca_system_score_codex":0.0013368598,"about_ca_system_score_gemma":0.002460149,"threshold_uncertainty_score":0.035554588},"labels":[],"label_agreement":null},{"id":"W4285280865","doi":"10.18653/v1/2022.lchange-1.19","title":"UAlberta at LSCDiscovery: Lexical Semantic Change Detection via Word Sense Disambiguation","year":2022,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Alberta Innovates; Alberta Machine Intelligence Institute","keywords":"Computer science; Semantic change; Task (project management); Natural language processing; Word (group theory); Word-sense disambiguation; Artificial intelligence; Semantics (computer science); Security token; SemEval; Frame (networking); Linguistics; WordNet; Programming language","score_opus":0.020984367770836326,"score_gpt":0.2614598771903304,"score_spread":0.2404755094194941,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4285280865","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06757113,0.008034628,0.4439158,0.0071688895,0.0026186998,0.0008941431,0.030635545,0.38834077,0.050820425],"genre_scores_gemma":[0.2855001,0.002728934,0.560153,0.0028524445,0.00062465714,0.00089709123,0.07434866,0.014865049,0.05803004],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9968772,0.00044093936,0.00021526814,0.0014416003,0.0007816027,0.00024344596],"domain_scores_gemma":[0.9965258,0.00060267496,0.00020748432,0.0013770552,0.00095464225,0.00033227657],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028076249,0.0024851924,0.0018142436,0.006913879,0.0016497046,0.0055834004,0.0022832484,0.0020467248,0.02067616],"category_scores_gemma":[0.008830299,0.0010830825,0.0016212356,0.0035155436,0.0011566702,0.0057047657,0.006011816,0.0026638831,0.027499115],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006578778,0.00037981948,0.012160898,0.0008938788,0.00026440425,0.0008750269,0.0010351054,0.0039685746,0.027523411,0.008079443,0.1659643,0.7781973],"study_design_scores_gemma":[0.0004145436,0.0004010743,0.024675852,0.0005138985,0.00039896925,0.0024509034,0.001522801,0.2417692,0.084765404,0.04879855,0.59379596,0.0004928662],"about_ca_topic_score_codex":0.021861676,"about_ca_topic_score_gemma":0.01815954,"teacher_disagreement_score":0.021861676,"about_ca_system_score_codex":0.001735616,"about_ca_system_score_gemma":0.0032950272,"threshold_uncertainty_score":0.06916869},"labels":[],"label_agreement":null},{"id":"W4285288434","doi":"10.18653/v1/2022.acl-long.233","title":"LAGr: Label Aligned Graphs for Better Systematic Generalization in Semantic Parsing","year":2022,"lang":"en","type":"article","venue":"Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Mila - Quebec Artificial Intelligence Institute; McGill University; Canadian Institute for Advanced Research","funders":"Samsung; Canadian Institute for Advanced Research; Microsoft Research","keywords":"Computer science; Parsing; Generalization; Artificial intelligence; Natural language processing; Inference; Graph; Sequence (biology); Theoretical computer science; Mathematics","score_opus":0.008235714887199415,"score_gpt":0.24240252831051956,"score_spread":0.23416681342332016,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4285288434","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0046967035,0.0001124967,0.9792909,0.00030492613,0.00005201177,0.00007756748,0.0005560963,0.0139561035,0.0009531537],"genre_scores_gemma":[0.100837894,0.00019834714,0.8866203,0.0007161041,0.00008204072,0.00027750788,0.005001573,0.0036385984,0.0026275762],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99702054,0.0012804712,0.00011275644,0.0010790826,0.0003895805,0.00011747479],"domain_scores_gemma":[0.99251187,0.0041565006,0.000393822,0.0021678004,0.0006069339,0.00016301541],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003422579,0.0021295748,0.0010807166,0.0019094361,0.0012923613,0.0016272585,0.0026749873,0.0026228938,0.0070715416],"category_scores_gemma":[0.012898197,0.0012042985,0.002252055,0.0016993282,0.0021101546,0.0074471314,0.0037732655,0.0049985168,0.004437056],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00054539996,0.0004426655,0.0025454564,0.00064204703,0.00025993126,0.00090601714,0.001786633,0.15884535,0.031211436,0.11455226,0.05214482,0.63611794],"study_design_scores_gemma":[0.00007794424,0.00007896793,0.00046082374,0.00006396568,0.000059089565,0.00021588829,0.00018624797,0.6642701,0.01242969,0.30647278,0.01562104,0.00006348714],"about_ca_topic_score_codex":0.0037584994,"about_ca_topic_score_gemma":0.010585495,"teacher_disagreement_score":0.0070715416,"about_ca_system_score_codex":0.0011377209,"about_ca_system_score_gemma":0.0022500535,"threshold_uncertainty_score":0.023656726},"labels":[],"label_agreement":null},{"id":"W4285289306","doi":"10.18653/v1/2022.acl-long.47","title":"AraT5: Text-to-Text Transformers for Arabic Language Generation","year":2022,"lang":"en","type":"article","venue":"Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":97,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Transformer; Arabic; Natural language processing; Artificial intelligence; Benchmark (surveying); Language model; Variety (cybernetics); Set (abstract data type); Test set; Programming language; Linguistics; Engineering","score_opus":0.008532702958726601,"score_gpt":0.24964838578847798,"score_spread":0.24111568282975138,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4285289306","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.033351175,0.00080876186,0.8778697,0.0006819098,0.0005744269,0.0006549584,0.0038324408,0.06872578,0.01350088],"genre_scores_gemma":[0.44046667,0.0006196779,0.5167119,0.0006703559,0.0001989583,0.0010641201,0.0178213,0.0031659575,0.019280983],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990827,0.00032383428,0.00007534253,0.00027342752,0.00016360443,0.00008095378],"domain_scores_gemma":[0.9985073,0.00060502795,0.000076108015,0.00036227444,0.00034267348,0.00010659648],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020284362,0.0017971967,0.0005942451,0.0015668672,0.00062737515,0.0016427587,0.0028660109,0.0013350708,0.017690111],"category_scores_gemma":[0.005529988,0.0005016063,0.001642474,0.0009194547,0.00066964043,0.004039873,0.0026601371,0.002647564,0.010081465],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00062560965,0.00044589167,0.0024851535,0.0004957362,0.00018212532,0.0002641901,0.00033534667,0.10426064,0.013580848,0.01888315,0.0576234,0.80081797],"study_design_scores_gemma":[0.00008706168,0.00023915913,0.0004315469,0.000036903242,0.000047497535,0.00016591493,0.00009638246,0.9402425,0.020801267,0.017595895,0.020217352,0.000038578193],"about_ca_topic_score_codex":0.0030657682,"about_ca_topic_score_gemma":0.0039725676,"teacher_disagreement_score":0.017690111,"about_ca_system_score_codex":0.0010167486,"about_ca_system_score_gemma":0.001528938,"threshold_uncertainty_score":0.059179366},"labels":[],"label_agreement":null},{"id":"W4285554708","doi":"10.18653/v1/2021.sigmorphon-1","title":"Proceedings of the 18th SIGMORPHON Workshop on Computational Research in Phonetics, Phonology, and Morphology","year":2021,"lang":"en","type":"paratext","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Vector Institute; Canadian Institute for Advanced Research; National Science Foundation","keywords":"Phonology; Phonetics; Morphology (biology); Computer science; Linguistics; Natural language processing; Artificial intelligence; Philosophy; Geology","score_opus":0.04664853304572412,"score_gpt":0.3512597287849922,"score_spread":0.30461119573926804,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4285554708","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.053296786,0.0765125,0.58493197,0.06682853,0.052749205,0.0009350461,0.0129058575,0.012306695,0.13953334],"genre_scores_gemma":[0.12653488,0.038840424,0.36907592,0.0056734155,0.015547959,0.0010195934,0.037854712,0.005985022,0.3994681],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99771845,0.0008791857,0.00017519755,0.00050106266,0.000519058,0.00020708759],"domain_scores_gemma":[0.99426585,0.0023441655,0.00018170486,0.0012475685,0.0011376012,0.000823153],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005547157,0.0015480939,0.001808023,0.0025260288,0.0012287871,0.006775075,0.0025742499,0.0021314055,0.056159005],"category_scores_gemma":[0.0077092005,0.00059003494,0.0011703382,0.0018728542,0.0019647242,0.006584967,0.0036798164,0.0033511706,0.0225993],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00081174576,0.0006025624,0.0016765865,0.000534966,0.00016292123,0.00040085922,0.00053927436,0.0024843412,0.008066394,0.013823644,0.45986223,0.5110344],"study_design_scores_gemma":[0.00014637003,0.00022919357,0.0045512742,0.00057751563,0.00014263454,0.00067073025,0.0007532706,0.03255785,0.008647771,0.045524027,0.9060959,0.00010337936],"about_ca_topic_score_codex":0.003745109,"about_ca_topic_score_gemma":0.008284594,"teacher_disagreement_score":0.056159005,"about_ca_system_score_codex":0.0017580071,"about_ca_system_score_gemma":0.002607342,"threshold_uncertainty_score":0.18787056},"labels":[],"label_agreement":null},{"id":"W4286560902","doi":"10.3758/s13428-022-01912-6","title":"Concreteness ratings for 62,000 English multiword expressions","year":2022,"lang":"en","type":"article","venue":"Behavior Research Methods","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":34,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada","keywords":"Concreteness; Computer science; Psychology; Word (group theory); Meaning (existential); Linguistics; Cognitive psychology; Natural language processing","score_opus":0.23104756777532953,"score_gpt":0.5707877905792769,"score_spread":0.33974022280394733,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4286560902","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9907775,0.00030765426,0.0010421615,0.000057391684,0.00004491838,0.00008742112,0.000925883,0.000072545045,0.006684556],"genre_scores_gemma":[0.98316956,0.00035757714,0.0043069087,0.00013482978,0.000057449906,0.00022356554,0.0031358795,0.0001231833,0.008491064],"study_design_codex":"observational","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99821186,0.00065103243,0.00032327135,0.00029602618,0.0004347491,0.000083137595],"domain_scores_gemma":[0.978625,0.0152660245,0.0017103668,0.0007969012,0.002962727,0.00063897873],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019946196,0.00036532906,0.0004029807,0.00092750514,0.0004148823,0.0005650772,0.00019918122,0.00066817517,0.0075169746],"category_scores_gemma":[0.02251662,0.00018155688,0.00031493057,0.0005529844,0.00030392,0.0009066702,0.0006420516,0.00048021937,0.0016733403],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.010555004,0.0015807605,0.550047,0.0019378276,0.00052204216,0.0010545402,0.018994877,0.0014422764,0.1376357,0.0021709253,0.012426301,0.2616328],"study_design_scores_gemma":[0.000100887686,0.0012455638,0.9788103,0.00010130699,0.00013036329,0.00070791057,0.0019823601,0.0018516382,0.007671286,0.00042901665,0.0069170906,0.000052302268],"about_ca_topic_score_codex":0.0010464261,"about_ca_topic_score_gemma":0.0018839148,"teacher_disagreement_score":0.0075169746,"about_ca_system_score_codex":0.00023756438,"about_ca_system_score_gemma":0.00014901742,"threshold_uncertainty_score":0.025146782},"labels":[],"label_agreement":null},{"id":"W4286713635","doi":"","title":"Extraction automatique de relations sémantiques d’hyperonymie et d’hyponymie dans un corpus métier","year":2021,"lang":"fr","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Berger (Canada)","funders":"","keywords":"Computer science; Artificial intelligence; Natural language processing","score_opus":0.015210053114646597,"score_gpt":0.2632322561098828,"score_spread":0.24802220299523622,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4286713635","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.32506597,0.003516514,0.5936612,0.0024223104,0.00079197244,0.00057640194,0.045265667,0.014066028,0.014633987],"genre_scores_gemma":[0.38745132,0.0017148196,0.5356889,0.00018912253,0.00024958962,0.0005827101,0.059626497,0.0022382343,0.012258859],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9974093,0.00052334357,0.0002989345,0.00089334603,0.00072441745,0.00015060873],"domain_scores_gemma":[0.9947855,0.0033492825,0.00027421885,0.0004148775,0.0010636727,0.00011249787],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011556074,0.0016723671,0.001399519,0.0062517785,0.002000525,0.003153008,0.0008240937,0.0015921162,0.006100875],"category_scores_gemma":[0.0069619096,0.0012538102,0.0013525706,0.004905103,0.00092346524,0.0037570803,0.0011544852,0.002358563,0.0035642139],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014177044,0.0003264292,0.009390749,0.0032532995,0.00044271967,0.0045031486,0.0060228137,0.0078050117,0.36560512,0.029582446,0.048411526,0.523239],"study_design_scores_gemma":[0.00059636805,0.00044382093,0.07467694,0.00063736684,0.0008687235,0.0068449085,0.007296956,0.24298896,0.3016207,0.026275441,0.33735815,0.0003917111],"about_ca_topic_score_codex":0.018567685,"about_ca_topic_score_gemma":0.019747118,"teacher_disagreement_score":0.018567685,"about_ca_system_score_codex":0.0013638904,"about_ca_system_score_gemma":0.0026284033,"threshold_uncertainty_score":0.036919236},"labels":[],"label_agreement":null},{"id":"W4286905174","doi":"10.18653/v1/2022.findings-emnlp.19","title":"Salient Phrase Aware Dense Retrieval: Can a Dense Retriever Imitate a Sparse One?","year":2022,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":36,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Phrase; Computer science; Salient; Artificial intelligence; Natural language processing; Labrador Retriever; Information retrieval; Medicine","score_opus":0.02471533952713897,"score_gpt":0.25931738518626657,"score_spread":0.2346020456591276,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4286905174","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.31709749,0.006974966,0.59658146,0.00770233,0.0015849414,0.00052206457,0.001348376,0.017409408,0.050778978],"genre_scores_gemma":[0.83733916,0.0014630056,0.14251895,0.0013670325,0.00057592813,0.00010021483,0.0013084352,0.0009129272,0.014414354],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992493,0.00017200739,0.000046767178,0.00019927554,0.00020231525,0.0001303259],"domain_scores_gemma":[0.9968527,0.0011253197,0.00018748327,0.0011166619,0.0005014203,0.00021641226],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016128984,0.0006244724,0.00149037,0.00096327934,0.00092288514,0.0026062587,0.001919517,0.0019070939,0.007887177],"category_scores_gemma":[0.009181601,0.0006085531,0.00046834772,0.0010122138,0.0012122812,0.010793729,0.0024735525,0.0013325138,0.0074251704],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0032099371,0.0009691234,0.0062191184,0.0008566311,0.00022434362,0.0008571787,0.0013940338,0.01877316,0.16003014,0.03572308,0.065193646,0.7065497],"study_design_scores_gemma":[0.0004641838,0.0014192307,0.0062392484,0.0001147378,0.00039792812,0.0021737039,0.0019570687,0.7519554,0.069804505,0.12436037,0.04088464,0.00022906662],"about_ca_topic_score_codex":0.0035229616,"about_ca_topic_score_gemma":0.005553595,"teacher_disagreement_score":0.007887177,"about_ca_system_score_codex":0.00039025026,"about_ca_system_score_gemma":0.00092705677,"threshold_uncertainty_score":0.026385248},"labels":[],"label_agreement":null},{"id":"W4286952897","doi":"10.48550/arxiv.2109.14788","title":"Tipping the Scales: A Corpus-Based Reconstruction of Adjective Scales in\\n the McGill Pain Questionnaire","year":2021,"lang":"","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Adjective; Adjective check list; McGill Pain Questionnaire; Psychology; Linguistics; Natural language processing; Visual analogue scale; Social psychology; Computer science; Medicine; Physical therapy; Philosophy; Noun","score_opus":0.03874347221352953,"score_gpt":0.20128992345956553,"score_spread":0.162546451246036,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4286952897","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.22564024,0.0017359875,0.64914143,0.0026534074,0.001216673,0.0030803876,0.06540006,0.0082209,0.042910945],"genre_scores_gemma":[0.3025472,0.0007289487,0.6188668,0.0003520088,0.00012282096,0.0031166032,0.06468721,0.0015224147,0.008055998],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99768996,0.0010981137,0.00020658561,0.00054919324,0.0003772736,0.00007884649],"domain_scores_gemma":[0.9918344,0.005830832,0.00035381957,0.0008853609,0.000992268,0.00010338545],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003244381,0.00097288407,0.00043661857,0.004622143,0.0016037708,0.0026199615,0.0011423532,0.00074914156,0.00835389],"category_scores_gemma":[0.016717443,0.0005149119,0.0008352729,0.004743252,0.0014191923,0.0021306844,0.0020237023,0.0020263465,0.0042980146],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00061236543,0.00049893424,0.03417083,0.0026475159,0.00017896094,0.0028596525,0.055249665,0.008583898,0.036972467,0.08256781,0.11858376,0.65707415],"study_design_scores_gemma":[0.00026572982,0.000282137,0.09074789,0.0012786002,0.00024856365,0.0037022547,0.039622504,0.20637263,0.016517205,0.064877585,0.57566684,0.00041801558],"about_ca_topic_score_codex":0.012982283,"about_ca_topic_score_gemma":0.019705733,"teacher_disagreement_score":0.012982283,"about_ca_system_score_codex":0.0011263207,"about_ca_system_score_gemma":0.0019490524,"threshold_uncertainty_score":0.027946532},"labels":[],"label_agreement":null},{"id":"W4287020707","doi":"10.18653/v1/2023.acl-demo.24","title":"YANMTT: Yet Another Neural Machine Translation Toolkit","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Surrey Place Centre","funders":"Ministry of Internal Affairs and Communications","keywords":"Computer science; Machine translation; Transformer; Artificial intelligence; Translation (biology); Sequence (biology); Machine learning","score_opus":0.025448397469178038,"score_gpt":0.2822492070972983,"score_spread":0.25680080962812024,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4287020707","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0025397057,0.0011756165,0.65951306,0.0006595252,0.0012279768,0.00016483113,0.011001033,0.30738875,0.016329475],"genre_scores_gemma":[0.07952541,0.0014225462,0.7337178,0.001462406,0.00035638688,0.0008583147,0.07075597,0.076923355,0.034977786],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9984218,0.0004202726,0.00018587302,0.000396072,0.00045219492,0.00012374682],"domain_scores_gemma":[0.99830616,0.0005170864,0.00007472176,0.0005718758,0.0004302988,0.000099769015],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013581786,0.0015420552,0.0010178995,0.0010819322,0.0007311793,0.0019063652,0.002860859,0.0016457285,0.0453306],"category_scores_gemma":[0.0069868467,0.000971093,0.0016148846,0.0012940894,0.0004905239,0.003082008,0.00360898,0.0032240008,0.047011796],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00051305105,0.00010804887,0.000897629,0.0018322375,0.00034635933,0.0005670818,0.00042859802,0.023478065,0.01840873,0.037278946,0.48842946,0.42771173],"study_design_scores_gemma":[0.00017588402,0.00010995105,0.000669793,0.00027091987,0.000121101744,0.0009319725,0.0001181458,0.20259382,0.025461879,0.06274834,0.70660406,0.0001940299],"about_ca_topic_score_codex":0.0038945228,"about_ca_topic_score_gemma":0.005495713,"teacher_disagreement_score":0.0453306,"about_ca_system_score_codex":0.0008187001,"about_ca_system_score_gemma":0.0015937771,"threshold_uncertainty_score":0.15164602},"labels":[],"label_agreement":null},{"id":"W4287414585","doi":"","title":"Revitalisation des langues autochtones via le prétraitement et la traduction automatique neuronale: le cas de l'inuktitut","year":2021,"lang":"fr","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Political science; Humanities; Philosophy","score_opus":0.0240613160779954,"score_gpt":0.2777224666815086,"score_spread":0.2536611506035132,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4287414585","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4006703,0.0022536735,0.5413142,0.0034400963,0.0006098804,0.00012973962,0.00061780546,0.0130265085,0.03793782],"genre_scores_gemma":[0.85947186,0.00067679054,0.10560526,0.00036036095,0.00011061313,0.00008660326,0.000574394,0.0016273387,0.03148673],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99909425,0.00025718892,0.000053708,0.00024247455,0.00025800822,0.000094383984],"domain_scores_gemma":[0.99619627,0.0016013703,0.00011496629,0.00108268,0.00091129757,0.000093352195],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00117482,0.00050521764,0.00061884994,0.00064638566,0.00085644965,0.0035888369,0.00096033775,0.0010329073,0.0065779174],"category_scores_gemma":[0.0062515,0.0004320763,0.00076354324,0.00062718714,0.001608121,0.003546249,0.0014946586,0.0015844483,0.002710398],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00071419094,0.0002608538,0.004334008,0.0006739021,0.00013783119,0.0013275526,0.004867617,0.034046005,0.1622052,0.11947645,0.011392039,0.6605644],"study_design_scores_gemma":[0.00009124976,0.00037598755,0.0060991184,0.0001683998,0.00020923385,0.0011438667,0.0020986986,0.65440553,0.16494368,0.083028585,0.08727676,0.00015892295],"about_ca_topic_score_codex":0.013887032,"about_ca_topic_score_gemma":0.0123430565,"teacher_disagreement_score":0.013887032,"about_ca_system_score_codex":0.0012261244,"about_ca_system_score_gemma":0.0015042968,"threshold_uncertainty_score":0.027612388},"labels":[],"label_agreement":null},{"id":"W4287752972","doi":"10.48550/arxiv.2006.13343","title":"One Model to Pronounce Them All: Multilingual Grapheme-to-Phoneme\\n Conversion With a Transformer Ensemble","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Grapheme; Transformer; Computer science; Word error rate; Task (project management); Speech recognition; Language model; Artificial intelligence; Natural language processing; Speech synthesis; Engineering; Voltage","score_opus":0.09770618884152414,"score_gpt":0.22129217475345705,"score_spread":0.12358598591193291,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4287752972","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.051620267,0.0011825736,0.90879756,0.0014710518,0.0010256178,0.00011284738,0.0015913661,0.020233197,0.013965527],"genre_scores_gemma":[0.7166188,0.0008535509,0.24530324,0.0011197781,0.0003448097,0.00018048876,0.0056488602,0.0019593956,0.027971124],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995377,0.00010349194,0.000019030642,0.00021911963,0.00006630303,0.000054385077],"domain_scores_gemma":[0.9994892,0.00016494021,0.000020661275,0.00015326367,0.00011568762,0.000056332407],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00079691445,0.0020371675,0.0008669111,0.00070563256,0.00060198683,0.001285584,0.0017079952,0.0011768512,0.0058919066],"category_scores_gemma":[0.0016038272,0.0005087623,0.0012635117,0.0006141122,0.000536507,0.0028232276,0.0019498838,0.003540358,0.005807059],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00084118027,0.0004944766,0.0031409855,0.00025560454,0.0005447372,0.000639536,0.0004294596,0.17009778,0.041914657,0.01633817,0.041010354,0.7242931],"study_design_scores_gemma":[0.00004040146,0.00010907957,0.00039795297,0.000021107475,0.00013237643,0.00022770415,0.000091612405,0.9610005,0.01480733,0.014133458,0.008997791,0.000040730294],"about_ca_topic_score_codex":0.0058858683,"about_ca_topic_score_gemma":0.013119155,"teacher_disagreement_score":0.0058919066,"about_ca_system_score_codex":0.0004840474,"about_ca_system_score_gemma":0.0010239981,"threshold_uncertainty_score":0.019710422},"labels":[],"label_agreement":null},{"id":"W4287816198","doi":"10.48550/arxiv.2004.04468","title":"A Multilingual Study of Multi-Sentence Compression using Word\\n Vertex-Labeled Graphs and Integer Linear Programming","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Grammaticality; Computer science; Automatic summarization; Sentence; Natural language processing; Artificial intelligence; Integer programming; Word (group theory); Graph; Algorithm; Theoretical computer science; Grammar; Mathematics; Linguistics","score_opus":0.11894219885826816,"score_gpt":0.2663279437168967,"score_spread":0.14738574485862854,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4287816198","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16122006,0.003085351,0.82220113,0.0012942225,0.00017700902,0.00016618073,0.0006134435,0.0025564216,0.00868617],"genre_scores_gemma":[0.6258156,0.0009916046,0.36572203,0.00026611047,0.00032038311,0.00012123815,0.0017747813,0.0006185702,0.004369729],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99812156,0.0010286573,0.00008387568,0.00037145943,0.00029356204,0.0001009188],"domain_scores_gemma":[0.989876,0.008167567,0.00051329203,0.0005315558,0.00073746307,0.0001741257],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019146239,0.00069340057,0.00053024996,0.0017458203,0.00059046526,0.0014174025,0.0008842196,0.00062202686,0.002713736],"category_scores_gemma":[0.009859138,0.00032974003,0.0006444919,0.0021308446,0.0008996118,0.0025324624,0.000837597,0.0012999127,0.0005611885],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00094713666,0.0006775365,0.0059847753,0.0009816734,0.00027924817,0.00058847346,0.00080093346,0.3048782,0.013783786,0.042320717,0.009250511,0.619507],"study_design_scores_gemma":[0.000037145164,0.00023767973,0.001288207,0.000021117841,0.000053582262,0.00016569711,0.00018544371,0.9698093,0.0088166185,0.014551078,0.0047985553,0.000035592526],"about_ca_topic_score_codex":0.009166889,"about_ca_topic_score_gemma":0.011136906,"teacher_disagreement_score":0.009166889,"about_ca_system_score_codex":0.0015164908,"about_ca_system_score_gemma":0.00095889275,"threshold_uncertainty_score":0.0182271},"labels":[],"label_agreement":null},{"id":"W4287829644","doi":"10.5281/zenodo.3693274","title":"ON THE RELEVANCE OF QUERY EXPANSION USING PARALLEL CORPORA AND WORD EMBEDDINGS TO BOOST TEXT DOCUMENT RETRIEVAL PRECISION","year":2020,"lang":"en","type":"article","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Relevance (law); Query expansion; Word (group theory); Information retrieval; Relevance feedback; Document retrieval; Natural language processing; Artificial intelligence; Mathematics; Image retrieval","score_opus":0.03762505963033836,"score_gpt":0.2720517554135033,"score_spread":0.23442669578316494,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4287829644","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4999086,0.016557524,0.4537657,0.0013506074,0.0007596283,0.0005893719,0.00091841945,0.008919451,0.017230656],"genre_scores_gemma":[0.75747186,0.0030533143,0.22969902,0.00030278097,0.00050769903,0.00024613785,0.0015597346,0.00080225384,0.006357145],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99754506,0.0010059524,0.00015785657,0.00050207827,0.00064081314,0.00014829931],"domain_scores_gemma":[0.9926086,0.0050260318,0.00028004494,0.00076310773,0.0012341548,0.00008794123],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030422125,0.0011653573,0.0011492348,0.0023511807,0.00050556305,0.0013625744,0.0006773669,0.00105437,0.002593392],"category_scores_gemma":[0.01874803,0.00033814204,0.00051536,0.0021191048,0.00064356264,0.0030059945,0.0010785239,0.0008184466,0.0018320769],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015732698,0.0006149123,0.004179006,0.0009666607,0.00023961009,0.00021318987,0.00031984458,0.05765439,0.09550299,0.0041384315,0.011996752,0.82260096],"study_design_scores_gemma":[0.00045823853,0.0015012386,0.011155348,0.00015853677,0.00055271585,0.00091552397,0.00032864575,0.82387704,0.13093916,0.012676406,0.017283237,0.0001538377],"about_ca_topic_score_codex":0.0025569957,"about_ca_topic_score_gemma":0.0027408244,"teacher_disagreement_score":0.0030422125,"about_ca_system_score_codex":0.0005280172,"about_ca_system_score_gemma":0.0007258118,"threshold_uncertainty_score":0.016088903},"labels":[],"label_agreement":null},{"id":"W4287855019","doi":"10.18653/v1/2022.semeval-1.16","title":"UAlberta at SemEval 2022 Task 2: Leveraging Glosses and Translations for Multilingual Idiomaticity Detection","year":2022,"lang":"en","type":"article","venue":"Proceedings of the 16th International Workshop on Semantic Evaluation (SemEval-2022)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Machine Intelligence Institute","keywords":"Computer science; SemEval; Literal (mathematical logic); Natural language processing; Artificial intelligence; Task (project management); Classifier (UML); Literal translation; Leverage (statistics); Linguistics; Source text; Programming language","score_opus":0.03128321214141992,"score_gpt":0.3218947493615454,"score_spread":0.29061153722012545,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4287855019","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13043673,0.010865306,0.28756708,0.0055889455,0.0024884539,0.0028173402,0.13868876,0.21706326,0.20448416],"genre_scores_gemma":[0.28895414,0.0020202775,0.35604006,0.0016007733,0.0002893249,0.0014827305,0.2766535,0.013842732,0.059116453],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9940204,0.0018135431,0.0003619234,0.0014548026,0.0018790626,0.00047024435],"domain_scores_gemma":[0.9927538,0.0018109446,0.0003118672,0.0018604593,0.0027581987,0.0005047324],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0056901923,0.0026315001,0.0020208296,0.005648793,0.0030782074,0.006579258,0.0024763895,0.0021055408,0.037762452],"category_scores_gemma":[0.014557877,0.0010638011,0.0011078102,0.0030869336,0.0011007938,0.00820922,0.006125794,0.002547967,0.027697872],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015817154,0.0005863218,0.00527743,0.0020668274,0.0003093613,0.0007324034,0.0015824441,0.0022831082,0.02828838,0.017693931,0.4455341,0.49406394],"study_design_scores_gemma":[0.00083358004,0.00045475957,0.024628224,0.0011197146,0.00037955443,0.0016883548,0.0035197127,0.11757643,0.05719564,0.034741435,0.7574422,0.0004203132],"about_ca_topic_score_codex":0.089876786,"about_ca_topic_score_gemma":0.119011655,"teacher_disagreement_score":0.089876786,"about_ca_system_score_codex":0.0034110232,"about_ca_system_score_gemma":0.0041471003,"threshold_uncertainty_score":0.17870724},"labels":[],"label_agreement":null},{"id":"W4287887329","doi":"10.18653/v1/2022.naacl-tutorials.6","title":"Contrastive Data and Learning for Natural Language Processing","year":2022,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":39,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Atomic Energy of Canada Limited; Westlake University; University of Virginia; Pennsylvania State University; University of Pennsylvania","keywords":"Zhàng; Computer science; Computational linguistics; Natural language processing; Artificial intelligence; Language technology; Linguistics; Natural language; Human language; Programming language; Comprehension approach; China; History; Philosophy","score_opus":0.01609528598964778,"score_gpt":0.2996274505530756,"score_spread":0.28353216456342784,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4287887329","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0049501415,0.0039858473,0.9782391,0.0020417203,0.00036233503,0.000116154784,0.0010512592,0.0032932062,0.005960184],"genre_scores_gemma":[0.12766552,0.0023143848,0.8566571,0.0005076514,0.0005574934,0.00061950594,0.003886707,0.0008357609,0.0069559496],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99747485,0.0011090928,0.00019758534,0.0006276456,0.0005066483,0.000084123974],"domain_scores_gemma":[0.9928039,0.004176361,0.0001888061,0.0020114803,0.0006571212,0.00016239248],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0053668837,0.0011098689,0.0013106489,0.0028268343,0.001175177,0.00333949,0.0025881198,0.0013927952,0.008552709],"category_scores_gemma":[0.017402144,0.0010759573,0.0013840628,0.0026557469,0.002783383,0.008846021,0.0038895889,0.0043087453,0.004185959],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003672917,0.00022438457,0.0017241668,0.00052905845,0.00016725027,0.00017436182,0.0003898929,0.011974307,0.0043913783,0.39631936,0.040166583,0.5435719],"study_design_scores_gemma":[0.00004005724,0.000053342534,0.0006329628,0.00008298454,0.000032703865,0.000095340525,0.00013118026,0.19472751,0.0051461156,0.7565636,0.042459637,0.000034614353],"about_ca_topic_score_codex":0.0024368104,"about_ca_topic_score_gemma":0.00579113,"teacher_disagreement_score":0.008552709,"about_ca_system_score_codex":0.001381418,"about_ca_system_score_gemma":0.0011368874,"threshold_uncertainty_score":0.02861166},"labels":[],"label_agreement":null},{"id":"W4287887703","doi":"10.18653/v1/2022.sigmorphon-1.23","title":"Generalizing Morphological Inflection Systems to Unseen Lemmas","year":2022,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Inflection; Computer science; Artificial intelligence; Transformer; Vocabulary; Task (project management); Pattern recognition (psychology); Speech recognition; Engineering","score_opus":0.02657961243404124,"score_gpt":0.27672569170379857,"score_spread":0.25014607926975735,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4287887703","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6770348,0.0013721078,0.25296614,0.0011681995,0.0007049677,0.00039259376,0.0051216404,0.047470007,0.013769572],"genre_scores_gemma":[0.849969,0.00043771617,0.12414406,0.0006700823,0.0001279676,0.00022652805,0.015691077,0.0019307988,0.006802728],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99820435,0.00044797672,0.00013244756,0.00089372817,0.00017705567,0.00014448215],"domain_scores_gemma":[0.9935534,0.0026611872,0.00019053901,0.0029196362,0.0005280766,0.00014710984],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002474076,0.002292631,0.0010090893,0.00090070366,0.0006880239,0.0019243951,0.0022714736,0.00187176,0.0046863044],"category_scores_gemma":[0.008776764,0.0008002871,0.0017436543,0.0011884187,0.0013133623,0.005756002,0.0034101626,0.004097753,0.0051372787],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00090793154,0.0005883865,0.009657298,0.0007845819,0.00045144744,0.0010546542,0.0019330657,0.11831924,0.06577075,0.0033147754,0.019780481,0.7774373],"study_design_scores_gemma":[0.00019084639,0.00072772213,0.008988342,0.00011322457,0.0002223393,0.001028773,0.0012465409,0.86234045,0.08003475,0.024313478,0.020626184,0.00016740146],"about_ca_topic_score_codex":0.0047514415,"about_ca_topic_score_gemma":0.009004204,"teacher_disagreement_score":0.0047514415,"about_ca_system_score_codex":0.0009908383,"about_ca_system_score_gemma":0.0008139873,"threshold_uncertainty_score":0.015677214},"labels":[],"label_agreement":null},{"id":"W4287888494","doi":"10.18653/v1/2022.bea-1.22","title":"‘Meet me at the ribary’ – Acceptability of spelling variants in free-text answers to listening comprehension prompts","year":2022,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Innovation Cluster (Canada)","funders":"Universität Duisburg-Essen","keywords":"Spelling; Operationalization; Computer science; Natural language processing; Task (project management); Similarity (geometry); Comprehension; Artificial intelligence; Active listening; German; Linguistics; Psychology; Communication","score_opus":0.014470269095385317,"score_gpt":0.2680494150222575,"score_spread":0.2535791459268722,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4287888494","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9871523,0.00011133366,0.0070487787,0.00013549686,0.000033287626,0.00004935094,0.0003293637,0.00016040994,0.004979703],"genre_scores_gemma":[0.99631715,0.000029122508,0.0025882947,0.00004781612,0.000015692594,0.00003090555,0.00038165937,0.000065594235,0.0005237496],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9880586,0.007563309,0.00084272755,0.0014282332,0.0018164383,0.00029052148],"domain_scores_gemma":[0.9237558,0.060804006,0.005455861,0.0041136364,0.0051612435,0.0007093621],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008311694,0.0005865138,0.00039375722,0.0010490798,0.0005303985,0.001874065,0.0005522712,0.0011764175,0.0034817727],"category_scores_gemma":[0.06609825,0.0002338195,0.00027105666,0.00054679037,0.0014746148,0.0015693315,0.0012913069,0.00069962715,0.0009692758],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0052580065,0.00050818507,0.2843419,0.002131648,0.00043786463,0.0031601014,0.14570594,0.0065762154,0.27334666,0.009000804,0.0103905685,0.25914207],"study_design_scores_gemma":[0.0001889659,0.0011493064,0.8595174,0.00024048262,0.00013126561,0.003289659,0.028061818,0.02758938,0.051114094,0.010315451,0.01804616,0.00035602952],"about_ca_topic_score_codex":0.0016747693,"about_ca_topic_score_gemma":0.0021551903,"teacher_disagreement_score":0.008311694,"about_ca_system_score_codex":0.00041425688,"about_ca_system_score_gemma":0.00018266439,"threshold_uncertainty_score":0.043956935},"labels":[],"label_agreement":null},{"id":"W4287891465","doi":"10.18653/v1/2022.gebnlp-1.25","title":"Indigenous Language Revitalization and the Dilemma of Gender Bias","year":2022,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Debiasing; Indigenous; Computer science; Identification (biology); First language; Gender bias; Indigenous language; Language identification; Natural language processing; Linguistics; Artificial intelligence; Natural language; Cognitive science; Psychology; Social psychology","score_opus":0.019912415933603083,"score_gpt":0.26064717738147963,"score_spread":0.24073476144787656,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4287891465","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.859989,0.001995502,0.080897495,0.0100566475,0.0004685974,0.00009711458,0.00044598887,0.0003497714,0.04569982],"genre_scores_gemma":[0.97217077,0.00064480456,0.017308095,0.0006152672,0.000033290045,0.00003418294,0.00015128635,0.00013339787,0.008908926],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99742967,0.0009714742,0.00012310574,0.00029202068,0.00082602125,0.0003577015],"domain_scores_gemma":[0.9948673,0.0011640218,0.00054998585,0.000761457,0.0024197716,0.0002375037],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024519733,0.00035259675,0.00032864252,0.0008150222,0.0025988347,0.0027520615,0.00073544803,0.00047634487,0.0026408539],"category_scores_gemma":[0.01313626,0.00013864366,0.00020976904,0.0014636186,0.0035482107,0.0032286858,0.0026987572,0.0010412029,0.0003807603],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00045398864,0.000091420196,0.07253315,0.0006090754,0.00010485735,0.001407766,0.106811486,0.0069102286,0.022051169,0.25450978,0.012150948,0.52236617],"study_design_scores_gemma":[0.000040053004,0.00028468724,0.103257746,0.00092055084,0.00021365608,0.0025375276,0.2610278,0.047790445,0.06965451,0.18088104,0.33301333,0.00037872294],"about_ca_topic_score_codex":0.28789195,"about_ca_topic_score_gemma":0.38643253,"teacher_disagreement_score":0.28789195,"about_ca_system_score_codex":0.005182788,"about_ca_system_score_gemma":0.0077684154,"threshold_uncertainty_score":0.57243246},"labels":[],"label_agreement":null},{"id":"W4288060620","doi":"10.18357/kula.221","title":"Re-purposing Excavation Database Content as Paradata","year":2022,"lang":"en","type":"article","venue":"KULA knowledge creation dissemination and preservation studies","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":35,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"Riksbankens Jubileumsfond; European Commission","keywords":"Metadata; Computer science; Documentation; Field (mathematics); Identification (biology); Structuring; Process (computing); Data science; Excavation; Reading (process); Plan (archaeology); Archaeology; World Wide Web","score_opus":0.09820098497722425,"score_gpt":0.39186040705082387,"score_spread":0.29365942207359963,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4288060620","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17436586,0.0020221584,0.67192465,0.006397971,0.0011218357,0.005219096,0.057624597,0.007761756,0.07356205],"genre_scores_gemma":[0.31751198,0.0011258139,0.6179843,0.0009581641,0.00024645007,0.0036741744,0.038457066,0.004395445,0.015646597],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.97866166,0.0064287954,0.004231226,0.0032982281,0.006623104,0.0007570577],"domain_scores_gemma":[0.8662123,0.057902854,0.0067391517,0.044854086,0.023168176,0.0011233939],"candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.024104992,0.0010031115,0.0009090451,0.016646141,0.002599229,0.013119826,0.0025768958,0.0011671225,0.0042941878],"category_scores_gemma":[0.087720245,0.0011673126,0.0011071244,0.017893463,0.003469078,0.012836963,0.008262751,0.002266109,0.0035327494],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00051944505,0.0002244238,0.04022616,0.004683125,0.00017591385,0.0016549599,0.15048406,0.003137483,0.023503887,0.12911572,0.03591946,0.6103554],"study_design_scores_gemma":[0.00004292636,0.00010986484,0.02812449,0.0027470458,0.0001494333,0.0011853371,0.06018271,0.008592186,0.02106653,0.06723518,0.8103509,0.00021343911],"about_ca_topic_score_codex":0.0064784987,"about_ca_topic_score_gemma":0.010299732,"teacher_disagreement_score":0.9868802,"about_ca_system_score_codex":0.0032087683,"about_ca_system_score_gemma":0.0060901716,"threshold_uncertainty_score":0.12748092},"labels":[],"label_agreement":null},{"id":"W4288429387","doi":"10.4995/rlyla.2022.16132","title":"Advances in the automatic lemmatization of Old English: class V strong verbs (L-Y)","year":2022,"lang":"en","type":"article","venue":"Revista de Lingüística y Lenguas Aplicadas","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Agencia Estatal de Investigación; Department of Environment and Conservation, Government of Newfoundland and Labrador; Strong","keywords":"Lemmatisation; Lemma (botany); Computer science; Natural language processing; Artificial intelligence; Class (philosophy); Variation (astronomy); Linguistics; Security token","score_opus":0.008919422198126501,"score_gpt":0.2692739860651642,"score_spread":0.26035456386703765,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4288429387","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03211942,0.0027038392,0.9385038,0.00056492724,0.00024550824,0.00022947174,0.0011177126,0.0056905863,0.01882479],"genre_scores_gemma":[0.19190115,0.002996789,0.7889189,0.00019935664,0.0002008777,0.00018157621,0.004265786,0.0028946064,0.008440938],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99833006,0.0005858701,0.00025656194,0.0004680459,0.0002960883,0.0000633447],"domain_scores_gemma":[0.9937844,0.0031807935,0.0005731593,0.0012019762,0.0011788657,0.00008078854],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026712385,0.0007555276,0.0005184796,0.0029748487,0.00075838086,0.0035818422,0.0010380612,0.00061653217,0.0064972076],"category_scores_gemma":[0.0055538663,0.00075386564,0.0007524603,0.0014554973,0.0015923193,0.004569344,0.0017986165,0.001571297,0.0053375294],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011667834,0.00010051415,0.004948101,0.0018622016,0.000062551044,0.00061040296,0.0048803627,0.0019399674,0.06658036,0.10407799,0.010948859,0.8038719],"study_design_scores_gemma":[0.000072616655,0.00021808333,0.01536704,0.0010358483,0.0001540735,0.0037325136,0.002874248,0.051593784,0.12535957,0.10881308,0.69055146,0.000227753],"about_ca_topic_score_codex":0.0011107313,"about_ca_topic_score_gemma":0.0016205789,"teacher_disagreement_score":0.0064972076,"about_ca_system_score_codex":0.0006111163,"about_ca_system_score_gemma":0.0013657719,"threshold_uncertainty_score":0.02173531},"labels":[],"label_agreement":null},{"id":"W4289305465","doi":"10.5281/zenodo.1420271","title":"What Makes A Code Change Easier To Review: An Empirical Investigation On Code Change Reviewability","year":2018,"lang":"en","type":"paratext","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Nederlandse Organisatie voor Wetenschappelijk Onderzoek; Schweizerischer Nationalfonds zur Förderung der Wissenschaftlichen Forschung; National Science Foundation","keywords":"Computer science; Code (set theory); Programming language; Code review; Empirical research; Static program analysis; Software development; Software; Mathematics; Statistics","score_opus":0.1778519253806048,"score_gpt":0.3667286719834998,"score_spread":0.188876746602895,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4289305465","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98567146,0.0012882215,0.0039066416,0.0018860673,0.00006278012,0.00048873766,0.0001638718,0.00011585571,0.0064165406],"genre_scores_gemma":[0.9952342,0.00046775778,0.0026282612,0.00028129198,0.000076324846,0.00027043925,0.00017505222,0.000089820365,0.0007768444],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.8874164,0.054600205,0.014383048,0.007589453,0.03341833,0.0025925275],"domain_scores_gemma":[0.13901155,0.6904666,0.1069484,0.015944358,0.041157234,0.0064719412],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.088810965,0.00034878525,0.0006744707,0.0077431435,0.0025839747,0.006470034,0.0016589082,0.0017270246,0.0024501],"category_scores_gemma":[0.5689576,0.00058397406,0.0006533966,0.0051665325,0.0037674233,0.009002403,0.0036264155,0.0024214354,0.00047419162],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00094677706,0.001492402,0.6254865,0.0038048576,0.0004813303,0.0017205405,0.1910443,0.0007501382,0.005216718,0.0032525808,0.0053376947,0.16046615],"study_design_scores_gemma":[0.00019558625,0.0012607866,0.86492354,0.0013020008,0.00032957026,0.0016082157,0.093297556,0.004757195,0.002692858,0.003693968,0.025700798,0.0002379657],"about_ca_topic_score_codex":0.0032564837,"about_ca_topic_score_gemma":0.0048746127,"teacher_disagreement_score":0.911189,"about_ca_system_score_codex":0.0034167953,"about_ca_system_score_gemma":0.004389889,"threshold_uncertainty_score":0.46968287},"labels":[],"label_agreement":null},{"id":"W4289552840","doi":"10.48550/arxiv.1809.01074","title":"A Novel Neural Sequence Model with Multiple Attentions for Word Sense\\n Disambiguation","year":2018,"lang":"","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Computer science; Bridging (networking); Word (group theory); Sentence; Natural language processing; Artificial intelligence; Sequence (biology); Encoder; Word-sense disambiguation; SemEval; Linguistics","score_opus":0.1342456584685779,"score_gpt":0.23751574908252324,"score_spread":0.10327009061394535,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4289552840","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03264166,0.0013707045,0.9572789,0.001468407,0.00046859845,0.0000754352,0.00037981282,0.0017792253,0.0045373985],"genre_scores_gemma":[0.7485922,0.0012665171,0.22473615,0.0011518194,0.00038784405,0.00027595228,0.0009341512,0.0003262136,0.022329109],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99963295,0.0000873683,0.000021534579,0.00014994535,0.000059720176,0.000048477486],"domain_scores_gemma":[0.99944764,0.00029118315,0.000048495087,0.00005117107,0.0001119722,0.00004949153],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000865508,0.0009271133,0.0010809257,0.0008659612,0.00055590237,0.0011201078,0.0025783696,0.0018671813,0.0042856922],"category_scores_gemma":[0.00229054,0.0007245623,0.0010297108,0.0011502288,0.00072922284,0.0026751885,0.0014260609,0.0021963092,0.0012411217],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00031113715,0.00020530031,0.0014233136,0.00016158247,0.00018728706,0.00036590672,0.00021764668,0.6989897,0.007364676,0.036598947,0.0068146912,0.2473598],"study_design_scores_gemma":[0.000008072861,0.000016135327,0.000054365035,0.000005261023,0.000012165676,0.000022850618,0.0000048917896,0.99110454,0.00037575755,0.007886099,0.0005041107,0.0000056744775],"about_ca_topic_score_codex":0.012627737,"about_ca_topic_score_gemma":0.018718123,"teacher_disagreement_score":0.012627737,"about_ca_system_score_codex":0.0012401202,"about_ca_system_score_gemma":0.0018479467,"threshold_uncertainty_score":0.025108516},"labels":[],"label_agreement":null},{"id":"W4289743854","doi":"10.48550/arxiv.1807.10805","title":"Improving Neural Sequence Labelling Using Additional Linguistic Information","year":2018,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Chunking (psychology); Computer science; Natural language processing; Sequence labeling; Artificial intelligence; Sequence (biology); Named-entity recognition; Word (group theory); Labelling; Sentence; Benchmark (surveying); Categorical variable; Task (project management); Machine learning; Linguistics","score_opus":0.0713100375631831,"score_gpt":0.2172160888038689,"score_spread":0.14590605124068579,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4289743854","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.065789394,0.0010205703,0.91978484,0.0009982954,0.00027224523,0.000104423336,0.0006069838,0.0055046603,0.005918595],"genre_scores_gemma":[0.61720824,0.000799751,0.35752788,0.0011632465,0.0002065846,0.00029330977,0.0050694924,0.00057671714,0.017154858],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99936193,0.0001982376,0.000039655457,0.00023035296,0.00011327352,0.00005652703],"domain_scores_gemma":[0.9969547,0.001776817,0.0002030051,0.0004954426,0.00048300618,0.00008704395],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001435092,0.00097363786,0.0009859138,0.0009854983,0.0006068109,0.0011226364,0.0018634178,0.0022685565,0.0036962721],"category_scores_gemma":[0.0067822793,0.0004909535,0.0008550313,0.0010443049,0.0006956874,0.004473015,0.0013151497,0.0024431823,0.0026014342],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003725146,0.00036719302,0.0019463617,0.0002716688,0.00008691524,0.00017088735,0.00023771654,0.49844927,0.013104508,0.015217794,0.013521476,0.45625377],"study_design_scores_gemma":[0.000008693547,0.000025968595,0.000120135184,0.0000133464055,0.000010004674,0.000020742902,0.000012546722,0.98903966,0.002172096,0.0075117187,0.0010574437,0.000007689505],"about_ca_topic_score_codex":0.007946831,"about_ca_topic_score_gemma":0.01164921,"teacher_disagreement_score":0.007946831,"about_ca_system_score_codex":0.0014528005,"about_ca_system_score_gemma":0.0016780149,"threshold_uncertainty_score":0.015801132},"labels":[],"label_agreement":null},{"id":"W4289874410","doi":"10.1684/pnv.2022.1043","title":"T-DAV : Test de dénomination d’actions par visionnement de vidéos. Développement, validation et normalisation","year":2022,"lang":"fr","type":"article","venue":"Gériatrie et Psychologie Neuropsychiatrie du Vieillissement","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval; Centre for Interdisciplinary Research in Rehabilitation","funders":"","keywords":"Nomination; Psychology; Political science","score_opus":0.04194733096082507,"score_gpt":0.34351712656876526,"score_spread":0.30156979560794017,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4289874410","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9186036,0.0020530599,0.030382603,0.0005416257,0.00038959357,0.0038712898,0.015958803,0.001469785,0.02672957],"genre_scores_gemma":[0.94433415,0.0011125992,0.025286285,0.00026863345,0.00008210448,0.004448241,0.010438952,0.00028466745,0.013744402],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9980173,0.00051316305,0.0003440226,0.0003187949,0.00066172227,0.0001449237],"domain_scores_gemma":[0.99307394,0.0026673167,0.0016152418,0.00049121,0.0017288876,0.00042340384],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022201163,0.0010551286,0.000552746,0.0021418934,0.0003232725,0.000815251,0.000803745,0.0007402576,0.006264447],"category_scores_gemma":[0.017179929,0.0002398988,0.0009585329,0.00063817,0.00055468583,0.001756937,0.0010211061,0.0007071603,0.0019447567],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0039127036,0.0021594858,0.44756484,0.0013437668,0.000860725,0.0016520192,0.003112107,0.0047828592,0.040812615,0.003077452,0.03747517,0.4532462],"study_design_scores_gemma":[0.000381418,0.0029378163,0.9426222,0.0001909446,0.00011778697,0.004674105,0.0013255522,0.0064483243,0.009305808,0.0022443687,0.029587224,0.00016453721],"about_ca_topic_score_codex":0.005670236,"about_ca_topic_score_gemma":0.004372513,"teacher_disagreement_score":0.006264447,"about_ca_system_score_codex":0.0005432713,"about_ca_system_score_gemma":0.0007191786,"threshold_uncertainty_score":0.020956695},"labels":[],"label_agreement":null},{"id":"W4289914272","doi":"10.3765/amp.v9i0.5168","title":"Comparative Reconstruction Probabilistically: The Role of Inventory and Phonotactics","year":2022,"lang":"en","type":"article","venue":"Proceedings of the Annual Meetings on Phonology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Phonotactics; Spurious relationship; Phonology; Computer science; Scope (computer science); Merge (version control); Natural language processing; Artificial intelligence; Linguistics; Econometrics; Machine learning; Mathematics; Information retrieval; Programming language","score_opus":0.009442116124865847,"score_gpt":0.23705018933607652,"score_spread":0.22760807321121068,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4289914272","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15701787,0.0005457405,0.8168138,0.00032214235,0.00005780642,0.000098905264,0.0004782945,0.0007052285,0.023960218],"genre_scores_gemma":[0.86439884,0.0001805051,0.1322595,0.00008753973,0.00004112063,0.00014912612,0.0006613419,0.0006007502,0.0016213134],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9899991,0.004961818,0.00057841785,0.0018989027,0.00220022,0.00036160214],"domain_scores_gemma":[0.9503595,0.03504698,0.0042916154,0.0066626156,0.0031696581,0.0004695371],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.015749538,0.0006145542,0.00066808256,0.0042117545,0.001745487,0.0053744134,0.0021833845,0.0012135429,0.0074355346],"category_scores_gemma":[0.0747825,0.00091468944,0.0009383883,0.0037743803,0.0056943484,0.008709034,0.0037810346,0.0017128274,0.00071166054],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006542445,0.000090364774,0.082139224,0.0007929873,0.0003478426,0.0006927632,0.008793933,0.057744186,0.015729709,0.6046127,0.0020960176,0.22630596],"study_design_scores_gemma":[0.000044315988,0.00037549424,0.087854266,0.0003752064,0.00028397865,0.0028643238,0.005419698,0.26009175,0.02035254,0.5929525,0.029079923,0.00030604927],"about_ca_topic_score_codex":0.0018381464,"about_ca_topic_score_gemma":0.0024543304,"teacher_disagreement_score":0.015749538,"about_ca_system_score_codex":0.001447444,"about_ca_system_score_gemma":0.0012931131,"threshold_uncertainty_score":0.083292544},"labels":[],"label_agreement":null},{"id":"W4292510811","doi":"10.6087/kcse.285","title":"Improving Journal Article Tag Suite for multilingual articles","year":2022,"lang":"en","type":"article","venue":"Science Editing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"U.S. National Library of Medicine; Canadian Medical Association","keywords":"Suite; Computer science; Variety (cybernetics); Set (abstract data type); World Wide Web; Information retrieval; Library science; Data science; Political science; Artificial intelligence; Programming language; Law","score_opus":0.015362391554324064,"score_gpt":0.2954643337713264,"score_spread":0.28010194221700235,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4292510811","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":"reporting","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":"reporting","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015236262,0.0007245102,0.8062707,0.005849253,0.0041557276,0.0028717637,0.006581593,0.14695325,0.011357033],"genre_scores_gemma":[0.030616535,0.00077556097,0.8972658,0.0019007624,0.001148345,0.0009400282,0.026310263,0.024167843,0.016874753],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.96466684,0.010519863,0.009254289,0.0029231668,0.0113452,0.0012906863],"domain_scores_gemma":[0.69888264,0.0616348,0.018137762,0.068464115,0.14303279,0.009847932],"candidate_categories":["metaresearch","scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.044559997,0.0023035037,0.0021132042,0.020748999,0.0038885155,0.014322166,0.003780182,0.0026153005,0.019373981],"category_scores_gemma":[0.16243066,0.0027624352,0.0024661594,0.012458086,0.00204044,0.022018516,0.0072965827,0.006073353,0.04148195],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00090326346,0.00074926065,0.009069917,0.0017713954,0.00021521821,0.0011812169,0.0034370425,0.0035799437,0.060935866,0.020061685,0.25602895,0.6420663],"study_design_scores_gemma":[0.0002105895,0.0005222978,0.005631893,0.00062741275,0.00025488768,0.0015400263,0.0021335625,0.03241908,0.08382942,0.013985834,0.85822666,0.0006183466],"about_ca_topic_score_codex":0.0061075804,"about_ca_topic_score_gemma":0.00835843,"teacher_disagreement_score":0.98567784,"about_ca_system_score_codex":0.0033227657,"about_ca_system_score_gemma":0.009386496,"threshold_uncertainty_score":0.23565865},"labels":[],"label_agreement":null},{"id":"W4293107821","doi":"10.4018/978-1-6684-5682-8.ch046","title":"A State-of-the-Art Review of Nigerian Languages Natural Language Processing Research","year":2022,"lang":"en","type":"review","venue":"IGI Global eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Yoruba; Hausa; Computer science; Igbo; Languages of Africa; Resource (disambiguation); State (computer science); Representation (politics); Natural language processing; Artificial intelligence; Linguistics; Programming language","score_opus":0.043877687317346314,"score_gpt":0.39902648803491236,"score_spread":0.35514880071756605,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4293107821","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00016876362,0.9979577,0.00014834538,0.0002575054,0.0001317324,0.0000068356826,0.000034571047,0.000007759759,0.0012867856],"genre_scores_gemma":[0.0007421329,0.99845576,0.00028235358,0.000111960224,0.00006800971,0.0000073779747,0.00003760046,0.000001716706,0.0002930921],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9995511,0.00007931871,0.00010297223,0.00008358349,0.00015512355,0.00002781506],"domain_scores_gemma":[0.99843115,0.0010368117,0.0001633218,0.000025343044,0.0002946161,0.000048757833],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009165106,0.0007772879,0.0011405015,0.0053851884,0.0004469653,0.0013857393,0.0008000916,0.00087509915,0.004733868],"category_scores_gemma":[0.0020797441,0.00043625807,0.00066832197,0.005537864,0.00051326107,0.0019647297,0.0005900267,0.0009723405,0.0016344696],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000046632056,0.00005196053,0.00033176673,0.07355858,0.00008393306,0.00018419535,0.0002523623,0.00027529962,0.0012662937,0.0032654265,0.02315118,0.8975323],"study_design_scores_gemma":[0.0000056862964,0.00007155422,0.0020413802,0.029160459,0.00023997338,0.0008119716,0.00024169295,0.000102408456,0.0005310984,0.0014987262,0.9652726,0.000022543638],"about_ca_topic_score_codex":0.0024153476,"about_ca_topic_score_gemma":0.0047598644,"teacher_disagreement_score":0.0053851884,"about_ca_system_score_codex":0.00070247427,"about_ca_system_score_gemma":0.0028283778,"threshold_uncertainty_score":0.015836358},"labels":[],"label_agreement":null},{"id":"W4293232299","doi":"10.1075/lia.21003.bel","title":"L’acquisition des objets directs et indirects en français L1","year":2022,"lang":"fr","type":"article","venue":"Language Interaction and Acquisition","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Humanities; Philosophy","score_opus":0.013993844401754836,"score_gpt":0.2960243486677767,"score_spread":0.2820305042660219,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4293232299","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9943798,0.00042263413,0.00046667946,0.00008403123,0.000005233637,0.000013135986,0.00063511427,0.000013749351,0.0039796513],"genre_scores_gemma":[0.99024886,0.0004543408,0.0017181869,0.00004501321,0.000005779444,0.0000346231,0.0008957271,0.000024095145,0.0065733655],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.999119,0.00018683485,0.000087136046,0.00023983688,0.00022033555,0.00014681809],"domain_scores_gemma":[0.9922035,0.004272642,0.0011468632,0.00029112125,0.0018291256,0.00025681854],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016434236,0.00067463337,0.0002854567,0.0011674311,0.0008024731,0.0013136951,0.00033106614,0.0004499069,0.007888692],"category_scores_gemma":[0.0051712673,0.0003233437,0.0002241151,0.00091101817,0.00093586714,0.0005380136,0.0008327362,0.00039883843,0.00072685006],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00096672116,0.00009893358,0.5874191,0.00084104296,0.00014764247,0.0054976605,0.115853086,0.0006005913,0.10167396,0.0012527673,0.0018684799,0.18378001],"study_design_scores_gemma":[0.000015270192,0.00019404653,0.9512949,0.00011655084,0.00007489305,0.0018483164,0.019291421,0.00028668126,0.009481285,0.00010415377,0.017241484,0.000050892726],"about_ca_topic_score_codex":0.25582328,"about_ca_topic_score_gemma":0.30765745,"teacher_disagreement_score":0.25582328,"about_ca_system_score_codex":0.0019503401,"about_ca_system_score_gemma":0.0013842158,"threshold_uncertainty_score":0.5086684},"labels":[],"label_agreement":null},{"id":"W4293241618","doi":"10.5539/cis.v15n2p2","title":"A Brief Presentation of the Knowledge Paths for Semiotics (KPS) Project: Creating Digital Research Tools","year":2022,"lang":"en","type":"article","venue":"Computer and Information Science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Terminology; Presentation (obstetrics); Semiotics; Context (archaeology); Knowledge acquisition; Digital library; Information retrieval; Object (grammar); Data science; Artificial intelligence; Linguistics","score_opus":0.05336518011743069,"score_gpt":0.36544044847904456,"score_spread":0.31207526836161387,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4293241618","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005546778,0.028046878,0.64427334,0.023927225,0.005408922,0.0014902657,0.0017676722,0.0017734438,0.28776538],"genre_scores_gemma":[0.076547265,0.06257495,0.5910318,0.0091569545,0.0033046342,0.0030493317,0.0036441516,0.0012797093,0.24941118],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99837273,0.0008056278,0.00016232317,0.00015448061,0.00038414611,0.00012072246],"domain_scores_gemma":[0.9989524,0.0004956154,0.00009899332,0.000099654666,0.00021043303,0.00014295116],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026265364,0.0009225695,0.0005196151,0.0025719246,0.002017939,0.006130115,0.0011153575,0.0029992117,0.030221384],"category_scores_gemma":[0.0029591687,0.00049492886,0.0006544125,0.0035524282,0.0023577318,0.0069528073,0.00346332,0.0034505774,0.012796095],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00004548746,0.0001353682,0.00029555632,0.001337806,0.000009615503,0.0006216862,0.0047245375,0.00096602377,0.0055453265,0.80002725,0.055953473,0.13033791],"study_design_scores_gemma":[0.0000034155078,0.00005064924,0.0001478109,0.0003484584,0.0000025224492,0.00058287,0.00074517046,0.00051069807,0.000701046,0.055388644,0.9414989,0.000019882993],"about_ca_topic_score_codex":0.0009006848,"about_ca_topic_score_gemma":0.0015023241,"teacher_disagreement_score":0.030221384,"about_ca_system_score_codex":0.0018660341,"about_ca_system_score_gemma":0.002348338,"threshold_uncertainty_score":0.101100564},"labels":[],"label_agreement":null},{"id":"W4293444055","doi":"","title":"La particule séparable re- facteur de cohésion textuelle en français médiéval","year":2007,"lang":"fr","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Linguistic Association","funders":"","keywords":"Philosophy","score_opus":0.01616728681497898,"score_gpt":0.2590042783100772,"score_spread":0.24283699149509824,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4293444055","genre_codex":"empirical","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8886324,0.0034577958,0.0064235185,0.004438767,0.00027467872,0.000022411366,0.0011571373,0.00019876561,0.09539445],"genre_scores_gemma":[0.9769564,0.0005048526,0.0013427072,0.00010611275,0.000096836484,0.00001172007,0.00031105694,0.000069601716,0.020600814],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9985763,0.0004315192,0.00007617791,0.0003519684,0.00032675982,0.00023730172],"domain_scores_gemma":[0.997718,0.0008764202,0.0003105628,0.00022314668,0.00067639176,0.00019550374],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012928427,0.0005063868,0.00034802497,0.002008447,0.0044028354,0.0035323347,0.00040172282,0.000626187,0.009730828],"category_scores_gemma":[0.0035048039,0.00026741246,0.00029675954,0.0025455758,0.0024236515,0.0018233305,0.0011601195,0.0017441557,0.0007526489],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00076864014,0.00011929491,0.081734985,0.0004709068,0.0003743958,0.0020657524,0.19028802,0.0015536014,0.023688994,0.45236173,0.018204518,0.22836918],"study_design_scores_gemma":[0.000073738345,0.00010659504,0.4436331,0.00020737744,0.00013897603,0.0018464348,0.041767202,0.0014571351,0.0072170887,0.014266218,0.48919037,0.00009581051],"about_ca_topic_score_codex":0.24530223,"about_ca_topic_score_gemma":0.26892617,"teacher_disagreement_score":0.24530223,"about_ca_system_score_codex":0.0053704665,"about_ca_system_score_gemma":0.0032259538,"threshold_uncertainty_score":0.4877488},"labels":[],"label_agreement":null},{"id":"W4293454439","doi":"10.48550/arxiv.1201.4733","title":"Du TAL au TIL","year":2012,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; Computer Research Institute of Montréal","funders":"","keywords":"Bridge (graph theory); Psycholinguistics; Computer science; Human language; Natural language processing; Artificial intelligence; State (computer science); Natural (archaeology); Natural language; Linguistics; Psychology; Programming language; Cognition; Philosophy; History","score_opus":0.047526828836771944,"score_gpt":0.18983477930368312,"score_spread":0.14230795046691117,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4293454439","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0058375676,0.0050166096,0.0044985306,0.008067171,0.0053727417,0.000104296596,0.0018142804,0.0010392452,0.96824956],"genre_scores_gemma":[0.034391444,0.0037936233,0.0044324617,0.0017757909,0.00042482797,0.000087477456,0.0019466308,0.00076643087,0.9523813],"study_design_codex":"not_applicable","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99910045,0.00014169829,0.00004535515,0.0002895051,0.00026412331,0.00015885415],"domain_scores_gemma":[0.99939394,0.00010064102,0.000039187693,0.00011340242,0.0002401178,0.000112787755],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0007206674,0.0010404955,0.00075007306,0.001352852,0.0026683423,0.0051395744,0.00075628795,0.0020121147,0.53958184],"category_scores_gemma":[0.002530897,0.0003647797,0.0006739072,0.0011719386,0.00097442954,0.0025595143,0.0034296776,0.0026212116,0.27443522],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037487425,0.00012203694,0.0028229677,0.0006605194,0.00005490024,0.0011006708,0.0014558454,0.00063442095,0.004286601,0.105793625,0.49345404,0.3892395],"study_design_scores_gemma":[0.000005793234,0.00001924781,0.00057649973,0.00009466851,0.0000058302467,0.00023969926,0.00034225176,0.00011498134,0.00038322413,0.0034501706,0.9947619,0.0000057626958],"about_ca_topic_score_codex":0.007600505,"about_ca_topic_score_gemma":0.009262419,"teacher_disagreement_score":0.53958184,"about_ca_system_score_codex":0.0020372102,"about_ca_system_score_gemma":0.0018662588,"threshold_uncertainty_score":0.65673065},"labels":[],"label_agreement":null},{"id":"W4293576324","doi":"10.1038/s41467-022-32012-w","title":"Synthesizing theories of human language with Bayesian program induction","year":2022,"lang":"en","type":"article","venue":"Nature Communications","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; Université du Québec; Mila - Quebec Artificial Intelligence Institute","funders":"Air Force Office of Scientific Research; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research; National Science Foundation","keywords":"Computer science; Bayesian probability; Computational biology; Data science; Artificial intelligence; Biology","score_opus":0.012914675866656981,"score_gpt":0.3204863804193879,"score_spread":0.3075717045527309,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4293576324","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013557959,0.00015070886,0.981898,0.0005864553,0.00001607825,0.00007445412,0.00025456882,0.002280485,0.0011812426],"genre_scores_gemma":[0.22093987,0.00029121037,0.77503735,0.00034970447,0.00006001304,0.00031389846,0.0015009873,0.0005665636,0.00094038283],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99686533,0.0013994947,0.00016956018,0.0006706488,0.00078199676,0.00011299129],"domain_scores_gemma":[0.9820225,0.0146213425,0.0007369573,0.0015590275,0.0008707976,0.00018931396],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0057615303,0.0010166793,0.0011524349,0.0026419808,0.0010176091,0.0027970062,0.0032595003,0.0011777288,0.0027735983],"category_scores_gemma":[0.029759957,0.000996142,0.0022760981,0.0013948007,0.0023462765,0.005322844,0.0030002282,0.0030728942,0.00085866364],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00026967796,0.00035615527,0.0078172935,0.0007288482,0.00043317166,0.00026656868,0.0015808869,0.35704127,0.0077533955,0.19847508,0.0070410757,0.41823655],"study_design_scores_gemma":[0.000025387826,0.000022922868,0.0002326879,0.000043784392,0.000034234938,0.00003401108,0.00007974327,0.76976484,0.002621924,0.2248245,0.0022978596,0.000018121294],"about_ca_topic_score_codex":0.0035439362,"about_ca_topic_score_gemma":0.0071093054,"teacher_disagreement_score":0.0057615303,"about_ca_system_score_codex":0.0019945551,"about_ca_system_score_gemma":0.0024280155,"threshold_uncertainty_score":0.030470252},"labels":[],"label_agreement":null},{"id":"W4293576334","doi":"10.7557/12.6441","title":"Low hanging fruit and the Boasian trilogy in digital lexicography of morphologically rich languages","year":2022,"lang":"en","type":"article","venue":"Nordlyd","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Lexical database; Lexicon; Bilingual dictionary; Linguistics; Resource (disambiguation); Grammar; Artificial intelligence; Natural language processing; Word (group theory); Lexicography","score_opus":0.006440518135382094,"score_gpt":0.24184870258585825,"score_spread":0.23540818445047615,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4293576334","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.61397606,0.029389191,0.046377316,0.012099299,0.00014451936,0.000120007215,0.0015180957,0.00023676582,0.2961387],"genre_scores_gemma":[0.9422642,0.008878493,0.028269222,0.00053349097,0.00003570637,0.000036897127,0.0006040355,0.00007259509,0.019305408],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9993005,0.00027571808,0.00006435854,0.00009353991,0.00016524032,0.000100701116],"domain_scores_gemma":[0.99845326,0.000826701,0.00021947532,0.00013720026,0.00023606408,0.00012735747],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011072635,0.00018089333,0.00026532463,0.008032052,0.0028244576,0.004775007,0.00043783872,0.00029719897,0.005475228],"category_scores_gemma":[0.0030089328,0.00016320351,0.000110472465,0.010982659,0.0070603983,0.006370427,0.0024620725,0.0005692583,0.0005401408],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015206194,0.000038810067,0.0390999,0.0010672127,0.000021859767,0.0009748052,0.12960292,0.0006728741,0.0036550767,0.4915354,0.008478855,0.32470027],"study_design_scores_gemma":[0.000011713271,0.00003941187,0.07925751,0.0010818098,0.000033546527,0.003269591,0.2676911,0.0014742669,0.0030961859,0.1412134,0.50276196,0.00006950875],"about_ca_topic_score_codex":0.14877754,"about_ca_topic_score_gemma":0.4388908,"teacher_disagreement_score":0.14877754,"about_ca_system_score_codex":0.0052995114,"about_ca_system_score_gemma":0.006592326,"threshold_uncertainty_score":0.2958231},"labels":[],"label_agreement":null},{"id":"W4293637667","doi":"10.48550/arxiv.1601.07124","title":"LIA-RAG: a system based on graphs and divergence of probabilities\\n applied to Speech-To-Text Summarization","year":2016,"lang":"","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Automatic summarization; Computer science; Task (project management); Divergence (linguistics); Natural language processing; Speech recognition; Artificial intelligence; Noise (video); Linguistics; Engineering","score_opus":0.030699011528510883,"score_gpt":0.18346938410013525,"score_spread":0.15277037257162437,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4293637667","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007748365,0.00042509477,0.8233197,0.00022114618,0.00017021695,0.00019497909,0.001761364,0.16447821,0.0016809938],"genre_scores_gemma":[0.10856328,0.0002558061,0.87424153,0.00027377228,0.00021179828,0.00037411563,0.006626427,0.0051201005,0.0043331133],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99832803,0.000582445,0.0001235779,0.00062252785,0.00027396594,0.00006951873],"domain_scores_gemma":[0.99725753,0.0013514152,0.00023223308,0.00054224255,0.00048517575,0.000131514],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001751511,0.0015469813,0.0011046518,0.0037112352,0.00080423645,0.0021288362,0.0019211035,0.0015024637,0.0066573704],"category_scores_gemma":[0.007020559,0.0005601339,0.0009461214,0.0013663456,0.0007201346,0.003681432,0.0019736746,0.0015920427,0.0065574343],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00095699535,0.00018050935,0.001507626,0.0011038583,0.00030730193,0.00043050127,0.00076669443,0.020175783,0.05681767,0.014468924,0.049354386,0.85392976],"study_design_scores_gemma":[0.00025508317,0.0004865162,0.0023351975,0.0001136446,0.0002402023,0.00045666195,0.00033033916,0.77621096,0.073643886,0.038671948,0.10705464,0.0002008704],"about_ca_topic_score_codex":0.0032914225,"about_ca_topic_score_gemma":0.003782957,"teacher_disagreement_score":0.0066573704,"about_ca_system_score_codex":0.0008124285,"about_ca_system_score_gemma":0.0008589788,"threshold_uncertainty_score":0.022271097},"labels":[],"label_agreement":null},{"id":"W4293863352","doi":"10.1109/siu55565.2022.9864730","title":"Using Word Embeddings in Detection of Temporal Expressions in Turkish Texts","year":2022,"lang":"en","type":"article","venue":"2022 30th Signal Processing and Communications Applications Conference (SIU)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Stantec (Canada)","funders":"","keywords":"Turkish; Scope (computer science); Computer science; Word (group theory); Artificial intelligence; Natural language processing; Set (abstract data type); The Internet; Field (mathematics); Speech recognition; Linguistics; Mathematics; World Wide Web","score_opus":0.03696915632119951,"score_gpt":0.3151557293805765,"score_spread":0.278186573059377,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4293863352","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6298346,0.0019786179,0.3537093,0.0006392022,0.00033448005,0.00027406536,0.0041603046,0.0042973496,0.0047720876],"genre_scores_gemma":[0.80563796,0.0007359222,0.18331337,0.00011101432,0.00006319062,0.00018389037,0.0071057705,0.00020432702,0.0026445629],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.998804,0.00038006844,0.00017610249,0.0004164698,0.00014626975,0.00007714515],"domain_scores_gemma":[0.9976229,0.0012600403,0.00037506808,0.00020020816,0.00049358857,0.00004820941],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008745364,0.0010359426,0.00039613803,0.0016895722,0.00028646982,0.0012766077,0.0004184477,0.0006170651,0.0012377093],"category_scores_gemma":[0.0056170262,0.00019686398,0.00059322285,0.0013276542,0.0003559941,0.002884025,0.00067010237,0.0008135852,0.001510445],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00096758956,0.00039573206,0.03339457,0.00095059973,0.00021977986,0.000884802,0.002220672,0.028991025,0.06473495,0.0042528454,0.007819058,0.85516834],"study_design_scores_gemma":[0.00004800476,0.0004944544,0.025478562,0.00022938687,0.00021826029,0.0012651636,0.0031239444,0.8868781,0.057089828,0.008338736,0.016718378,0.00011716595],"about_ca_topic_score_codex":0.0025461088,"about_ca_topic_score_gemma":0.0024868643,"teacher_disagreement_score":0.0025461088,"about_ca_system_score_codex":0.000387118,"about_ca_system_score_gemma":0.0005164652,"threshold_uncertainty_score":0.0050626397},"labels":[],"label_agreement":null},{"id":"W4295199811","doi":"","title":"Procédés de traduction humaine mis en évidence dans les ruptures lexicogrammaticales avec la traduction d'un système de traduction automatique probabiliste","year":2013,"lang":"fr","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Trois-Rivières","funders":"","keywords":"Philosophy; Humanities","score_opus":0.015522552665172287,"score_gpt":0.24775272099669401,"score_spread":0.2322301683315217,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4295199811","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.45931733,0.0008622782,0.5267728,0.0013137406,0.0002800454,0.00016526037,0.00037485408,0.005210931,0.0057027717],"genre_scores_gemma":[0.8606147,0.00014355297,0.1337997,0.00011744867,0.000055377277,0.000067018205,0.00023615643,0.0005868895,0.0043790913],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99460405,0.0021520406,0.00042036234,0.0010788422,0.0014692362,0.00027545195],"domain_scores_gemma":[0.959114,0.031265724,0.0017980479,0.003980115,0.0032292055,0.0006129036],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004673901,0.0010397299,0.0010664962,0.001528812,0.0015257216,0.0029636417,0.0014819539,0.0022702925,0.009877038],"category_scores_gemma":[0.04490404,0.0008842398,0.0008540225,0.0014636576,0.001953178,0.0031008553,0.0019303041,0.0022996736,0.0021447642],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.007812561,0.0004985684,0.017959272,0.0011643623,0.00047886674,0.0028774892,0.005358882,0.17808317,0.13390677,0.037321266,0.0044341204,0.6101047],"study_design_scores_gemma":[0.00019480246,0.00064989505,0.012384499,0.00012920356,0.000251745,0.001798931,0.0012927759,0.71656793,0.21319443,0.043487888,0.009865869,0.00018200668],"about_ca_topic_score_codex":0.0042517837,"about_ca_topic_score_gemma":0.0045476505,"teacher_disagreement_score":0.009877038,"about_ca_system_score_codex":0.0014128912,"about_ca_system_score_gemma":0.0014143364,"threshold_uncertainty_score":0.033041954},"labels":[],"label_agreement":null},{"id":"W4296817069","doi":"10.18280/isi.270420","title":"Recursive LSTM for the Classification of Named Entity Recognition for Hindi Language","year":2022,"lang":"en","type":"article","venue":"Ingénierie des systèmes d information","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Named-entity recognition; Natural language processing; Artificial intelligence; Classifier (UML); Hindi; Architecture; Installation; Word (group theory); Information extraction; F1 score; Precision and recall; Contrast (vision); Entity linking; Recall; Information retrieval; Task (project management); Linguistics; Knowledge base","score_opus":0.023173097123378204,"score_gpt":0.2689661435906806,"score_spread":0.2457930464673024,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4296817069","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08205645,0.004718312,0.8620581,0.0018149482,0.0017984102,0.0002892852,0.005511428,0.028352596,0.013400456],"genre_scores_gemma":[0.6468808,0.0020667268,0.31951275,0.00073925394,0.00030864775,0.0003520551,0.008902443,0.00039771284,0.020839643],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995307,0.00008481659,0.000047595495,0.00017896623,0.00008514547,0.000072815164],"domain_scores_gemma":[0.99959093,0.0001330656,0.000040936007,0.000053156153,0.00016697678,0.00001488465],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00070092856,0.0007443215,0.00041213733,0.0006934134,0.0004526931,0.0008533111,0.0011178198,0.00092178705,0.00535951],"category_scores_gemma":[0.0016339023,0.0002713918,0.00073152117,0.00097615126,0.00028296543,0.0018653594,0.000569232,0.0014564452,0.0034908743],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028392987,0.00016756318,0.0019046529,0.0004229118,0.00012333931,0.00034019176,0.0002982204,0.041105326,0.037438873,0.0047704275,0.022508271,0.89063644],"study_design_scores_gemma":[0.000018740715,0.00020933364,0.002432436,0.00008270794,0.00008242731,0.00022871223,0.00016514973,0.943503,0.031147083,0.0071921567,0.014893423,0.00004472213],"about_ca_topic_score_codex":0.010331394,"about_ca_topic_score_gemma":0.012321485,"teacher_disagreement_score":0.010331394,"about_ca_system_score_codex":0.0008175315,"about_ca_system_score_gemma":0.001039941,"threshold_uncertainty_score":0.020542502},"labels":[],"label_agreement":null},{"id":"W4297747537","doi":"","title":"Module NooJ du français. Traitement automatique de corpus de français parlé régional","year":2014,"lang":"fr","type":"article","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Moncton","funders":"","keywords":"Computer science","score_opus":0.008986875020506883,"score_gpt":0.2221124291164217,"score_spread":0.21312555409591483,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4297747537","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.67536366,0.0029518406,0.1482584,0.0012377958,0.0005192441,0.00048323325,0.094940536,0.03897664,0.03726872],"genre_scores_gemma":[0.6619766,0.0010005427,0.14717895,0.00019684453,0.00010431975,0.0003728804,0.13108075,0.0040916754,0.05399756],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9992799,0.00010653988,0.000052816362,0.00030331002,0.0001623288,0.00009512741],"domain_scores_gemma":[0.99899274,0.00031171797,0.0000624435,0.00014640906,0.0004177336,0.000068998386],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007410259,0.001093931,0.0004737969,0.002062138,0.00068065844,0.0016865532,0.0004893279,0.0005574256,0.009129735],"category_scores_gemma":[0.0023918173,0.00039166244,0.00066435017,0.0014207122,0.00033488465,0.0010003991,0.00064267754,0.0005254491,0.0047440245],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014850325,0.00018594692,0.03659789,0.0012958461,0.00048700423,0.0014581501,0.003002274,0.014981751,0.20797448,0.0061841225,0.11101545,0.61533207],"study_design_scores_gemma":[0.0001977691,0.0003449283,0.2120454,0.00024142106,0.0004403944,0.0017120886,0.00294893,0.09375879,0.19114046,0.003388623,0.49358144,0.00019972857],"about_ca_topic_score_codex":0.16834952,"about_ca_topic_score_gemma":0.16868417,"teacher_disagreement_score":0.16834952,"about_ca_system_score_codex":0.0011735752,"about_ca_system_score_gemma":0.0018635916,"threshold_uncertainty_score":0.3347392},"labels":[],"label_agreement":null},{"id":"W4297878801","doi":"10.1590/scielopreprints.4755","title":"ARCHIVING AND LANGUAGE DOCUMENTATION","year":2022,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Documentation; Relevance (law); Focus (optics); Indigenous; Indigenous language; Library science; Linguistics; Political science; World Wide Web; Computer science","score_opus":0.010051730414711966,"score_gpt":0.30364803710381866,"score_spread":0.2935963066891067,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4297878801","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0245214,0.04370256,0.46254215,0.056926783,0.004990481,0.00039845012,0.0017067874,0.0037394161,0.401472],"genre_scores_gemma":[0.3471783,0.04519582,0.435408,0.006505694,0.0034529918,0.0004431121,0.0029238597,0.0028927433,0.15599945],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.98705065,0.006535875,0.001228051,0.0014188587,0.0033863513,0.00038019722],"domain_scores_gemma":[0.979788,0.0072813593,0.001558857,0.0071208114,0.0037167633,0.0005341996],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008750267,0.0005630715,0.00059461,0.0053778407,0.00458478,0.017691437,0.0014983681,0.0019605146,0.013983179],"category_scores_gemma":[0.029193835,0.0005293687,0.0006512645,0.007011344,0.008135621,0.014628841,0.007953698,0.0032856101,0.0045463643],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003708526,0.000032853466,0.0012730529,0.00063144503,0.00002519794,0.00029107183,0.012533899,0.0005540265,0.0019182211,0.63472927,0.03428587,0.31368807],"study_design_scores_gemma":[0.000009826814,0.000024722849,0.0011833397,0.0010281132,0.000023938679,0.0012996637,0.0053984146,0.00152927,0.0023359738,0.23283406,0.7542882,0.000044559605],"about_ca_topic_score_codex":0.008002601,"about_ca_topic_score_gemma":0.007093879,"teacher_disagreement_score":0.017691437,"about_ca_system_score_codex":0.0033620798,"about_ca_system_score_gemma":0.0067256014,"threshold_uncertainty_score":0.04677844},"labels":[],"label_agreement":null},{"id":"W4298141459","doi":"","title":"Des modules combinatoires intégrés à un analyseur syntaxique pour une classification automatique de documents","year":2003,"lang":"fr","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science","score_opus":0.02051212816341251,"score_gpt":0.26940283746914157,"score_spread":0.24889070930572904,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4298141459","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02931597,0.0005352694,0.96308684,0.0003895341,0.00005917328,0.00013102026,0.00033587724,0.0023382048,0.003808059],"genre_scores_gemma":[0.14399792,0.00051374454,0.84008676,0.00023509726,0.00013477345,0.00028768918,0.0012873757,0.00102211,0.012434605],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99425584,0.0016287747,0.0006032585,0.0015217436,0.0017192481,0.000271116],"domain_scores_gemma":[0.99521554,0.0026142667,0.00034810777,0.0006026953,0.0010648381,0.00015463137],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0039437544,0.0017419046,0.0012906939,0.005747271,0.0013227452,0.005350775,0.001084805,0.0012140817,0.0067876945],"category_scores_gemma":[0.0074966885,0.001193588,0.0036680573,0.004001785,0.0023881872,0.007068713,0.0019322359,0.0024292967,0.0023902114],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000496559,0.00026250078,0.00974967,0.00046994706,0.0005463582,0.0005465909,0.0019703805,0.013155691,0.063817106,0.26666573,0.0071795275,0.63513994],"study_design_scores_gemma":[0.00011592471,0.00031888188,0.019942727,0.000358154,0.000774882,0.0012263436,0.0013249861,0.37567905,0.08823058,0.4198932,0.09190605,0.00022916938],"about_ca_topic_score_codex":0.009872274,"about_ca_topic_score_gemma":0.010263572,"teacher_disagreement_score":0.009872274,"about_ca_system_score_codex":0.0029193026,"about_ca_system_score_gemma":0.0018984518,"threshold_uncertainty_score":0.022707105},"labels":[],"label_agreement":null},{"id":"W4298470066","doi":"","title":"La lemmatisation et l'encodage grammatical permettent-ils de reconnaître l'auteur d'un texte ?","year":2002,"lang":"fr","type":"article","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Linguistic Association","funders":"","keywords":"Humanities; Philosophy; Auteur theory; Art; Literature","score_opus":0.02081450290037599,"score_gpt":0.24670236879907698,"score_spread":0.225887865898701,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4298470066","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05438455,0.0019997288,0.8439206,0.018334908,0.0036210392,0.00012558606,0.0006902397,0.010741997,0.0661814],"genre_scores_gemma":[0.59687847,0.0020526461,0.30655083,0.0032857615,0.0021826853,0.0001710386,0.0011479874,0.009548655,0.07818185],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9940148,0.0028394412,0.00047684112,0.0012662524,0.00088484946,0.00051785196],"domain_scores_gemma":[0.98736274,0.005120262,0.00091583154,0.0039564758,0.002433781,0.00021094109],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003966708,0.0014635635,0.0014339718,0.0013768524,0.0018152584,0.0076604644,0.0012138509,0.0043233475,0.022444285],"category_scores_gemma":[0.022472242,0.0014233536,0.0013615588,0.001468563,0.005499149,0.020939969,0.0023682264,0.0033022624,0.015755156],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000673561,0.00013833173,0.005905915,0.0007243653,0.00015282692,0.0020019982,0.0044713607,0.0020855104,0.042845584,0.60489583,0.027433768,0.30867097],"study_design_scores_gemma":[0.00014718332,0.00019721712,0.0053025377,0.0004671641,0.00034968505,0.00456074,0.0039044241,0.04527817,0.08989258,0.47465557,0.37502438,0.00022034356],"about_ca_topic_score_codex":0.0040872777,"about_ca_topic_score_gemma":0.004901004,"teacher_disagreement_score":0.022444285,"about_ca_system_score_codex":0.0014181263,"about_ca_system_score_gemma":0.0020860028,"threshold_uncertainty_score":0.07508361},"labels":[],"label_agreement":null},{"id":"W4299149419","doi":"10.1007/978-3-031-01880-0_2","title":"Background: Corpora and Evaluation Methods","year":2011,"lang":"en","type":"book-chapter","venue":"Synthesis lectures on data management","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Automatic summarization; Computer science; Natural language processing; Terminology; Information retrieval; Artificial intelligence; Linguistics","score_opus":0.12909839802379802,"score_gpt":0.36666621950415146,"score_spread":0.23756782148035344,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4299149419","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0043393527,0.1135256,0.33462548,0.014791128,0.008545862,0.0034299989,0.12605299,0.021383734,0.37330586],"genre_scores_gemma":[0.032341428,0.07594899,0.4653009,0.005752387,0.011406948,0.013130109,0.2213294,0.013852574,0.16093734],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9916618,0.002889308,0.0011054365,0.0012211222,0.002829095,0.00029315014],"domain_scores_gemma":[0.97779787,0.009660398,0.00078547833,0.0046747276,0.0063739773,0.00070756284],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011861021,0.001924264,0.0019076095,0.01141107,0.0022308126,0.007908425,0.0037325465,0.0022427153,0.106478915],"category_scores_gemma":[0.036325045,0.0018157485,0.00097290165,0.016130548,0.0020571973,0.008550759,0.003450925,0.003625573,0.09412237],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013155509,0.00019881378,0.00034058097,0.0032555896,0.00004388778,0.00009901797,0.00026241867,0.0017861358,0.0019061938,0.034946904,0.542401,0.414628],"study_design_scores_gemma":[0.00005256507,0.00006729339,0.0010953067,0.0015468545,0.000049359787,0.0002828196,0.00011539412,0.0016383545,0.0039502005,0.043340296,0.9478043,0.000057329326],"about_ca_topic_score_codex":0.003631196,"about_ca_topic_score_gemma":0.003764516,"teacher_disagreement_score":0.106478915,"about_ca_system_score_codex":0.002276164,"about_ca_system_score_gemma":0.004100743,"threshold_uncertainty_score":0.35620743},"labels":[],"label_agreement":null},{"id":"W4299697650","doi":"10.1007/978-3-031-01880-0_4","title":"Summarizing Text Conversations","year":2011,"lang":"en","type":"book-chapter","venue":"Synthesis lectures on data management","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Natural language processing; Linguistics; Philosophy","score_opus":0.05137841069642054,"score_gpt":0.2638444147118555,"score_spread":0.21246600401543494,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4299697650","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013901742,0.006654597,0.34041756,0.019152695,0.010791068,0.00058475486,0.01899474,0.019260127,0.57024276],"genre_scores_gemma":[0.20956472,0.0053194007,0.14664835,0.0027893428,0.005523974,0.0008582744,0.04118906,0.010771195,0.5773357],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9985386,0.0006035158,0.00008478641,0.00028781075,0.00038301296,0.0001022454],"domain_scores_gemma":[0.9969831,0.0013399965,0.0001498841,0.00047466965,0.00092441327,0.00012798829],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010920168,0.0012313225,0.0005278677,0.0036681949,0.002674967,0.004994449,0.0010009478,0.0009862422,0.10804025],"category_scores_gemma":[0.009821166,0.0005153288,0.00046159176,0.0029648647,0.0009678321,0.0066062366,0.0028458012,0.0019613078,0.055046435],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001926373,0.000042274365,0.00041068057,0.00036280783,0.000022653378,0.00022640519,0.004503747,0.00069153676,0.0041335607,0.1540418,0.4984479,0.33692393],"study_design_scores_gemma":[0.0000128667225,0.000020079748,0.0004348539,0.00028587115,0.000024173933,0.000118844015,0.0021242693,0.0027257083,0.0036997548,0.06723307,0.9232952,0.000025211504],"about_ca_topic_score_codex":0.0019135835,"about_ca_topic_score_gemma":0.002268753,"teacher_disagreement_score":0.10804025,"about_ca_system_score_codex":0.0011588864,"about_ca_system_score_gemma":0.0011351904,"threshold_uncertainty_score":0.36143064},"labels":[],"label_agreement":null},{"id":"W4300790807","doi":"","title":"A Software for the Analysis of Scripted Dialogs Based on Surface Markers","year":2003,"lang":"en","type":"article","venue":"DOAJ (DOAJ: Directory of Open Access Journals)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval; Université du Québec à Trois-Rivières","funders":"","keywords":"Computer science; Software; Surface (topology); Programming language; Mathematics; Geometry","score_opus":0.1541525535487932,"score_gpt":0.504193049879603,"score_spread":0.3500404963308098,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4300790807","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0048391176,0.00009850432,0.8145374,0.000061442646,0.00005852887,0.00033975163,0.0027025524,0.17379904,0.0035635405],"genre_scores_gemma":[0.04595894,0.00017329841,0.9285008,0.00015963033,0.00003682691,0.0008610906,0.005649069,0.009721754,0.008938651],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9994271,0.000080743164,0.0000641456,0.00022585971,0.00016747166,0.000034747045],"domain_scores_gemma":[0.99883205,0.00063018064,0.0000970027,0.00020857187,0.00017537067,0.000056778466],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000829473,0.0013138421,0.00066663465,0.0017373742,0.00064439705,0.0019609625,0.0011300714,0.00072116236,0.021696925],"category_scores_gemma":[0.0034684269,0.0007779117,0.00091210776,0.0011053133,0.0005965821,0.002127196,0.0013096724,0.0015918049,0.006843463],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008921363,0.00029269658,0.0035422035,0.0014745295,0.00025907825,0.0007799855,0.0015508559,0.00863266,0.10916846,0.029428815,0.059803877,0.7841747],"study_design_scores_gemma":[0.0004514419,0.00048434307,0.0125663085,0.00028574307,0.00026906104,0.0032649417,0.0005860759,0.44306153,0.12676138,0.06128715,0.35067555,0.00030657309],"about_ca_topic_score_codex":0.002714007,"about_ca_topic_score_gemma":0.0025953886,"teacher_disagreement_score":0.021696925,"about_ca_system_score_codex":0.00054430193,"about_ca_system_score_gemma":0.0012752071,"threshold_uncertainty_score":0.07258344},"labels":[],"label_agreement":null},{"id":"W4300987828","doi":"","title":"Lucy-n~: une extension n-synchrone de Lustre","year":2010,"lang":"fr","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Prevention of Organ Failure","funders":"","keywords":"Lustre (file system); Computer science; Art; World Wide Web","score_opus":0.015130161318044873,"score_gpt":0.250733965886814,"score_spread":0.23560380456876914,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4300987828","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012159397,0.00031848458,0.937528,0.00040788873,0.0006297155,0.0001385645,0.0006248013,0.03371112,0.014482103],"genre_scores_gemma":[0.28562474,0.00068283844,0.63036656,0.0016747039,0.0010007431,0.0011051575,0.0029151307,0.018541645,0.058088534],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99673444,0.0006375191,0.00026017055,0.00086905825,0.0011477424,0.00035091906],"domain_scores_gemma":[0.99730647,0.0010914695,0.0001101339,0.0008435659,0.000463915,0.00018454321],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024358311,0.0021585103,0.0018173202,0.001274943,0.0014167825,0.0027852736,0.0021914535,0.0018806877,0.018713461],"category_scores_gemma":[0.0051002535,0.0012809724,0.0020680302,0.0010705813,0.0025678838,0.0060052,0.004684876,0.0044011865,0.008327179],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0030681647,0.00040766242,0.0013310598,0.0011894671,0.00019722217,0.00083983934,0.0012432059,0.033616416,0.05994402,0.54293656,0.048902232,0.3063242],"study_design_scores_gemma":[0.0013262758,0.00082623464,0.001054495,0.0003908857,0.00027169057,0.00071963685,0.00028187942,0.21923426,0.09299163,0.3395812,0.34290728,0.00041460505],"about_ca_topic_score_codex":0.0037272468,"about_ca_topic_score_gemma":0.004267067,"teacher_disagreement_score":0.018713461,"about_ca_system_score_codex":0.0015465145,"about_ca_system_score_gemma":0.0016238814,"threshold_uncertainty_score":0.06260282},"labels":[],"label_agreement":null},{"id":"W4301778581","doi":"10.1007/978-3-031-01862-6_2","title":"Background","year":2018,"lang":"en","type":"book-chapter","venue":"Synthesis lectures on data management","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Question answering; Natural language processing; Dialog box; Parsing; Natural language; Section (typography); Context (archaeology); Artificial intelligence; Natural language understanding; Semantic interpretation; Syntax; Linguistics; World Wide Web","score_opus":0.04900264844609401,"score_gpt":0.29041669427853817,"score_spread":0.24141404583244416,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4301778581","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0011031062,0.005476496,0.011212724,0.0044696177,0.003746625,0.00030451495,0.00850158,0.0011771147,0.9640083],"genre_scores_gemma":[0.0061240434,0.005245336,0.0058216676,0.0025508308,0.0013191293,0.00038220486,0.009005492,0.00056504534,0.9689863],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9992524,0.000084833424,0.000029753042,0.00021609737,0.00030620332,0.00011068437],"domain_scores_gemma":[0.9990207,0.00014914406,0.000042874457,0.00013515381,0.00040365945,0.0002484797],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00084562204,0.0010675046,0.0006510498,0.0016669346,0.0015796963,0.0033299983,0.0018868966,0.0013999316,0.54480577],"category_scores_gemma":[0.002806334,0.00028153168,0.00049136695,0.0015143459,0.0005999031,0.0024730712,0.0022489855,0.0016388694,0.40653628],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010051907,0.00010819353,0.00025443154,0.00046881146,0.0000054433976,0.000108021464,0.00020085332,0.00021236848,0.00075403985,0.081518225,0.7007273,0.21554182],"study_design_scores_gemma":[0.000003850557,0.000011171395,0.000080677506,0.00008361505,0.0000018147557,0.000042392883,0.000032747623,0.000028890196,0.00014127578,0.006571313,0.99299896,0.0000031723757],"about_ca_topic_score_codex":0.002485386,"about_ca_topic_score_gemma":0.0033718178,"teacher_disagreement_score":0.54480577,"about_ca_system_score_codex":0.0013573497,"about_ca_system_score_gemma":0.0026525902,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W4301884066","doi":"10.3115/v1/w15-09","title":"Proceedings of the 11th Workshop on Multiword Expressions","year":2015,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of New Brunswick","funders":"","keywords":"Computer science; Natural language processing","score_opus":0.03667260991321998,"score_gpt":0.30996375619625416,"score_spread":0.2732911462830342,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4301884066","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024444435,0.034776833,0.78381777,0.01591985,0.015120326,0.0007241254,0.0042831004,0.008620653,0.11229294],"genre_scores_gemma":[0.1249268,0.02020879,0.59062743,0.004189991,0.005598831,0.0011034708,0.023174837,0.007531091,0.22263885],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99598503,0.0017389469,0.00038031765,0.0009381631,0.0007227458,0.0002348418],"domain_scores_gemma":[0.9945931,0.0023576096,0.00013761754,0.001442795,0.0011641611,0.00030470928],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0057162694,0.0016599685,0.0018961718,0.0021472047,0.0015270779,0.0068154256,0.0029146592,0.001886972,0.057550393],"category_scores_gemma":[0.0102422675,0.000722502,0.0019288567,0.0025724985,0.0014927626,0.0105357645,0.004318965,0.0029730673,0.022458717],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005966434,0.00036254092,0.0011464509,0.00093372975,0.00018278322,0.0006096038,0.0021057401,0.0026800844,0.00936297,0.07385516,0.2651429,0.6430214],"study_design_scores_gemma":[0.000055323297,0.00009395071,0.0012699044,0.000520256,0.00010178957,0.0006133067,0.00091480266,0.016204845,0.0073642973,0.077440284,0.8953546,0.00006658064],"about_ca_topic_score_codex":0.0016872463,"about_ca_topic_score_gemma":0.0023858547,"teacher_disagreement_score":0.057550393,"about_ca_system_score_codex":0.0016921317,"about_ca_system_score_gemma":0.0017798693,"threshold_uncertainty_score":0.19252527},"labels":[],"label_agreement":null},{"id":"W4302574100","doi":"10.21428/594757db.146247f6","title":"A working model for textual Membership Query Synthesis","year":2022,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Artificial intelligence; Task (project management); Constraint (computer-aided design); Set (abstract data type); Encoder; Domain (mathematical analysis); Natural language processing; Selection (genetic algorithm); Generative grammar; Machine learning; Mathematics","score_opus":0.04543900458264734,"score_gpt":0.2781397287591677,"score_spread":0.23270072417652038,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4302574100","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022679523,0.00038252768,0.9707365,0.0009355355,0.00006589386,0.00009631011,0.0005816855,0.0018297271,0.002692196],"genre_scores_gemma":[0.69197994,0.00033997933,0.29209313,0.0008344816,0.0002458211,0.00048219718,0.0020653866,0.0005149476,0.011444069],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986376,0.0004644142,0.00008558645,0.0004577084,0.00025728083,0.000097413744],"domain_scores_gemma":[0.9972671,0.0016400543,0.00013364291,0.00046237977,0.00037274224,0.00012413225],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028498622,0.00064207695,0.0008744981,0.0009481288,0.0004534333,0.0019103513,0.0026600487,0.0020128693,0.006767754],"category_scores_gemma":[0.0096213985,0.0004657038,0.00103265,0.00086485007,0.0013559478,0.0042689173,0.0016883106,0.0016620622,0.0019599772],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000878681,0.0004223269,0.0031460817,0.0005619853,0.00014307765,0.00048623388,0.0015721787,0.3697317,0.039588623,0.242918,0.015261966,0.32528916],"study_design_scores_gemma":[0.00002436053,0.00003344061,0.00010049502,0.000008989945,0.000009631144,0.000045647215,0.000028215032,0.96379316,0.0019959975,0.03220293,0.0017479295,0.00000928428],"about_ca_topic_score_codex":0.0023411948,"about_ca_topic_score_gemma":0.0024055315,"teacher_disagreement_score":0.006767754,"about_ca_system_score_codex":0.0011332969,"about_ca_system_score_gemma":0.0009395495,"threshold_uncertainty_score":0.022640407},"labels":[],"label_agreement":null},{"id":"W43028483","doi":"10.63317/4wof6fn4wqjp","title":"NLGbAse: A Free Linguistic Resource for Natural Language Processing Systems","year":2010,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Metadata; Annotation; Natural language processing; Context (archaeology); Artificial intelligence; Domain (mathematical analysis); Natural language; Resource (disambiguation); Field (mathematics); Information retrieval; World Wide Web","score_opus":0.012585524952696717,"score_gpt":0.2918051607984203,"score_spread":0.27921963584572357,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W43028483","genre_codex":"software","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":"software","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004693298,0.0010583684,0.23654985,0.0013334778,0.0005216927,0.0015118499,0.22506481,0.5074107,0.02185592],"genre_scores_gemma":[0.025181856,0.0008874528,0.30804068,0.0010098126,0.00020202436,0.0037474674,0.5901749,0.058354646,0.012401215],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9964562,0.0009452146,0.00063855975,0.00059514417,0.0011640483,0.00020080109],"domain_scores_gemma":[0.9848157,0.007616879,0.0011016423,0.0036045376,0.0020695722,0.00079161924],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0046179397,0.0021526446,0.0021846802,0.0069345487,0.0018654214,0.0041737924,0.004476358,0.003097374,0.047389954],"category_scores_gemma":[0.026101006,0.0021019753,0.0012144343,0.005525252,0.0013286625,0.009782952,0.007362831,0.0034596205,0.050931867],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009760351,0.00023837698,0.0011083017,0.00428166,0.00019252731,0.001421986,0.0014081651,0.0025770606,0.020804267,0.017228141,0.8219476,0.12781589],"study_design_scores_gemma":[0.0003651357,0.00010819093,0.0021277135,0.0005043096,0.00007859536,0.0011179212,0.00028475604,0.012237463,0.020038899,0.022434365,0.9404384,0.00026418993],"about_ca_topic_score_codex":0.0054496857,"about_ca_topic_score_gemma":0.006860369,"teacher_disagreement_score":0.047389954,"about_ca_system_score_codex":0.0015703352,"about_ca_system_score_gemma":0.0035445997,"threshold_uncertainty_score":0.15853518},"labels":[],"label_agreement":null},{"id":"W4303437233","doi":"10.7202/1092196ar","title":"Comparaison d’un texte original et de ses rétrotraductions : que disent les mesures textométriques ?","year":2022,"lang":"fr","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Philosophy","score_opus":0.05619626575687161,"score_gpt":0.3546783508723819,"score_spread":0.29848208511551033,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4303437233","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8139375,0.0042437855,0.03483154,0.0025073055,0.0006536778,0.0003154037,0.0072880513,0.000390564,0.13583219],"genre_scores_gemma":[0.95020604,0.0015853643,0.021541867,0.00023002253,0.0002452017,0.0003475344,0.0042847376,0.0008894301,0.020669766],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99625707,0.0014746073,0.0003683401,0.00069389114,0.0010828641,0.0001232141],"domain_scores_gemma":[0.9835781,0.009228085,0.0012840729,0.0022867878,0.0034183734,0.00020467877],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003792996,0.00056411553,0.00041244,0.006234109,0.0018264567,0.0051982007,0.00059292384,0.00058641424,0.010705987],"category_scores_gemma":[0.019063681,0.00029108766,0.00025542884,0.009199651,0.0033510653,0.004434879,0.002345759,0.0010607839,0.0020962209],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00095932634,0.00012608286,0.048537984,0.0034137352,0.00022240911,0.001634408,0.32561475,0.0013186224,0.038646616,0.08612643,0.014887724,0.47851193],"study_design_scores_gemma":[0.00005720743,0.00026456662,0.17185746,0.0011130611,0.0001958145,0.0022048492,0.12773854,0.0024463688,0.018118624,0.015370992,0.66047245,0.00015998566],"about_ca_topic_score_codex":0.00793864,"about_ca_topic_score_gemma":0.0135263,"teacher_disagreement_score":0.010705987,"about_ca_system_score_codex":0.0019682148,"about_ca_system_score_gemma":0.0013721446,"threshold_uncertainty_score":0.03581506},"labels":[],"label_agreement":null},{"id":"W4306398997","doi":"10.5430/wjel.v12n8p242","title":"The Syntactic Structure of an Introductory PP in Standard Arabic: A Non-Transformational Approach","year":2022,"lang":"en","type":"article","venue":"World Journal of English Language","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Complement (music); Linguistics; Computer science; Utterance; Phrase structure rules; Phrase; Transformational grammar; Arabic; Transformational leadership; Natural language processing; Head (geology); Syntactic structure; Grammar; Artificial intelligence; Syntax; Philosophy; Psychology","score_opus":0.004209481971302761,"score_gpt":0.23484551088409455,"score_spread":0.2306360289127918,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4306398997","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.46627167,0.0016766554,0.41199178,0.0023347065,0.00019032434,0.000421076,0.0015676152,0.0012695971,0.11427657],"genre_scores_gemma":[0.94058985,0.0006130117,0.053120404,0.00014130161,0.00010036511,0.00012567984,0.00085079554,0.00026059133,0.004197985],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99905604,0.00041295937,0.00007745252,0.00020433751,0.00018260622,0.0000666442],"domain_scores_gemma":[0.9971644,0.0015960402,0.00031765425,0.000431775,0.00043607064,0.000054147495],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001069357,0.00050832704,0.00039622452,0.002086806,0.001662625,0.002953152,0.00089399103,0.0007747023,0.0058483914],"category_scores_gemma":[0.003577687,0.00062871363,0.0005025422,0.0028649676,0.003591399,0.0045408923,0.0016218617,0.0016750451,0.0012794276],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002537231,0.00009957976,0.0058369227,0.00049476343,0.00002916448,0.0021767535,0.027768487,0.002647236,0.024193807,0.8439644,0.0022435584,0.09029154],"study_design_scores_gemma":[0.000069368645,0.00055401085,0.032867096,0.00051894924,0.00016166638,0.005988198,0.023682935,0.06374036,0.04474787,0.644987,0.18244724,0.00023535032],"about_ca_topic_score_codex":0.0017772467,"about_ca_topic_score_gemma":0.0011554989,"teacher_disagreement_score":0.0058483914,"about_ca_system_score_codex":0.0018902142,"about_ca_system_score_gemma":0.0010795534,"threshold_uncertainty_score":0.019564748},"labels":[],"label_agreement":null},{"id":"W4306964455","doi":"10.31234/osf.io/qjnpv","title":"The Plausibility of Sampling as an Algorithmic Theory of Sentence Processing","year":2022,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Mila - Quebec Artificial Intelligence Institute; Canadian Institute for Advanced Research; McGill University","funders":"","keywords":"Computer science; Variance (accounting); Sampling (signal processing); Context (archaeology); Reading (process); Class (philosophy); Parsing; Sentence; Artificial intelligence; Natural language processing; Algorithm; Linguistics","score_opus":0.035177885962474044,"score_gpt":0.337574704823559,"score_spread":0.30239681886108494,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4306964455","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.066286504,0.0011568695,0.91928136,0.0039557572,0.00014678897,0.00012384942,0.00028033726,0.0011910249,0.0075775343],"genre_scores_gemma":[0.7526074,0.0010715155,0.23920953,0.0015575531,0.00062635296,0.00038991278,0.0005650484,0.0005258519,0.0034468374],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99432653,0.0023462411,0.00022280999,0.001499181,0.001309928,0.00029526956],"domain_scores_gemma":[0.9204278,0.0643631,0.0036866313,0.008484319,0.0022304044,0.0008077867],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009475955,0.0012420476,0.0014177407,0.0021811489,0.0015069082,0.0043806727,0.0029737598,0.0022965544,0.0049592466],"category_scores_gemma":[0.061149273,0.0011776373,0.0019115119,0.0021797763,0.0061459458,0.014985383,0.001941599,0.0053220256,0.00094759866],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006869675,0.00037965583,0.019111944,0.0005667614,0.00040368715,0.00032857695,0.0012820077,0.1939338,0.006605951,0.67131424,0.004892098,0.10049425],"study_design_scores_gemma":[0.000037949736,0.00015642561,0.0026244572,0.00003123337,0.000051227675,0.00021178507,0.000065832864,0.3882801,0.0022147966,0.6045193,0.0017583496,0.000048488026],"about_ca_topic_score_codex":0.0024810953,"about_ca_topic_score_gemma":0.0025713095,"teacher_disagreement_score":0.009475955,"about_ca_system_score_codex":0.0026828249,"about_ca_system_score_gemma":0.0019071638,"threshold_uncertainty_score":0.050114274},"labels":[],"label_agreement":null},{"id":"W4307532870","doi":"","title":"A Second Life for TIIARA: From Bilingual to Multilingual!","year":2016,"lang":"en","type":"article","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Linguistics; Psychology; Philosophy","score_opus":0.01334676257450387,"score_gpt":0.25369497129074303,"score_spread":0.24034820871623916,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4307532870","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0253102,0.0088789165,0.030405948,0.11952336,0.026749026,0.00020082683,0.0019880785,0.0054915464,0.7814521],"genre_scores_gemma":[0.3192283,0.0051875515,0.032989394,0.024694784,0.0063131326,0.00042460384,0.0039081234,0.009279818,0.5979743],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9969934,0.0011997664,0.000115012546,0.00047779133,0.0005770828,0.0006368233],"domain_scores_gemma":[0.99253803,0.0010113735,0.000208535,0.0010716612,0.001910706,0.0032596728],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004003642,0.000970355,0.0007819303,0.001324895,0.00917601,0.012208437,0.0014629546,0.0028755725,0.13076949],"category_scores_gemma":[0.00857318,0.0004935612,0.0006415858,0.0014315309,0.002984554,0.014391079,0.0094802985,0.0060429918,0.057250362],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00070963387,0.00026736045,0.0026081474,0.00056482793,0.00002262387,0.00092960714,0.027339967,0.00012114452,0.0043921396,0.12488054,0.60321593,0.23494805],"study_design_scores_gemma":[0.00002364007,0.00006651205,0.000644063,0.0001829037,0.000009696814,0.00056090846,0.008695365,0.00015792748,0.001013552,0.011773365,0.9768346,0.000037460944],"about_ca_topic_score_codex":0.0064004418,"about_ca_topic_score_gemma":0.012752719,"teacher_disagreement_score":0.13076949,"about_ca_system_score_codex":0.0030813373,"about_ca_system_score_gemma":0.0066074287,"threshold_uncertainty_score":0.43746752},"labels":[],"label_agreement":null},{"id":"W4307751734","doi":"10.3390/app122111038","title":"Framework for Handling Rare Word Problems in Neural Machine Translation System Using Multi-Word Expressions","year":2022,"lang":"en","type":"article","venue":"Applied Sciences","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Brandon University","funders":"","keywords":"Computer science; Artificial intelligence; Natural language processing; Machine translation; Fluency; Word (group theory); Test set; Set (abstract data type); Word embedding; Vocabulary; Translation (biology); Embedding; Linguistics; Programming language","score_opus":0.0708010812538504,"score_gpt":0.32704081290534803,"score_spread":0.2562397316514976,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4307751734","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0027334555,0.00025124318,0.99358916,0.00024200657,0.00005205614,0.00010167458,0.000059195223,0.0007302225,0.002240935],"genre_scores_gemma":[0.1673171,0.0008511153,0.8210146,0.00024767086,0.00017056454,0.00076213124,0.0005042136,0.00024984102,0.0088828],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989428,0.00027997635,0.00012460386,0.00030038366,0.00024867035,0.00010368495],"domain_scores_gemma":[0.9995597,0.00013015594,0.0000449698,0.000055539924,0.0001833628,0.00002626968],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015097815,0.00079324556,0.00085855566,0.0010776401,0.0010069794,0.001619817,0.0019723452,0.0011032163,0.0048252456],"category_scores_gemma":[0.0017425738,0.00045982984,0.0014182925,0.00085447426,0.00088550913,0.0019829914,0.0015825718,0.0015285261,0.0019818454],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017542492,0.00022394018,0.0017555649,0.00051874004,0.00016903867,0.0015594695,0.0008274172,0.4476537,0.019597456,0.22653134,0.00679933,0.2941885],"study_design_scores_gemma":[0.00002242288,0.00007662735,0.00019881841,0.000034023782,0.000047527625,0.00019250708,0.000060828483,0.95071906,0.002961969,0.038110435,0.0075535285,0.000022152264],"about_ca_topic_score_codex":0.009112061,"about_ca_topic_score_gemma":0.010254761,"teacher_disagreement_score":0.009112061,"about_ca_system_score_codex":0.0010452748,"about_ca_system_score_gemma":0.0025694196,"threshold_uncertainty_score":0.018118024},"labels":[],"label_agreement":null},{"id":"W4308105639","doi":"10.3389/fpls.2022.1011948","title":"Plastaumatic: Automating plastome assembly and annotation","year":2022,"lang":"en","type":"article","venue":"Frontiers in Plant Science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Natural Sciences and Engineering Research Council of Canada; Génome Québec; Compute Canada","keywords":"Annotation; Computational biology; Chloroplast DNA; Biology; Computer science; Genome; Genetics; Gene","score_opus":0.0074268614437464665,"score_gpt":0.23705410943405894,"score_spread":0.22962724799031248,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4308105639","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.052730333,0.0010067581,0.7406317,0.0003754736,0.00036041893,0.0007145025,0.031973124,0.16402642,0.008181358],"genre_scores_gemma":[0.06421024,0.00045277516,0.84089106,0.00024632283,0.000061237224,0.0006623726,0.0718318,0.014399606,0.0072445483],"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99852043,0.00016238898,0.00015981177,0.00073845737,0.00030640303,0.00011250437],"domain_scores_gemma":[0.99875903,0.0003534539,0.0001537976,0.00035726328,0.00029405867,0.00008238955],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017358831,0.0014406389,0.0009127745,0.001718024,0.0009791851,0.001979437,0.0014849515,0.000601099,0.007886815],"category_scores_gemma":[0.0033496467,0.0011224989,0.0014773498,0.0013343602,0.000501756,0.0015297964,0.0021118945,0.001914653,0.010607602],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015033156,0.00017570738,0.00562729,0.0019150564,0.00021182715,0.0008540608,0.002136463,0.004316846,0.55662197,0.005194812,0.06746948,0.35397312],"study_design_scores_gemma":[0.0002585928,0.00048421783,0.018478377,0.00036148663,0.000195469,0.0026733424,0.0007440356,0.07336245,0.48200434,0.009710636,0.41133434,0.00039264123],"about_ca_topic_score_codex":0.0019145382,"about_ca_topic_score_gemma":0.002521946,"teacher_disagreement_score":0.007886815,"about_ca_system_score_codex":0.00049270527,"about_ca_system_score_gemma":0.0013093781,"threshold_uncertainty_score":0.026384056},"labels":[],"label_agreement":null},{"id":"W4308571509","doi":"10.5334/johd.97","title":"CREMMA Medii Aevi: Literary Manuscript Text Recognition in Latin","year":2023,"lang":"en","type":"article","venue":"Journal of Open Humanities Data","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Latin Americans; Computer science; Representation (politics); Natural language processing; Artificial intelligence; Segmentation; Medieval Latin; Information retrieval; Linguistics; History; Classics; Political science","score_opus":0.23807547872461002,"score_gpt":0.3631402314952308,"score_spread":0.12506475277062076,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4308571509","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.34479994,0.013774905,0.04615871,0.0039389567,0.0039842306,0.0009403466,0.45280516,0.05229851,0.08129923],"genre_scores_gemma":[0.2153117,0.0019358837,0.07432689,0.0009791129,0.0006426053,0.00084281183,0.6760392,0.0013174431,0.02860445],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9993241,0.000116116564,0.000097228534,0.00023888412,0.00015988504,0.00006382694],"domain_scores_gemma":[0.9988243,0.00023327926,0.00010659789,0.0003107163,0.0004113583,0.000113755115],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000636499,0.0011022341,0.0004920916,0.0035583205,0.0009871394,0.0018980344,0.0011778253,0.0011208731,0.008566272],"category_scores_gemma":[0.002731885,0.00016818239,0.00088205317,0.0022429319,0.0005243726,0.0014111586,0.0012719623,0.0007248707,0.011284828],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008196988,0.00042695802,0.018782612,0.0020980136,0.0001613163,0.0014578274,0.0013214715,0.004214742,0.029346272,0.00385063,0.48930684,0.44821364],"study_design_scores_gemma":[0.00013039625,0.00020976088,0.041115377,0.00045848425,0.00008157677,0.0025272656,0.0024872893,0.024165021,0.03960425,0.0029501948,0.88614553,0.00012477777],"about_ca_topic_score_codex":0.0073521296,"about_ca_topic_score_gemma":0.017526759,"teacher_disagreement_score":0.008566272,"about_ca_system_score_codex":0.00092796126,"about_ca_system_score_gemma":0.0009354121,"threshold_uncertainty_score":0.02865702},"labels":[],"label_agreement":null},{"id":"W4310413652","doi":"10.48550/arxiv.2211.14402","title":"An Analysis of Social Biases Present in BERT Variants Across Multiple Languages","year":2022,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Natural language processing; Persian; Artificial intelligence; Set (abstract data type); Linguistics; Adjective; Noun","score_opus":0.0646907881846757,"score_gpt":0.27292015475053366,"score_spread":0.20822936656585794,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4310413652","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9845575,0.00010270163,0.011850494,0.00016026806,0.000013850487,0.000020910362,0.0002942652,0.00020224878,0.0027977647],"genre_scores_gemma":[0.995139,0.000041173804,0.003678157,0.000025884297,0.000005812689,0.0000154684,0.00045804112,0.000077027245,0.00055946485],"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99869627,0.00069121964,0.00007796865,0.00018156646,0.0002631239,0.000089903006],"domain_scores_gemma":[0.9900563,0.0064132023,0.00069146365,0.0012729599,0.0013420853,0.00022408368],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032782946,0.00046958486,0.00032019996,0.00089205493,0.00057686184,0.0011538882,0.00050094695,0.0004119818,0.0011817485],"category_scores_gemma":[0.013154397,0.00026693635,0.0004047212,0.0009751391,0.00084041513,0.0016860892,0.001132993,0.00067449163,0.00033720522],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011313861,0.0002541763,0.645119,0.00035568746,0.00055395724,0.0017434488,0.015772972,0.12477753,0.03916134,0.020210626,0.004520679,0.14639927],"study_design_scores_gemma":[0.000049613842,0.00039482696,0.18569605,0.00008241462,0.0002309738,0.0013939005,0.0077182674,0.74286693,0.026977565,0.024227517,0.010182242,0.00017974488],"about_ca_topic_score_codex":0.0063221273,"about_ca_topic_score_gemma":0.012301355,"teacher_disagreement_score":0.0063221273,"about_ca_system_score_codex":0.0007637274,"about_ca_system_score_gemma":0.0004981732,"threshold_uncertainty_score":0.017337501},"labels":[],"label_agreement":null},{"id":"W4310892181","doi":"10.14705/rpnet.2022.61.1458","title":"Evaluating automatic speech recognition for L2 pronunciation feedback: a focus on Google Translate","year":2022,"lang":"en","type":"book-chapter","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Concordia University; Université du Québec à Trois-Rivières","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Pronunciation; Transcription (linguistics); Speech recognition; Focus (optics); Computer science; Phonetic transcription; Natural language processing; Word (group theory); Artificial intelligence; Linguistics","score_opus":0.06380145554938882,"score_gpt":0.33311084750550557,"score_spread":0.26930939195611675,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4310892181","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.95298135,0.0023792628,0.013323501,0.00041677852,0.0001240681,0.00048001736,0.004423736,0.00465987,0.021211378],"genre_scores_gemma":[0.9303504,0.0010497114,0.025642691,0.00032522093,0.0000859819,0.00034879567,0.015414257,0.0010711458,0.02571176],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99596417,0.0019134894,0.00013866466,0.00035715615,0.0014323108,0.00019410596],"domain_scores_gemma":[0.9884737,0.0066935276,0.00028445525,0.0006147773,0.003599617,0.00033395138],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003402166,0.0011290804,0.00063576235,0.0010892584,0.0005568627,0.0012171582,0.00093566877,0.0007504431,0.0050233062],"category_scores_gemma":[0.009184733,0.00019151912,0.0002562069,0.0013788397,0.00046379663,0.000862427,0.0007127778,0.00043356908,0.0041258354],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0023864894,0.00063566683,0.036689527,0.0019112892,0.00014232057,0.0009097626,0.008492476,0.005216936,0.13224931,0.00050568255,0.030285463,0.78057504],"study_design_scores_gemma":[0.0005753104,0.007819387,0.49580935,0.00049166736,0.0004833122,0.0038558943,0.01680376,0.07149736,0.27616575,0.00059967214,0.12537184,0.0005266903],"about_ca_topic_score_codex":0.077465005,"about_ca_topic_score_gemma":0.140771,"teacher_disagreement_score":0.077465005,"about_ca_system_score_codex":0.0017729367,"about_ca_system_score_gemma":0.0013578269,"threshold_uncertainty_score":0.15402818},"labels":[],"label_agreement":null},{"id":"W4310974014","doi":"10.31234/osf.io/tjah9","title":"Subjective equivalence—a basic requirement for strict framing effects: Commentary on Huizenga et al. (2023)","year":2022,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University; Defence Research and Development Canada","funders":"","keywords":"Framing (construction); Equivalence (formal languages); Logical equivalence; Framing effect; Epistemology; Frame problem; Psychology; Mathematical economics; Social psychology; Computer science; Mathematics; Linguistics; Discrete mathematics; Philosophy; Artificial intelligence; History","score_opus":0.02618538921481071,"score_gpt":0.3311226748552362,"score_spread":0.3049372856404255,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4310974014","genre_codex":"commentary","genre_gemma":"commentary","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"commentary","genre_consensus":"commentary","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00037228726,0.010375547,0.0017799977,0.9596322,0.024355935,0.000021609096,0.0001375092,0.000050754243,0.0032739774],"genre_scores_gemma":[0.0163299,0.0060361163,0.0033554868,0.9346238,0.03594972,0.00023053797,0.000070892376,0.00009280616,0.0033107707],"study_design_codex":"not_applicable","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9816175,0.0066689244,0.0016721083,0.004132927,0.004994241,0.00091427757],"domain_scores_gemma":[0.84675705,0.12486923,0.002940605,0.0043184357,0.019277621,0.0018369809],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.041975446,0.0014147217,0.0025714661,0.0021672475,0.0064540687,0.0053054797,0.0126353325,0.031916145,0.004832725],"category_scores_gemma":[0.11106299,0.0008903794,0.0020648686,0.0022632913,0.025975835,0.013292119,0.005299343,0.057503268,0.00494303],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017067473,0.000030919335,0.00018307932,0.00046302754,0.00004997965,0.00025180564,0.003568439,0.00016088318,0.00022334309,0.11198232,0.86418897,0.018726455],"study_design_scores_gemma":[0.0001795766,0.000058249134,0.0016605183,0.0020273228,0.00010480375,0.00041613242,0.0023418714,0.0007101038,0.00084710785,0.24703923,0.74439865,0.00021655827],"about_ca_topic_score_codex":0.027761523,"about_ca_topic_score_gemma":0.026405545,"teacher_disagreement_score":0.041975446,"about_ca_system_score_codex":0.008201021,"about_ca_system_score_gemma":0.0069931317,"threshold_uncertainty_score":0.22199005},"labels":[],"label_agreement":null},{"id":"W4311428537","doi":"10.3384/nejlt.2000-1533.2022.4315","title":"Part-of-Speech and Morphological Tagging of Algerian Judeo-Arabic","year":2022,"lang":"en","type":"article","venue":"Northern European Journal of Language Technology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Azrieli Foundation; Israel Science Foundation","keywords":"Computer science; Natural language processing; Annotation; Arabic; Preprocessor; Pipeline (software); Artificial intelligence; Part of speech; Linguistics","score_opus":0.010133932396906404,"score_gpt":0.23726246352599364,"score_spread":0.22712853112908724,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4311428537","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.951947,0.0008960935,0.026737956,0.00038590963,0.00021868387,0.00012472286,0.0061394116,0.0014614221,0.012088937],"genre_scores_gemma":[0.90054953,0.00048680865,0.07479942,0.0001019851,0.00010279026,0.00014423576,0.014365961,0.00035931516,0.00909005],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9994821,0.00015174753,0.000062828396,0.00016664427,0.00010098641,0.000035757697],"domain_scores_gemma":[0.9972621,0.0012015442,0.00029127634,0.0004148098,0.0007452202,0.00008502127],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006144456,0.00038897927,0.00020173068,0.0026105125,0.0011415058,0.0007654896,0.0004077101,0.00049357984,0.0033584964],"category_scores_gemma":[0.0025351332,0.00019230414,0.000189204,0.0017045666,0.00052318914,0.00082727947,0.0005433368,0.00035952474,0.0022313676],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011929076,0.00032035023,0.05989053,0.001373649,0.00009209669,0.0037612293,0.014223553,0.008956791,0.29579237,0.0062351357,0.01867493,0.5894864],"study_design_scores_gemma":[0.00008433074,0.00026773722,0.39323506,0.0002942482,0.00014858009,0.0041103596,0.011457548,0.11674066,0.28890082,0.005413044,0.17913564,0.00021198511],"about_ca_topic_score_codex":0.004796105,"about_ca_topic_score_gemma":0.010171592,"teacher_disagreement_score":0.004796105,"about_ca_system_score_codex":0.00045128312,"about_ca_system_score_gemma":0.00050798553,"threshold_uncertainty_score":0.011235237},"labels":[],"label_agreement":null},{"id":"W4311991579","doi":"10.48550/arxiv.2212.08132","title":"WEKA-Based: Key Features and Classifier for French of Five Countries","year":2022,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Python (programming language); Sketch; Classifier (UML); Key (lock); Artificial intelligence; Computer science; Natural language processing","score_opus":0.035928748567932794,"score_gpt":0.20816817346162816,"score_spread":0.17223942489369537,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4311991579","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.098097876,0.002664338,0.6881666,0.000869892,0.0005675409,0.0016330187,0.058773547,0.13618702,0.013040148],"genre_scores_gemma":[0.30220965,0.00087360776,0.5730897,0.00031436177,0.00009541028,0.002513205,0.10258507,0.002928284,0.015390762],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990356,0.00013935665,0.000113477705,0.00030502834,0.00023040912,0.0001760921],"domain_scores_gemma":[0.9990232,0.00038999654,0.000059494796,0.0001566421,0.0003337475,0.00003693795],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008434774,0.0016567111,0.0006815244,0.0038597428,0.0010580352,0.001483707,0.0012204208,0.0010593034,0.010197208],"category_scores_gemma":[0.005293256,0.0004324853,0.0017916996,0.00231014,0.00036398892,0.0015270807,0.00093962596,0.0012622067,0.008681414],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00057386566,0.00017491118,0.010242933,0.0007427372,0.00029263686,0.0006559945,0.00043251025,0.018186247,0.026955463,0.0040571787,0.102042995,0.8356425],"study_design_scores_gemma":[0.000243145,0.00037379903,0.037095938,0.00034914515,0.00057506195,0.0017823715,0.0013467929,0.49912706,0.106915504,0.015285293,0.33656594,0.00034000727],"about_ca_topic_score_codex":0.03838758,"about_ca_topic_score_gemma":0.03771161,"teacher_disagreement_score":0.03838758,"about_ca_system_score_codex":0.0012638416,"about_ca_system_score_gemma":0.0017074578,"threshold_uncertainty_score":0.07632834},"labels":[],"label_agreement":null},{"id":"W4312215389","doi":"10.5334/johd.94","title":"The TRANSCOMP Dataset of Literary Translations from 120 Languages and a Parallel Collection of English-language Originals","year":2022,"lang":"en","type":"article","venue":"Journal of Open Humanities Data","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Linguistics; Computer science; Word (group theory); Metadata; Literary translation; Natural language processing; Literature; History; Art; World Wide Web; Philosophy","score_opus":0.05700619531125964,"score_gpt":0.3531706120189455,"score_spread":0.2961644167076859,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4312215389","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01692666,0.0006954038,0.0012579839,0.00026831587,0.00030630655,0.00018185312,0.9729258,0.0012944535,0.0061432268],"genre_scores_gemma":[0.0042513204,0.00011949383,0.0017604561,0.00004961857,0.000054101736,0.00022814053,0.99156135,0.00014753577,0.0018280803],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.998027,0.0004092733,0.00034514855,0.00039999958,0.0006134851,0.00020507594],"domain_scores_gemma":[0.9951062,0.0013380281,0.00035271898,0.0012033127,0.0015288254,0.0004709347],"candidate_categories":["open_science"],"consensus_categories":[],"category_scores_codex":[0.0014093602,0.0015498238,0.001075152,0.009182373,0.0014801543,0.0021518185,0.0019239032,0.0016331566,0.025257546],"category_scores_gemma":[0.0070095584,0.00040697772,0.0010374532,0.010157584,0.0010114526,0.0018378086,0.0032282649,0.0015729871,0.03728485],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00046369262,0.00038794233,0.005401305,0.0027172912,0.000089235706,0.0009803771,0.0011020616,0.0006481677,0.0042311256,0.0023725692,0.9373074,0.044298757],"study_design_scores_gemma":[0.00027945824,0.00011904451,0.022420362,0.00047261373,0.000050434846,0.0010682032,0.0018398586,0.0012739473,0.0034415692,0.001405845,0.967553,0.00007570979],"about_ca_topic_score_codex":0.005032044,"about_ca_topic_score_gemma":0.013128739,"teacher_disagreement_score":0.9980761,"about_ca_system_score_codex":0.0008320847,"about_ca_system_score_gemma":0.0017822395,"threshold_uncertainty_score":0.08449495},"labels":[],"label_agreement":null},{"id":"W4312301825","doi":"10.1109/saci55618.2022.9919604","title":"A Deep Learning Method for Sentence Embeddings Based on Hadamard Matrix Encodings","year":2022,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Artificial intelligence; Sentence; Parsing; Embedding; Word embedding; Dependency grammar; Word (group theory); Natural language processing; Deep learning; String (physics); Dependency (UML); Artificial neural network; Convolutional neural network; Mathematics","score_opus":0.011807292276903771,"score_gpt":0.31441266645296306,"score_spread":0.3026053741760593,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4312301825","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0049880752,0.00026476942,0.9898913,0.0001974722,0.00015991452,0.000062373445,0.00054109754,0.003028112,0.00086690905],"genre_scores_gemma":[0.135676,0.00044904067,0.8501567,0.00023790149,0.00019666662,0.00028549484,0.004214754,0.00059347495,0.008189891],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9993594,0.0001734242,0.00006788248,0.00017865951,0.00016965193,0.00005088922],"domain_scores_gemma":[0.9990828,0.0003634845,0.000077774064,0.00015613495,0.0002711414,0.00004861674],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008455713,0.0011876763,0.000637869,0.0013444922,0.00039454777,0.0010165011,0.0013140943,0.0010076612,0.006506596],"category_scores_gemma":[0.0034670725,0.0005315684,0.00093266706,0.0012264521,0.00047896226,0.0032230907,0.0013133222,0.0024383396,0.0029018924],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025832723,0.00016708838,0.00075719133,0.00028236702,0.0001275873,0.00022604037,0.00021291347,0.06444659,0.020045528,0.048552856,0.022325434,0.84259814],"study_design_scores_gemma":[0.000026377962,0.000104295024,0.0002655681,0.000034313332,0.000028141414,0.00013758386,0.00004354427,0.9421239,0.010324642,0.03746407,0.009417285,0.000030277426],"about_ca_topic_score_codex":0.002638785,"about_ca_topic_score_gemma":0.00480304,"teacher_disagreement_score":0.006506596,"about_ca_system_score_codex":0.0006926849,"about_ca_system_score_gemma":0.00096179056,"threshold_uncertainty_score":0.021766722},"labels":[],"label_agreement":null},{"id":"W4312353038","doi":"10.1109/ijcnn55064.2022.9892285","title":"Named Entity Recognition for Audio De-Identification","year":2022,"lang":"en","type":"article","venue":"2022 International Joint Conference on Neural Networks (IJCNN)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Mitacs","keywords":"Pipeline (software); Computer science; Task (project management); Named-entity recognition; Speech recognition; Identification (biology); Natural language processing; Audio mining; Artificial intelligence; Acoustic model; Speech processing; Programming language","score_opus":0.04035002746334673,"score_gpt":0.2873490868881336,"score_spread":0.24699905942478687,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4312353038","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015049714,0.002953917,0.93327844,0.0008051882,0.0005163353,0.00038266083,0.010386644,0.028950488,0.007676499],"genre_scores_gemma":[0.15797172,0.001833861,0.7848202,0.00039553485,0.00024527838,0.00047757864,0.043100607,0.00078707276,0.01036817],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99632865,0.0010479076,0.00034168424,0.0012973404,0.0007811256,0.00020328182],"domain_scores_gemma":[0.99482226,0.001654947,0.0004761971,0.0018253762,0.0011202246,0.00010101125],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030179578,0.0013774879,0.00091223686,0.0033073416,0.0011608043,0.0015821952,0.0017070733,0.0013181217,0.00810804],"category_scores_gemma":[0.008510816,0.00038470703,0.0011688381,0.0027092823,0.00056693354,0.0042371685,0.0019614617,0.001483993,0.009284289],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00039439075,0.00014137717,0.0026642052,0.0007968777,0.0001695832,0.00047930476,0.00039837105,0.008874414,0.053236637,0.010850634,0.04746371,0.87453043],"study_design_scores_gemma":[0.00007511585,0.00027863236,0.012390841,0.00033580486,0.0002920815,0.0022228444,0.0010535693,0.35880524,0.21314885,0.052829813,0.3582981,0.00026918025],"about_ca_topic_score_codex":0.0043763407,"about_ca_topic_score_gemma":0.0057663913,"teacher_disagreement_score":0.00810804,"about_ca_system_score_codex":0.0007884038,"about_ca_system_score_gemma":0.0011242519,"threshold_uncertainty_score":0.027124107},"labels":[],"label_agreement":null},{"id":"W4312528406","doi":"10.5220/0011524400003335","title":"Towards a Unified Multilingual Ontology for Rhetorical Figures","year":2022,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Rhetorical question; Ontology; Natural language processing; Linguistics; World Wide Web; Information retrieval; Artificial intelligence; Epistemology; Philosophy","score_opus":0.03405988176905151,"score_gpt":0.330774394569948,"score_spread":0.2967145128008965,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4312528406","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0046745436,0.00022992853,0.98110056,0.00095131603,0.00023293856,0.00019129325,0.001323252,0.002698981,0.008597282],"genre_scores_gemma":[0.08486124,0.00065004156,0.897895,0.00057985686,0.00022530876,0.00035745092,0.005007036,0.0021675006,0.008256587],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9944126,0.0013956453,0.0011770728,0.0012081593,0.0013762971,0.00043025505],"domain_scores_gemma":[0.9939095,0.0009797481,0.00039175706,0.0015763558,0.0025883536,0.0005542967],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004537137,0.000898589,0.0014637078,0.0076180585,0.003363905,0.009774668,0.0031623063,0.002252576,0.0064771487],"category_scores_gemma":[0.008365322,0.0017042948,0.0031439685,0.004943691,0.0026971353,0.01749197,0.008620078,0.005589286,0.0050352653],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006654372,0.00018370926,0.001476384,0.00045460288,0.0000993688,0.00039459346,0.002868074,0.0035684519,0.005692484,0.8475288,0.014660598,0.12300632],"study_design_scores_gemma":[0.000043041542,0.00006294992,0.0009871182,0.00053272117,0.00032936217,0.00066410075,0.0029310987,0.06153144,0.009329349,0.47428185,0.4491608,0.00014613783],"about_ca_topic_score_codex":0.012350505,"about_ca_topic_score_gemma":0.018053139,"teacher_disagreement_score":0.012350505,"about_ca_system_score_codex":0.0032097187,"about_ca_system_score_gemma":0.007809032,"threshold_uncertainty_score":0.024557233},"labels":[],"label_agreement":null},{"id":"W4312738848","doi":"10.4000/books.aaccademia.11009","title":"Tackling Italian University Assessment Tests with Transformer-Based Language Models","year":2022,"lang":"en","type":"book-chapter","venue":"Accademia University Press eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Canadian Institute for Advanced Research","keywords":"Transformer; Computer science; Security token; Natural language processing; Artificial intelligence; Task (project management); Cloze test; Test (biology); Mathematics education; Reading (process); Reading comprehension; Linguistics; Psychology; Engineering; Computer security; Systems engineering","score_opus":0.018302011155825897,"score_gpt":0.2352954042576568,"score_spread":0.2169933931018309,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4312738848","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10708567,0.0007145416,0.8714277,0.00040764984,0.00008743994,0.00015711102,0.00049295014,0.009366491,0.010260457],"genre_scores_gemma":[0.6480911,0.00053453184,0.33524108,0.00019971766,0.0000688398,0.00018698612,0.0025156874,0.0005978787,0.012564219],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99937457,0.00025484213,0.00003502675,0.00018238471,0.00010099769,0.000052119718],"domain_scores_gemma":[0.9984762,0.0011395311,0.00007361676,0.0001139417,0.00014925895,0.00004748814],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010598217,0.0012819769,0.00055304554,0.00072794425,0.0002573836,0.0016303695,0.0010340746,0.0008113203,0.003216284],"category_scores_gemma":[0.0037772271,0.00039961975,0.0007367967,0.00058492384,0.00036360067,0.0014411192,0.0010477159,0.001732647,0.0023102842],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022734511,0.00017185097,0.0042152186,0.00017806458,0.00008288902,0.00033738505,0.0003853771,0.20738271,0.009735116,0.00806857,0.007706812,0.76150864],"study_design_scores_gemma":[0.0000121068915,0.000070903414,0.0010450254,0.0000151063905,0.000023537783,0.000118839096,0.00011507377,0.9843976,0.0045857667,0.0071488833,0.002449615,0.000017561997],"about_ca_topic_score_codex":0.01116878,"about_ca_topic_score_gemma":0.013589353,"teacher_disagreement_score":0.01116878,"about_ca_system_score_codex":0.0009567327,"about_ca_system_score_gemma":0.0013926082,"threshold_uncertainty_score":0.022207558},"labels":[],"label_agreement":null},{"id":"W4312790626","doi":"10.2139/ssrn.4303223","title":"Failures of Lemma Access: Do Synonym and Subsumative Speech Errors Exist?","year":2022,"lang":"en","type":"article","venue":"SSRN Electronic Journal","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Synonym (taxonomy); Lemma (botany); Computer science; Mathematics; Speech recognition; Linguistics; Natural language processing; Philosophy; Biology","score_opus":0.011395622991343906,"score_gpt":0.280521255885531,"score_spread":0.2691256328941871,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4312790626","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9353763,0.00074409053,0.032422952,0.002986856,0.00028434535,0.000088065586,0.0016032936,0.0013833249,0.025110772],"genre_scores_gemma":[0.99538606,0.00012917121,0.0027307693,0.00017706079,0.000091958114,0.000020656735,0.00039467422,0.00024334672,0.00082623505],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99302214,0.0021278043,0.0011029163,0.0018266385,0.001155976,0.00076453306],"domain_scores_gemma":[0.9065543,0.060573176,0.009914167,0.016460778,0.0049425256,0.0015549728],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0068924245,0.00082254276,0.001394644,0.0024851405,0.0013959908,0.004264143,0.0018697467,0.003747658,0.016888926],"category_scores_gemma":[0.079569794,0.0010250815,0.00051938137,0.00188513,0.0032468939,0.012070656,0.0035976993,0.002248858,0.005518923],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0057894136,0.00086747215,0.51414096,0.00119028,0.00037553968,0.021969175,0.04092265,0.0020809036,0.040764913,0.08304498,0.01227544,0.27657822],"study_design_scores_gemma":[0.0005672103,0.0008154528,0.39821956,0.0006664417,0.0007062009,0.06761533,0.04780384,0.04612586,0.068652324,0.32762268,0.04069086,0.0005142775],"about_ca_topic_score_codex":0.0010104413,"about_ca_topic_score_gemma":0.00071175484,"teacher_disagreement_score":0.016888926,"about_ca_system_score_codex":0.00047678873,"about_ca_system_score_gemma":0.00095644017,"threshold_uncertainty_score":0.056499064},"labels":[],"label_agreement":null},{"id":"W4313230118","doi":"10.1007/978-1-4842-8829-0_12","title":"Working with Symbol Warping Tools","year":2022,"lang":"en","type":"book-chapter","venue":"Apress eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Delta-Q Technologies (Canada)","funders":"","keywords":"Image warping; Symbol (formal); Computer science; Arithmetic; Mathematics; Artificial intelligence; Programming language","score_opus":0.042860588142739096,"score_gpt":0.24056359322480647,"score_spread":0.19770300508206737,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4313230118","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004154817,0.009942252,0.73174495,0.0022948885,0.0017351629,0.00011084119,0.00039364953,0.012025985,0.23759744],"genre_scores_gemma":[0.04846212,0.017630044,0.5350023,0.0018206189,0.00077363424,0.00018902466,0.0014678448,0.007821007,0.38683346],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99949193,0.000076922246,0.00002622847,0.00013511533,0.00022565863,0.00004416587],"domain_scores_gemma":[0.99915457,0.0004061863,0.00003022324,0.00019578863,0.00017055128,0.000042635664],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005555583,0.0014310866,0.0005406208,0.001229564,0.0009897382,0.004360641,0.001363598,0.0011529871,0.056157213],"category_scores_gemma":[0.002803884,0.0006094595,0.00074907485,0.0016439118,0.001399857,0.010322997,0.002090532,0.0033468618,0.03554094],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000046832036,0.000067702524,0.0001867169,0.0005772865,0.000031497486,0.00021378923,0.0012808986,0.0015563357,0.0147554865,0.1707041,0.10016156,0.71041775],"study_design_scores_gemma":[0.000009465492,0.000054879984,0.00013830511,0.0002274005,0.000029018956,0.0007646126,0.0003522066,0.0034899062,0.020797372,0.104116976,0.86998415,0.000035616325],"about_ca_topic_score_codex":0.00064153044,"about_ca_topic_score_gemma":0.0009961488,"teacher_disagreement_score":0.056157213,"about_ca_system_score_codex":0.00054583565,"about_ca_system_score_gemma":0.0006644961,"threshold_uncertainty_score":0.1878646},"labels":[],"label_agreement":null},{"id":"W4313445006","doi":"10.1007/978-3-031-04394-9_47","title":"Manual Transcription","year":2023,"lang":"en","type":"book-chapter","venue":"Springer texts in education","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Saskatchewan","funders":"","keywords":"Transcription (linguistics); Computer science; Data quality; Data science; Linguistics; Engineering; Philosophy; Operations management","score_opus":0.017097987988141175,"score_gpt":0.2896870523222362,"score_spread":0.272589064334095,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4313445006","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0047268714,0.0007129597,0.110993125,0.0016515236,0.0048845075,0.0037725635,0.27504274,0.058320228,0.5398955],"genre_scores_gemma":[0.013904208,0.00073472946,0.1411601,0.0011863345,0.0008564745,0.0043787933,0.38554922,0.022245057,0.42998517],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99731004,0.00050698797,0.00034286224,0.0007031826,0.0009304713,0.00020647956],"domain_scores_gemma":[0.98988605,0.0021108896,0.00022040427,0.003295404,0.0042442568,0.00024298865],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0019313747,0.0025791703,0.00157362,0.006735553,0.0022531385,0.0028512583,0.0021487654,0.0012221013,0.68058306],"category_scores_gemma":[0.011577341,0.0009960083,0.0012425635,0.004316481,0.00083599397,0.002018009,0.0031090025,0.0022136907,0.6009845],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013403776,0.000057817735,0.00021243085,0.0006552823,0.000012044729,0.00016284952,0.00033509775,0.00012554775,0.0062189987,0.0035329645,0.83175176,0.15680118],"study_design_scores_gemma":[0.00004957598,0.000034585853,0.000753566,0.00015992795,0.000018225326,0.00028842484,0.00036213102,0.0003909708,0.005416894,0.002924255,0.98956674,0.000034620505],"about_ca_topic_score_codex":0.0030794689,"about_ca_topic_score_gemma":0.004770821,"teacher_disagreement_score":0.68058306,"about_ca_system_score_codex":0.0011147987,"about_ca_system_score_gemma":0.0034878848,"threshold_uncertainty_score":0.4556095},"labels":[],"label_agreement":null},{"id":"W4315433003","doi":"10.16995/dscn.8084","title":"Title Pending 8084","year":2023,"lang":"fr","type":"article","venue":"Digital Studies / Le champ numérique","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"NoSQL; Computer science; JavaScript; Agile software development; World Wide Web; Set (abstract data type); Citation; Software engineering; Database; Programming language; Scalability","score_opus":0.05747828914191096,"score_gpt":0.3198880235993798,"score_spread":0.26240973445746885,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4315433003","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00039519582,0.00090020645,0.000701265,0.0020287414,0.007630727,0.0002787775,0.0038204407,0.00056740316,0.98367727],"genre_scores_gemma":[0.0007445468,0.00037223185,0.00014983107,0.00032951226,0.00046769105,0.000040397237,0.0013086852,0.00022189428,0.9963652],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99864334,0.00010665687,0.00013397775,0.00033536277,0.00062795094,0.00015275837],"domain_scores_gemma":[0.9952968,0.0007068986,0.00017830981,0.00045131685,0.0024723392,0.000894408],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.0014997314,0.0010639273,0.001449039,0.002383814,0.002647574,0.00927273,0.0016063375,0.0037987642,0.939376],"category_scores_gemma":[0.005810281,0.0007122013,0.0008233433,0.0028552252,0.00091169286,0.0032549547,0.0027578296,0.0020547933,0.9082262],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010747844,0.000072932824,0.00022224755,0.00039060472,0.0000098375485,0.00013553097,0.00005287775,0.000050102335,0.0016715559,0.0062164026,0.8595572,0.1315132],"study_design_scores_gemma":[0.000010320848,0.000024531575,0.00023804624,0.00007683387,0.0000025152044,0.000051581508,0.00002610735,0.000029962843,0.00013668556,0.00029305104,0.9991055,0.0000048798097],"about_ca_topic_score_codex":0.002341424,"about_ca_topic_score_gemma":0.0042673307,"teacher_disagreement_score":0.060624003,"about_ca_system_score_codex":0.0015912888,"about_ca_system_score_gemma":0.0026149012,"threshold_uncertainty_score":0.08647269},"labels":[],"label_agreement":null},{"id":"W4315866265","doi":"10.5430/wjel.v13n2p23","title":"Data-Driven Learning Tasks and Involvement Load Hypothesis","year":2023,"lang":"en","type":"article","venue":"World Journal of English Language","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Vocabulary; Session (web analytics); Reading (process); Recall; Vocabulary learning; Test (biology); Cognition; Psychology; Significant difference; Computer science; Meaning (existential); Cognitive psychology; Mathematics education; Linguistics","score_opus":0.026842372409745796,"score_gpt":0.2777499247030637,"score_spread":0.2509075522933179,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4315866265","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.90134376,0.00023273141,0.07355399,0.0013165644,0.000057271907,0.001188648,0.00041950465,0.00022676334,0.021660754],"genre_scores_gemma":[0.9824808,0.00006648189,0.014430345,0.00024237492,0.000051879353,0.0011580917,0.00032298957,0.0000595787,0.0011873173],"study_design_codex":"observational","study_design_gemma":"not_applicable","domain_scores_codex":[0.984885,0.00638206,0.0013538899,0.0023469164,0.004357989,0.0006740502],"domain_scores_gemma":[0.77307004,0.18792814,0.014043697,0.015251399,0.0058759386,0.003830674],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.021446932,0.0011806947,0.0010918785,0.001908384,0.0010826447,0.0036235203,0.0027893241,0.0017040719,0.0061918986],"category_scores_gemma":[0.10612961,0.00064860564,0.0009401951,0.00093485875,0.004717026,0.0075953817,0.0056223203,0.0017118035,0.00094446517],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0145180095,0.021394597,0.51483524,0.0025010193,0.0005278582,0.0011356008,0.023726068,0.015133128,0.032042775,0.09030071,0.002184614,0.28170037],"study_design_scores_gemma":[0.0036274856,0.012960567,0.4472631,0.0005837625,0.00047046994,0.0019384755,0.007888181,0.1548619,0.046339627,0.3107351,0.012938301,0.0003930665],"about_ca_topic_score_codex":0.00060627994,"about_ca_topic_score_gemma":0.00025559598,"teacher_disagreement_score":0.021446932,"about_ca_system_score_codex":0.001328653,"about_ca_system_score_gemma":0.0010540561,"threshold_uncertainty_score":0.11342353},"labels":[],"label_agreement":null},{"id":"W4316673150","doi":"10.18280/ria.360610","title":"DZ-OPINION: Algerian Dialect Opinion Analysis Model with Deep Learning Techniques","year":2022,"lang":"en","type":"article","venue":"Revue d intelligence artificielle","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Sentiment analysis; Reputation; Public opinion; Computer science; Arabic; Product (mathematics); Process (computing); Data science; Opinion leadership; Social network analysis; Artificial intelligence; World Wide Web; Social media; Political science; Public relations; Linguistics","score_opus":0.024789487260832976,"score_gpt":0.2835657569791419,"score_spread":0.2587762697183089,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4316673150","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.40678033,0.0046851817,0.5302996,0.0051449346,0.001427693,0.00031059186,0.0056623165,0.006242791,0.039446652],"genre_scores_gemma":[0.89925724,0.00069934665,0.07216136,0.0010864799,0.0003276815,0.00017356312,0.004677576,0.00014774723,0.021469038],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99984145,0.000035543027,0.000009921848,0.000052280502,0.00002524636,0.00003552517],"domain_scores_gemma":[0.99982363,0.000037620386,0.000013186541,0.000008539991,0.00010697424,0.000010020108],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00039870024,0.0007059416,0.00035185984,0.00048145532,0.0003134114,0.000761533,0.000807828,0.0006532634,0.0035183076],"category_scores_gemma":[0.0007134237,0.0001390419,0.0005581718,0.0002682835,0.00015744647,0.00063087477,0.00041867554,0.00092492433,0.0014288846],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001082165,0.00043826515,0.015635446,0.00031075042,0.00047581215,0.000602433,0.0005280828,0.13917829,0.017376503,0.008551973,0.065634064,0.7501861],"study_design_scores_gemma":[0.000024186069,0.00005678187,0.0014302147,0.000019156654,0.000055816745,0.00005700166,0.00007960168,0.98860174,0.0029855217,0.0022371819,0.004441026,0.000011698286],"about_ca_topic_score_codex":0.013610584,"about_ca_topic_score_gemma":0.016949369,"teacher_disagreement_score":0.013610584,"about_ca_system_score_codex":0.0007024673,"about_ca_system_score_gemma":0.0005236436,"threshold_uncertainty_score":0.027062714},"labels":[],"label_agreement":null},{"id":"W4317039284","doi":"10.22541/essoar.167397464.40579434/v1","title":"SKB Task Force GWFTS: Pragmatic Validation Using Predictive Modeling Exercises","year":2023,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Korea Atomic Energy Research Institute; Nuclear Waste Management Organization; U.S. Department of Energy","keywords":"Task (project management); Computer science; Task force; Engineering; Systems engineering; Political science","score_opus":0.044101297656532486,"score_gpt":0.318591841068891,"score_spread":0.2744905434123585,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4317039284","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016078804,0.00045293456,0.850304,0.003740329,0.0013373068,0.0014006839,0.024222504,0.048007555,0.05445596],"genre_scores_gemma":[0.19954206,0.00048580352,0.6577372,0.001171225,0.0005157507,0.0029808355,0.0792226,0.020851051,0.03749347],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9891116,0.0072787283,0.0005070917,0.0011195185,0.0016677033,0.00031524408],"domain_scores_gemma":[0.9624491,0.025883613,0.0004294147,0.006641907,0.004104131,0.00049188966],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013187244,0.002455267,0.0014433978,0.0016421238,0.0017343417,0.0036136794,0.0031042253,0.0025312111,0.07300988],"category_scores_gemma":[0.06633308,0.0013057366,0.0020596432,0.0010789349,0.0013583817,0.0055072918,0.0058833137,0.0049608187,0.03022499],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009908138,0.00073893595,0.0013022443,0.001768399,0.00030496228,0.0003487638,0.0012772442,0.03175747,0.006914829,0.07174338,0.44417843,0.4386745],"study_design_scores_gemma":[0.0010379746,0.00025930614,0.0014585373,0.0007915184,0.00023511227,0.00027263022,0.00086852326,0.5410591,0.023285849,0.16422033,0.2663071,0.00020411583],"about_ca_topic_score_codex":0.0055830353,"about_ca_topic_score_gemma":0.0057145627,"teacher_disagreement_score":0.07300988,"about_ca_system_score_codex":0.0010684778,"about_ca_system_score_gemma":0.0033661711,"threshold_uncertainty_score":0.24424237},"labels":[],"label_agreement":null},{"id":"W4317504322","doi":"10.1145/3580495","title":"Filtering and Extended Vocabulary based Translation for Low-resource Language Pair of Sanskrit-Hindi","year":2023,"lang":"en","type":"article","venue":"ACM Transactions on Asian and Low-Resource Language Information Processing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Sanskrit; Computer science; Machine translation; Hindi; Natural language processing; Artificial intelligence; Vocabulary; Transformer; Sentence; Phrase; Language translation; Linguistics; Engineering","score_opus":0.00992098405943372,"score_gpt":0.2561712473687812,"score_spread":0.2462502633093475,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4317504322","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13471554,0.00076250505,0.85515475,0.00017616802,0.00019675975,0.00015978981,0.00033328286,0.003235278,0.005265936],"genre_scores_gemma":[0.5433568,0.00048707615,0.44112548,0.00017924933,0.00008660204,0.0002136454,0.0026110525,0.0003487521,0.011591369],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9996331,0.00008793104,0.00002889885,0.0001351415,0.00007468477,0.00004012702],"domain_scores_gemma":[0.99960643,0.00013711999,0.000029016497,0.00007172816,0.00013702732,0.000018675397],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00047447512,0.0006355652,0.0005959225,0.0006878676,0.0005542416,0.00058344775,0.00053598307,0.0004566738,0.0028818673],"category_scores_gemma":[0.0012418122,0.00018869757,0.0006970569,0.00065228634,0.0003485652,0.000979682,0.0006324204,0.00060094835,0.0016646802],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00039949222,0.00022990865,0.0017171397,0.00040571133,0.00008049883,0.0006683163,0.00054531934,0.026735995,0.17400932,0.008978787,0.005261214,0.78096825],"study_design_scores_gemma":[0.000061377476,0.00066350255,0.003380689,0.000039254246,0.00012836793,0.0012979363,0.00035361652,0.81358033,0.15226378,0.008518704,0.019653529,0.000058916583],"about_ca_topic_score_codex":0.0041163554,"about_ca_topic_score_gemma":0.007836582,"teacher_disagreement_score":0.0041163554,"about_ca_system_score_codex":0.00032270345,"about_ca_system_score_gemma":0.00084529613,"threshold_uncertainty_score":0.009640753},"labels":[],"label_agreement":null},{"id":"W4317517908","doi":"10.24197/her.24.2022.1-15","title":"La traduction automatique neuronale : un problème de temporalité","year":2023,"lang":"es","type":"article","venue":"Hermeneus","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Philosophy","score_opus":0.014030232791307625,"score_gpt":0.2786719093285806,"score_spread":0.264641676537273,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4317517908","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04266446,0.012943191,0.75796,0.017660774,0.0013601789,0.00011723256,0.0012460712,0.0017872908,0.16426069],"genre_scores_gemma":[0.6625028,0.009198711,0.20393933,0.0031209174,0.00081713736,0.00039463022,0.0015818378,0.0013188772,0.11712568],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99723035,0.0006115251,0.000253759,0.0010179336,0.0007130449,0.00017337997],"domain_scores_gemma":[0.9911191,0.0037241026,0.0006809785,0.0022437568,0.0020130023,0.00021902037],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031456007,0.0005946848,0.0006271205,0.0015070919,0.002077138,0.0070474828,0.0019777082,0.0018579086,0.011128408],"category_scores_gemma":[0.013221628,0.0006685844,0.0008483418,0.002138662,0.008085225,0.011245165,0.003249222,0.002795981,0.0038545155],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012247969,0.000020423893,0.0019699584,0.00048560483,0.00005690742,0.00042400346,0.0029484169,0.005222968,0.005985855,0.81335115,0.0077204485,0.16169174],"study_design_scores_gemma":[0.000027995915,0.000059213566,0.0026113538,0.0003569949,0.000046069108,0.0011431498,0.0016691476,0.028348483,0.008312254,0.76137245,0.19595835,0.00009462265],"about_ca_topic_score_codex":0.014664462,"about_ca_topic_score_gemma":0.009918449,"teacher_disagreement_score":0.014664462,"about_ca_system_score_codex":0.0038468975,"about_ca_system_score_gemma":0.0036665278,"threshold_uncertainty_score":0.037228167},"labels":[],"label_agreement":null},{"id":"W4317565429","doi":"10.5281/zenodo.7552977","title":"AML: A Novel Accuracy Metric for Critical Evaluation of Log Parsing Techniques (results)","year":2023,"lang":"en","type":"paratext","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Metric (unit); Computer science; Parsing; Artificial intelligence; Engineering","score_opus":0.10022410498489066,"score_gpt":0.3688968886522967,"score_spread":0.26867278366740605,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4317565429","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19918887,0.010466557,0.5459684,0.0016613802,0.0022356592,0.0012426879,0.06661114,0.1535606,0.019064695],"genre_scores_gemma":[0.5018263,0.0009890078,0.3866288,0.00052639126,0.00045825416,0.001477067,0.08938271,0.011768666,0.006942937],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9730636,0.007968762,0.003625577,0.004126442,0.010009404,0.0012063016],"domain_scores_gemma":[0.9464891,0.02952734,0.0029997667,0.00857094,0.011547284,0.000865498],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013102051,0.0031280452,0.0018251948,0.007322494,0.0010413278,0.0045197923,0.0030362643,0.0027982513,0.006735582],"category_scores_gemma":[0.055036563,0.0005067726,0.0013602037,0.004459859,0.0012008956,0.004523776,0.0025618381,0.0015816212,0.0057051363],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0047675446,0.00094518904,0.041780863,0.0047229016,0.0016525417,0.0005404874,0.0007819966,0.06532543,0.05100328,0.00796595,0.2202419,0.60027194],"study_design_scores_gemma":[0.0006837675,0.0028218634,0.052989755,0.0004981258,0.0011779414,0.0019181379,0.0011529058,0.5909499,0.22895624,0.023827035,0.094493784,0.0005305377],"about_ca_topic_score_codex":0.0033690534,"about_ca_topic_score_gemma":0.0038600015,"teacher_disagreement_score":0.013102051,"about_ca_system_score_codex":0.0017267697,"about_ca_system_score_gemma":0.0018611407,"threshold_uncertainty_score":0.069291115},"labels":[],"label_agreement":null},{"id":"W4317642541","doi":"10.1109/icimcis56303.2022.10017635","title":"English-to-Sundanese Translation using Neural Machine Translation: Collection and Analytics","year":2022,"lang":"en","type":"article","venue":"2022 International Conference on Informatics, Multimedia, Cyber and Information System (ICIMCIS)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Sydney Steel (Canada)","funders":"","keywords":"Machine translation; Computer science; Natural language processing; Artificial intelligence; Evaluation of machine translation; Transformer; Example-based machine translation; German; Text corpus; Machine translation software usability; Speech recognition; Linguistics; Engineering","score_opus":0.03258167084101219,"score_gpt":0.2761862244814549,"score_spread":0.24360455364044273,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4317642541","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.51607215,0.0035129932,0.2983151,0.0033189252,0.0013457501,0.0019755838,0.10416324,0.025002558,0.04629363],"genre_scores_gemma":[0.4276992,0.0014481914,0.28925782,0.0003363338,0.00031934038,0.0019396499,0.2629784,0.0021530774,0.013868002],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.997168,0.00092187995,0.0003773964,0.0006059196,0.00078881904,0.00013789514],"domain_scores_gemma":[0.9940057,0.0013476112,0.00028055307,0.0014858746,0.002745083,0.0001351413],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026147508,0.0010225729,0.0007236421,0.0037767782,0.0013462694,0.0016294167,0.0010214136,0.0006277499,0.004714801],"category_scores_gemma":[0.009465274,0.00043151877,0.0007726608,0.005285346,0.0008058083,0.0023300247,0.0028835877,0.001173499,0.0048418036],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008246819,0.0011441006,0.041713975,0.0019205245,0.00025336677,0.0020557665,0.0019215047,0.019692333,0.03430632,0.012238302,0.14056975,0.74335927],"study_design_scores_gemma":[0.00029482588,0.00085630384,0.10135355,0.00044286475,0.0003349724,0.0028617135,0.0050139404,0.3791307,0.14552665,0.03180023,0.3320082,0.00037596663],"about_ca_topic_score_codex":0.008609742,"about_ca_topic_score_gemma":0.011422797,"teacher_disagreement_score":0.008609742,"about_ca_system_score_codex":0.0009651806,"about_ca_system_score_gemma":0.002243742,"threshold_uncertainty_score":0.017119229},"labels":[],"label_agreement":null},{"id":"W4318203162","doi":"10.3765/amp.v10i0.5445","title":"Paradoxes of MaxEnt markedness","year":2023,"lang":"en","type":"article","venue":"Proceedings of the Annual Meetings on Phonology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Nautical Research Society","funders":"","keywords":"Probabilistic logic; Rule-based machine translation; Linguistics; Markedness; Phonology; Mathematics; Grammar; Principle of maximum entropy; Generalization; Computer science; Natural language processing; Artificial intelligence; Philosophy","score_opus":0.009555969793891108,"score_gpt":0.24851489498880525,"score_spread":0.23895892519491413,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4318203162","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.23887055,0.0048466073,0.43523067,0.040372316,0.0009610084,0.00009824615,0.0009923759,0.0014572749,0.277171],"genre_scores_gemma":[0.97141165,0.00044715556,0.019223496,0.0020506557,0.00035363354,0.00008072846,0.00015183023,0.00025502912,0.0060257926],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9943539,0.0020050535,0.00038254692,0.001543291,0.0012506107,0.0004647245],"domain_scores_gemma":[0.9785957,0.015521996,0.00085022335,0.0034559672,0.0009103068,0.00066587183],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0072903438,0.00049531186,0.0010405827,0.0016099634,0.0033895303,0.0037477184,0.0020691582,0.0027799176,0.008254447],"category_scores_gemma":[0.03586089,0.00065521314,0.0014825644,0.0011765984,0.013161361,0.01197639,0.0065804976,0.0059567867,0.00077184616],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000013705039,0.0000048468046,0.00016548319,0.000020272513,0.0000059640315,0.000042155792,0.0001919113,0.00055607775,0.000114422284,0.9960532,0.0006313333,0.0022006235],"study_design_scores_gemma":[0.0000042609067,0.000002280142,0.00006969756,0.0000052062724,0.0000021319704,0.000030345409,0.000020215943,0.001451463,0.00007923161,0.997207,0.0011241052,0.000004138174],"about_ca_topic_score_codex":0.0011717951,"about_ca_topic_score_gemma":0.0009950864,"teacher_disagreement_score":0.008254447,"about_ca_system_score_codex":0.0027742195,"about_ca_system_score_gemma":0.00081839587,"threshold_uncertainty_score":0.038555503},"labels":[],"label_agreement":null},{"id":"W4318245003","doi":"10.3765/elm.2.5389","title":"Informational content vs. discourse orientation: experimental and computational perspectives","year":2023,"lang":"en","type":"article","venue":"Experiments in Linguistic Meaning","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"Université du Québec à Montréal","keywords":"Linguistics; Meaning (existential); Semantics (computer science); Content (measure theory); Computer science; Computational linguistics; Psychology; Computational model; Natural language processing; Artificial intelligence; Mathematics; Philosophy","score_opus":0.03544889131103577,"score_gpt":0.3524142309365706,"score_spread":0.3169653396255348,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4318245003","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.88385487,0.0017941978,0.07065354,0.0027138747,0.00014261092,0.0007015019,0.0011950282,0.00039269714,0.038551655],"genre_scores_gemma":[0.9743736,0.00048082383,0.022324486,0.00048810942,0.00010300753,0.00076235895,0.00055501476,0.00011596511,0.0007967414],"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.989038,0.0079286825,0.0005712302,0.0011811281,0.0009846122,0.000296319],"domain_scores_gemma":[0.85678947,0.13025284,0.0042275633,0.0064976374,0.0016326354,0.00059988885],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010012924,0.0008521777,0.00061996985,0.00092511775,0.00062639924,0.0032065834,0.0015100626,0.0016206154,0.0068306285],"category_scores_gemma":[0.06781933,0.0006848521,0.0004374234,0.0008518955,0.0041712504,0.0065487786,0.002115966,0.0015986292,0.00062761916],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.01072032,0.014064597,0.09091767,0.0071221846,0.0012755154,0.00085213524,0.028608952,0.030867668,0.37232026,0.24276403,0.005086032,0.1954006],"study_design_scores_gemma":[0.0018715211,0.010203849,0.15311174,0.00065132,0.0011076347,0.0019572452,0.016325848,0.12489198,0.229439,0.438471,0.021193149,0.0007757269],"about_ca_topic_score_codex":0.0009401974,"about_ca_topic_score_gemma":0.000547037,"teacher_disagreement_score":0.010012924,"about_ca_system_score_codex":0.0006569486,"about_ca_system_score_gemma":0.0004275167,"threshold_uncertainty_score":0.052954018},"labels":[],"label_agreement":null},{"id":"W4318465586","doi":"10.1007/s10032-023-00427-w","title":"Large-scale genealogical information extraction from handwritten Quebec parish records","year":2023,"lang":"en","type":"article","venue":"International Journal on Document Analysis and Recognition (IJDAR)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Université du Québec à Chicoutimi","funders":"Association Nationale de la Recherche et de la Technologie","keywords":"Computer science; Workflow; Consistency (knowledge bases); Scale (ratio); Sample (material); Information extraction; Population; Artificial intelligence; Natural language processing; Information retrieval; Data mining; Database; Geography; Medicine; Cartography","score_opus":0.012081826517106386,"score_gpt":0.29168839802054464,"score_spread":0.27960657150343826,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4318465586","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.72820264,0.0045982026,0.04237662,0.00082825485,0.00022904044,0.0006873042,0.19073896,0.0068289894,0.02551006],"genre_scores_gemma":[0.7460549,0.0014165135,0.06518684,0.00011382943,0.00010714866,0.00024247047,0.16006298,0.00033881256,0.026476564],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99975544,0.000014022102,0.000016634325,0.00006584187,0.00009088822,0.000057106517],"domain_scores_gemma":[0.99885094,0.0001992776,0.00010491933,0.00012426393,0.0006522605,0.000068351],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00016544537,0.0004757779,0.00039769392,0.010188202,0.0012525634,0.0010144158,0.0007713487,0.0005526625,0.004731523],"category_scores_gemma":[0.0012512713,0.00021366986,0.00034991498,0.009637579,0.0002928137,0.0004428312,0.00035273642,0.00041407152,0.0020568664],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00038113847,0.0002023995,0.11139287,0.0009723843,0.00025170017,0.0033416164,0.0018255118,0.019422784,0.04472473,0.0024188163,0.09632257,0.71874356],"study_design_scores_gemma":[0.00005820305,0.000121372344,0.55873233,0.0003811162,0.00029948325,0.0015596783,0.0043293717,0.17968415,0.027815055,0.0018164116,0.22505592,0.00014699786],"about_ca_topic_score_codex":0.7610443,"about_ca_topic_score_gemma":0.88046783,"teacher_disagreement_score":0.23895568,"about_ca_system_score_codex":0.0028657399,"about_ca_system_score_gemma":0.004257022,"threshold_uncertainty_score":0.48072582},"labels":[],"label_agreement":null},{"id":"W4318776785","doi":"10.3390/data8020034","title":"Neural Coreference Resolution for Dutch Parliamentary Documents with the DutchParliament Dataset","year":2023,"lang":"en","type":"article","venue":"Data","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Nederlandse Organisatie voor Wetenschappelijk Onderzoek; Canadian Institute of Steel Construction","keywords":"Coreference; Computer science; Natural language processing; Artificial intelligence; Task (project management); Resolution (logic); Metadata; Annotation; Cluster analysis; Information retrieval; World Wide Web","score_opus":0.06760614956412057,"score_gpt":0.3378913798641697,"score_spread":0.27028523030004914,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4318776785","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3183012,0.0076265186,0.06787196,0.0018288315,0.0012481667,0.0018785232,0.5303452,0.0123960255,0.05850358],"genre_scores_gemma":[0.10643534,0.0004759914,0.045741197,0.00025092065,0.00006543515,0.00093276665,0.83736926,0.00037679222,0.008352206],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9971468,0.00077800214,0.00028827484,0.00093949656,0.00061971036,0.00022783894],"domain_scores_gemma":[0.99718493,0.00092038064,0.00020649875,0.0008219732,0.00073969516,0.00012643509],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022048054,0.0012544397,0.00082826416,0.003170518,0.0022811703,0.0011293617,0.0023973137,0.0019733722,0.007767016],"category_scores_gemma":[0.008530172,0.00035423538,0.0011949508,0.0040204404,0.0007091681,0.0018435882,0.0019640117,0.0015692113,0.005618265],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018482602,0.0008390049,0.013687454,0.0029661062,0.00049730495,0.002526313,0.0023359451,0.020236962,0.021089947,0.008037458,0.605945,0.31999028],"study_design_scores_gemma":[0.0006416556,0.00030247305,0.04213128,0.00042532862,0.00027404795,0.002223929,0.0035455325,0.12722144,0.03983228,0.0070797573,0.7760644,0.0002578863],"about_ca_topic_score_codex":0.050392583,"about_ca_topic_score_gemma":0.10341415,"teacher_disagreement_score":0.050392583,"about_ca_system_score_codex":0.0021007643,"about_ca_system_score_gemma":0.0022566176,"threshold_uncertainty_score":0.10019851},"labels":[],"label_agreement":null},{"id":"W4318826499","doi":"10.2139/ssrn.4340420","title":"Latent Utility and Permutation Invariance: A Revealed Preference Approach","year":2023,"lang":"en","type":"article","venue":"SSRN Electronic Journal","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Preference; Permutation (music); Mathematics; Econometrics; Statistics; Psychology; Physics","score_opus":0.025930422003346956,"score_gpt":0.2625932317837991,"score_spread":0.23666280978045218,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4318826499","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.040049654,0.00034258683,0.9507264,0.0010189405,0.00006136711,0.00014859404,0.00048525567,0.00017176331,0.006995316],"genre_scores_gemma":[0.8618668,0.0010365907,0.12635125,0.00030964514,0.00029274763,0.00050589227,0.00057780463,0.0001618142,0.0088974945],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.98539275,0.010309434,0.0004849481,0.0017561447,0.001287813,0.0007690026],"domain_scores_gemma":[0.9313657,0.052614495,0.00322986,0.009824351,0.0021120014,0.0008535173],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.014853961,0.0015361805,0.004073048,0.0023584869,0.0012285145,0.0050819847,0.0045141405,0.0024089094,0.013537357],"category_scores_gemma":[0.057805818,0.0015533564,0.003924458,0.004382501,0.0041073384,0.011124215,0.0030335882,0.0046439636,0.0011415554],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024180161,0.00019689344,0.0018023325,0.000119355296,0.0003765377,0.00018283814,0.00032580158,0.01713414,0.0005331176,0.9570114,0.0013150005,0.020760663],"study_design_scores_gemma":[0.000048838145,0.0000932966,0.00094194105,0.000020548205,0.0000949193,0.00008591567,0.00011030917,0.15790741,0.00019817986,0.8397837,0.00066348485,0.000051430296],"about_ca_topic_score_codex":0.004001051,"about_ca_topic_score_gemma":0.0035494533,"teacher_disagreement_score":0.014853961,"about_ca_system_score_codex":0.0021794734,"about_ca_system_score_gemma":0.002127865,"threshold_uncertainty_score":0.07855618},"labels":[],"label_agreement":null},{"id":"W4319346507","doi":"10.5334/pme.727","title":"&lt;p&gt;Strategic Paragraphing 2.0: Techniques for Enhancing Inter-Paragraph Coherence&lt;/p&gt;","year":2023,"lang":"en","type":"article","venue":"Perspectives on Medical Education","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Paragraph; Coherence (philosophical gambling strategy); Computer science; Mathematics; Statistics; World Wide Web","score_opus":0.01644959554107794,"score_gpt":0.33067782958825237,"score_spread":0.3142282340471744,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4319346507","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0028155532,0.00061762147,0.95680773,0.0022862952,0.0007527532,0.00023146409,0.00035029888,0.013835873,0.02230246],"genre_scores_gemma":[0.04390046,0.0009757415,0.90534824,0.0010800638,0.00047898982,0.00039768603,0.0007705556,0.006662796,0.04038551],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99733233,0.0012106283,0.00020588306,0.00038817676,0.00074189866,0.00012105158],"domain_scores_gemma":[0.9862574,0.00799679,0.0007096584,0.0027854154,0.0019257226,0.00032510303],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003858183,0.0011917776,0.00045633307,0.0015973541,0.0013197126,0.0044278693,0.0018383467,0.001605605,0.041137163],"category_scores_gemma":[0.016914755,0.00090450316,0.0007713094,0.0018868354,0.0020198298,0.005402573,0.002291906,0.0026625292,0.017780349],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024316588,0.000087549655,0.0004962821,0.0006991422,0.000032905595,0.0002569429,0.0020691939,0.0015317623,0.026211424,0.0661217,0.14047703,0.7617728],"study_design_scores_gemma":[0.00013144646,0.0001996829,0.0011780184,0.00031671787,0.00007645789,0.0011580451,0.0010577796,0.060735576,0.076485075,0.09947298,0.75905234,0.0001357782],"about_ca_topic_score_codex":0.001232908,"about_ca_topic_score_gemma":0.0023170377,"teacher_disagreement_score":0.041137163,"about_ca_system_score_codex":0.0007010244,"about_ca_system_score_gemma":0.0013924,"threshold_uncertainty_score":0.13761753},"labels":[],"label_agreement":null},{"id":"W4319443012","doi":"10.1007/978-981-19-8590-4_6","title":"Reflections on Remixing Open Access Content into Open Educational Resources: A New Paradigm for Sustainable Data-Driven Language Learning Systems Design in Higher Education","year":2023,"lang":"en","type":"book-chapter","venue":"Future education and learning spaces","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"University of Waikato","keywords":"Open educational resources; Computer science; Open education; Multimedia; World Wide Web","score_opus":0.16203991385795963,"score_gpt":0.4263413651949371,"score_spread":0.26430145133697747,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4319443012","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010942099,0.009579267,0.04939676,0.49331224,0.006064378,0.00012011431,0.00013520007,0.00036146137,0.4300884],"genre_scores_gemma":[0.2870531,0.016669855,0.053137697,0.07323198,0.0025405998,0.00035439187,0.00014199427,0.0015095893,0.5653607],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99455255,0.0025555291,0.00015102014,0.0004605129,0.0017937704,0.000486519],"domain_scores_gemma":[0.9920359,0.0061122533,0.0001219029,0.0004748231,0.00078227103,0.0004728569],"candidate_categories":["open_science"],"consensus_categories":[],"category_scores_codex":[0.010498658,0.00062006497,0.0003486925,0.0006756825,0.003980001,0.015166077,0.0022862486,0.009044095,0.014748459],"category_scores_gemma":[0.008772994,0.00053808995,0.0005621235,0.001250266,0.017714787,0.03090543,0.006371297,0.011766615,0.0032521791],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00001848867,0.000043223656,0.00006649269,0.0000852783,0.0000027632962,0.00022011762,0.014822808,0.0005367278,0.0007555271,0.8737418,0.084960684,0.0247461],"study_design_scores_gemma":[0.000008068827,0.000016359467,0.000084246865,0.0001772657,0.0000030648632,0.00016885188,0.010777631,0.00058996223,0.0012791053,0.15268481,0.83418936,0.000021289123],"about_ca_topic_score_codex":0.012035786,"about_ca_topic_score_gemma":0.018977206,"teacher_disagreement_score":0.99771374,"about_ca_system_score_codex":0.0073836343,"about_ca_system_score_gemma":0.005279052,"threshold_uncertainty_score":0.05552286},"labels":[{"model":"gemma","categories":["open_science","scholarly_communication"],"domain":null,"study_design":"theoretical_or_conceptual","genre":"other","about_ca_system":false,"about_ca_topic":false,"confidence":"low"},{"model":"gpt","categories":["scholarly_communication","open_science"],"domain":null,"study_design":"theoretical_or_conceptual","genre":"commentary","about_ca_system":false,"about_ca_topic":false,"confidence":"medium"}],"label_agreement":"agree"},{"id":"W4319597719","doi":"10.5334/johd.95","title":"MultiHATHI: A Complete Collection of Multilingual Prose Fiction in the HathiTrust Digital Library","year":2023,"lang":"en","type":"article","venue":"Journal of Open Humanities Data","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Bespoke; Metadata; Computer science; Digital library; Classifier (UML); English language; World Wide Web; Information retrieval; Natural language processing; Artificial intelligence; Linguistics; Literature; Art; Poetry","score_opus":0.13862741170011722,"score_gpt":0.3461929316426599,"score_spread":0.2075655199425427,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4319597719","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03420883,0.0033760616,0.0011150219,0.00060794235,0.00033503523,0.00010473407,0.93176776,0.0020775096,0.026407147],"genre_scores_gemma":[0.018916339,0.0006375014,0.0026408422,0.00012636386,0.00012871233,0.00014126622,0.96943116,0.00031841607,0.007659328],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99917006,0.00013802068,0.00010528737,0.00016874819,0.00029712613,0.00012068461],"domain_scores_gemma":[0.99735814,0.00070718717,0.00028033418,0.00053062744,0.0007534109,0.00037018655],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006231831,0.0008974321,0.0006937363,0.012471175,0.001797711,0.0026166989,0.0010388914,0.0011156502,0.02161973],"category_scores_gemma":[0.004316409,0.00029395576,0.00059531425,0.013865233,0.0007140151,0.0029672186,0.0024306204,0.0009840286,0.02651446],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025627547,0.000107474174,0.011461786,0.0025616784,0.00006383966,0.0005822787,0.0024780063,0.00044023036,0.0022560398,0.0031025058,0.9140636,0.062626414],"study_design_scores_gemma":[0.000033990495,0.00003248051,0.029533774,0.00037159552,0.000034295954,0.00048719443,0.0017505964,0.00089329766,0.0018872044,0.0009480707,0.96397346,0.00005401648],"about_ca_topic_score_codex":0.016643526,"about_ca_topic_score_gemma":0.05296014,"teacher_disagreement_score":0.02161973,"about_ca_system_score_codex":0.0014397914,"about_ca_system_score_gemma":0.00199101,"threshold_uncertainty_score":0.07232523},"labels":[],"label_agreement":null},{"id":"W4320082227","doi":"10.53482/2022_53_402","title":"Quantifying syntax similarity with a polynomial representation of dependency trees","year":2022,"lang":"en","type":"article","venue":"Glottometrics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"National Institute of General Medical Sciences; Government of Canada; Australian Government; National Science Foundation","keywords":"Dependency (UML); Syntax; Computer science; Abstract syntax tree; Natural language processing; Artificial intelligence; Abstract syntax; Representation (politics); Graph; Word grammar; Sentence; Dependency graph; Similarity (geometry); Theoretical computer science; Generative grammar; Relational grammar; Emergent grammar","score_opus":0.042532923176786454,"score_gpt":0.3129962674554597,"score_spread":0.27046334427867325,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4320082227","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.058825437,0.00020566514,0.9356467,0.00018150317,0.000024293866,0.00008727042,0.0008770708,0.00088880904,0.0032633175],"genre_scores_gemma":[0.6999678,0.00032121132,0.2952383,0.00010397648,0.00009104233,0.00022726151,0.0019418095,0.00039467297,0.0017139529],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9979101,0.0006278269,0.00015587122,0.00046064224,0.0006559183,0.00018958017],"domain_scores_gemma":[0.9917548,0.0043454207,0.001379641,0.0013997784,0.0008687452,0.00025166318],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001956124,0.00057780225,0.000626281,0.0074181524,0.0007510079,0.0020684956,0.0010832058,0.00082052924,0.0031254322],"category_scores_gemma":[0.014959539,0.0002837113,0.0008990523,0.008148298,0.001736716,0.004927348,0.0014951954,0.0010543389,0.00075982546],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028645477,0.00024970714,0.014549588,0.00042125385,0.00012999027,0.00035842397,0.001366628,0.18836881,0.030634059,0.5037256,0.005871046,0.25403845],"study_design_scores_gemma":[0.000021288037,0.00010042776,0.00870621,0.00003835148,0.000038556147,0.0002950326,0.0002729018,0.6646866,0.0058369962,0.31366068,0.0062596956,0.000083231665],"about_ca_topic_score_codex":0.003609201,"about_ca_topic_score_gemma":0.003123451,"teacher_disagreement_score":0.0074181524,"about_ca_system_score_codex":0.0014983147,"about_ca_system_score_gemma":0.00130844,"threshold_uncertainty_score":0.010871053},"labels":[],"label_agreement":null},{"id":"W4320492200","doi":"10.5539/ells.v13n1p33","title":"An Error Analysis of Coordinating Conjunction Misuse in Chinese ESL Learners’ Writings: A Corpus-based Approach","year":2023,"lang":"en","type":"article","venue":"English Language and Literature Studies","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Sentence; Linguistics; Conjunction (astronomy); China; Computer science; Corpus linguistics; Error analysis; Natural language processing; Psychology; Artificial intelligence; History","score_opus":0.012395309514158515,"score_gpt":0.3115324833737897,"score_spread":0.2991371738596312,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4320492200","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9803398,0.00047131552,0.0124967145,0.00014319956,0.000038729555,0.00033060316,0.0018084772,0.00010919429,0.004261954],"genre_scores_gemma":[0.97713,0.00032874258,0.015900243,0.000044425855,0.000039580453,0.00055772456,0.0039772433,0.00006276142,0.001959374],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.996172,0.0011084537,0.0007244869,0.00086261454,0.0009776996,0.00015475739],"domain_scores_gemma":[0.97963333,0.011263127,0.0023741785,0.0014395515,0.004983319,0.00030651529],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034629414,0.00054615334,0.00058908865,0.007315333,0.001419199,0.0018131429,0.00081323215,0.000629111,0.002112616],"category_scores_gemma":[0.013549244,0.00033075068,0.00051680894,0.005969993,0.0012839185,0.0019159319,0.0019933633,0.00078264484,0.00046553934],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00059503136,0.00074065634,0.6489064,0.0020515067,0.0002918234,0.0031107548,0.06788175,0.0021809184,0.02637837,0.0047679907,0.005463313,0.23763146],"study_design_scores_gemma":[0.00007782572,0.00042745,0.85193276,0.00046446256,0.00046922633,0.0023071088,0.053428147,0.04209794,0.022762563,0.0022726178,0.023590116,0.00016964287],"about_ca_topic_score_codex":0.011939095,"about_ca_topic_score_gemma":0.0136045795,"teacher_disagreement_score":0.011939095,"about_ca_system_score_codex":0.0011467383,"about_ca_system_score_gemma":0.0015274573,"threshold_uncertainty_score":0.023739219},"labels":[],"label_agreement":null},{"id":"W4320858207","doi":"10.18653/v1/2023.mwe-1","title":"Proceedings of the 19th Workshop on Multiword Expressions (MWE 2023)","year":2023,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Standing Committee on Language Education and Research; Agencia Estatal de Investigación; Agence Nationale de la Recherche; Riksbankens Jubileumsfond; Natural Sciences and Engineering Research Council of Canada; Xunta de Galicia; Irish Research Council; Science Foundation Ireland; European Regional Development Fund; Technological University Dublin; Agentschap Innoveren en Ondernemen","keywords":"Computer science; Programming language","score_opus":0.04115302236897675,"score_gpt":0.31105639783421796,"score_spread":0.26990337546524124,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4320858207","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022698933,0.05109461,0.6245316,0.024448164,0.033330336,0.00090127817,0.013514563,0.020351248,0.20912933],"genre_scores_gemma":[0.06260678,0.017478772,0.3544295,0.0050677476,0.005222425,0.0010045721,0.04727759,0.015474708,0.49143788],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9969773,0.001161959,0.00027182416,0.0007527671,0.00060408824,0.00023205827],"domain_scores_gemma":[0.99645776,0.001147082,0.00009322579,0.0011282436,0.0008670242,0.00030666662],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0044396394,0.0018005222,0.0020872918,0.0017619493,0.0012097717,0.0062550562,0.0027768684,0.0024336851,0.15544613],"category_scores_gemma":[0.0077603967,0.00076880486,0.0019581583,0.0018949738,0.0011100676,0.009734803,0.005125224,0.0030193832,0.08353872],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004515196,0.00022854224,0.000491035,0.00056721375,0.00011791753,0.00043177034,0.00057867775,0.0010166198,0.0073396442,0.02395469,0.4792035,0.48561877],"study_design_scores_gemma":[0.000032808934,0.000051814404,0.00081977336,0.00028891323,0.00003925824,0.00041695556,0.00028555276,0.004413498,0.0028934227,0.023037918,0.9676894,0.000030811912],"about_ca_topic_score_codex":0.0019551031,"about_ca_topic_score_gemma":0.0032543964,"teacher_disagreement_score":0.15544613,"about_ca_system_score_codex":0.0013335465,"about_ca_system_score_gemma":0.0013455726,"threshold_uncertainty_score":0.5200191},"labels":[],"label_agreement":null},{"id":"W4320910543","doi":"10.7202/1096256ar","title":"Examining translation behaviour of Turkish student translators in scientific text translation with think-aloud protocols","year":2023,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Turkish; Think aloud protocol; Computer science; Context (archaeology); Machine translation; Process (computing); Linguistics; Translation (biology); Protocol (science); Mathematics education; Natural language processing; Psychology; Human–computer interaction; Medicine","score_opus":0.0859470067226058,"score_gpt":0.333959970036745,"score_spread":0.2480129633141392,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4320910543","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98689365,0.000082128885,0.010540224,0.00020669974,0.000031640444,0.0006110257,0.00004755867,0.000077310484,0.0015097526],"genre_scores_gemma":[0.9727851,0.0002104412,0.021018136,0.00036884303,0.00002920791,0.002021898,0.00014943934,0.000095233874,0.0033217056],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9745455,0.018488428,0.0018785005,0.0020632197,0.0021318775,0.00089239824],"domain_scores_gemma":[0.9325586,0.039040998,0.0062464387,0.0037006543,0.016602837,0.0018504631],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.021914328,0.00078772905,0.00067885715,0.0013540359,0.0022284724,0.0032017073,0.001180083,0.0015551323,0.0021485975],"category_scores_gemma":[0.07356755,0.00074112375,0.0005610676,0.00092345855,0.0020784778,0.0016497988,0.0023378145,0.0016280356,0.0020002516],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010198005,0.0014139563,0.031579945,0.00058115576,0.000057835114,0.0010484903,0.86878705,0.00036178489,0.03296571,0.0006588294,0.00096894446,0.06055653],"study_design_scores_gemma":[0.00056764646,0.009918798,0.08564902,0.0008261129,0.00028349928,0.0032152508,0.7705333,0.009173825,0.072174996,0.0041081035,0.04298746,0.0005619609],"about_ca_topic_score_codex":0.00064301083,"about_ca_topic_score_gemma":0.0011286126,"teacher_disagreement_score":0.021914328,"about_ca_system_score_codex":0.0012433265,"about_ca_system_score_gemma":0.002539771,"threshold_uncertainty_score":0.11589545},"labels":[],"label_agreement":null},{"id":"W4323240463","doi":"10.5220/0011620900003393","title":"Adding Time and Subject Line Features to the Donor Journey","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Acadia University","funders":"","keywords":"Subject (documents); Computer science; Line (geometry); World Wide Web; Mathematics; Geometry","score_opus":0.01512059567864465,"score_gpt":0.2842979874119204,"score_spread":0.26917739173327576,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4323240463","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06007544,0.0016342029,0.70701504,0.01635752,0.005476455,0.00019440507,0.0027950946,0.017141055,0.18931082],"genre_scores_gemma":[0.41260484,0.0014748154,0.385466,0.002319613,0.0014572762,0.00018861458,0.0045250277,0.013405182,0.17855857],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9984875,0.0004488017,0.000120107914,0.00028334034,0.00046192968,0.00019829112],"domain_scores_gemma":[0.99315006,0.0014525879,0.00041093057,0.0025351683,0.0016971998,0.00075411645],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002540955,0.0006345819,0.00069144135,0.0017506267,0.0018537757,0.005581881,0.0014973739,0.0012330032,0.045203995],"category_scores_gemma":[0.012833954,0.00046364806,0.0008323427,0.0024989317,0.0009874965,0.016642433,0.004481046,0.0028316556,0.0193257],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010566439,0.00025134548,0.0034167173,0.00026668666,0.00003259619,0.0003198946,0.0019484549,0.0024213106,0.0060278405,0.23466699,0.083514646,0.66607684],"study_design_scores_gemma":[0.00009616867,0.00024800393,0.0013446531,0.00021297649,0.00007945254,0.00055447797,0.0020228028,0.023271143,0.009502144,0.2122276,0.7503258,0.000114770206],"about_ca_topic_score_codex":0.0025299666,"about_ca_topic_score_gemma":0.003415469,"teacher_disagreement_score":0.045203995,"about_ca_system_score_codex":0.0010723309,"about_ca_system_score_gemma":0.0019960126,"threshold_uncertainty_score":0.15122241},"labels":[],"label_agreement":null},{"id":"W4323314718","doi":"10.4324/9781003168348-19","title":"Translation Technology in Canada","year":2023,"lang":"en","type":"book-chapter","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Dictation; Translation (biology); Computer science; Natural language processing; Artificial intelligence; Speech recognition; Messenger RNA; Biology","score_opus":0.017639060776293358,"score_gpt":0.23044096988058502,"score_spread":0.21280190910429167,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4323314718","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0076943412,0.063780926,0.0028694815,0.021463363,0.0031199448,0.00006323792,0.0012763039,0.0004977298,0.8992347],"genre_scores_gemma":[0.08208443,0.07376226,0.005336702,0.0035214492,0.00029475326,0.00004926545,0.0011873319,0.00034551337,0.83341825],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99795496,0.000113467126,0.00007187532,0.00020003541,0.00117637,0.00048330319],"domain_scores_gemma":[0.9987527,0.00012170331,0.000030815347,0.000050087983,0.0007937237,0.00025091495],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008721548,0.00074302213,0.0003774567,0.002955131,0.0091279615,0.0073935767,0.0010407328,0.0016650114,0.03141693],"category_scores_gemma":[0.0017210288,0.00034641143,0.00044048257,0.008764544,0.0032703376,0.001737156,0.0017560591,0.0019496286,0.007166238],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":true,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000039560684,0.000025220224,0.0007811878,0.00064493465,0.000011624365,0.00067467464,0.004023857,0.0009189442,0.0014250895,0.35335326,0.3739749,0.26412675],"study_design_scores_gemma":[0.0000022260192,0.000004136158,0.00067284773,0.000093976894,0.00000285477,0.00009928608,0.0003444236,0.000092071015,0.00020620029,0.0017424768,0.99672985,0.000009633011],"about_ca_topic_score_codex":0.97865605,"about_ca_topic_score_gemma":0.98530245,"teacher_disagreement_score":0.89487153,"about_ca_system_score_codex":0.10512847,"about_ca_system_score_gemma":0.17911549,"threshold_uncertainty_score":0.762764},"labels":[],"label_agreement":null},{"id":"W4323544055","doi":"10.5007/2175-7968.2023.e85397","title":"Challenging machine translation engines: some Spanish-English linguistic problems put to the test","year":2023,"lang":"en","type":"article","venue":"Cadernos de Tradução","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"University of Ottawa","keywords":"Machine translation; Sentence; Computer science; Test (biology); Natural language processing; Linguistics; Work (physics); Quality (philosophy); Machine translation software usability; Artificial intelligence; Example-based machine translation; Engineering","score_opus":0.027478511805198662,"score_gpt":0.2651449271276318,"score_spread":0.23766641532243316,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4323544055","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.85189444,0.02152303,0.041167486,0.014998023,0.0033585953,0.0013439481,0.009139562,0.011510438,0.045064457],"genre_scores_gemma":[0.8390586,0.0040560015,0.07717457,0.005933892,0.0006255845,0.0006243341,0.046226677,0.0041739913,0.022126451],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9811298,0.007289635,0.0023164677,0.0024591282,0.0058887573,0.0009161152],"domain_scores_gemma":[0.96762633,0.015812697,0.0011535182,0.0032342316,0.011029171,0.0011440082],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.017023688,0.0021497705,0.0020642106,0.0024731548,0.0026259094,0.005141462,0.0017970384,0.003448617,0.00303258],"category_scores_gemma":[0.05295728,0.000674302,0.0012407429,0.0044647646,0.001890063,0.006069883,0.00356656,0.003398973,0.0032216068],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0044852947,0.003329366,0.037523728,0.007652574,0.0011561225,0.006287339,0.02443488,0.022022553,0.044190824,0.011749176,0.23875311,0.5984151],"study_design_scores_gemma":[0.0026565306,0.005996177,0.05986127,0.0016335894,0.00094516494,0.009748386,0.033001576,0.09280211,0.096790545,0.016420841,0.67942685,0.000717001],"about_ca_topic_score_codex":0.0129805,"about_ca_topic_score_gemma":0.017706236,"teacher_disagreement_score":0.017023688,"about_ca_system_score_codex":0.0023128123,"about_ca_system_score_gemma":0.0029664792,"threshold_uncertainty_score":0.09003091},"labels":[],"label_agreement":null},{"id":"W4323967470","doi":"10.1007/978-3-031-21780-7_10","title":"Rules Are Rules: Rhetorical Figures and Algorithms","year":2023,"lang":"en","type":"book-chapter","venue":"Studies in computational intelligence","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Rhetorical question; Linguistics; Computer science; Reciprocal; Function (biology); Antithesis; Trope (literature); Rhetorical device; Rank (graph theory); Algorithm; Natural language processing; Artificial intelligence; Mathematics; Philosophy","score_opus":0.11351547814976995,"score_gpt":0.3775380277179588,"score_spread":0.2640225495681888,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4323967470","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0032041317,0.030612363,0.32168958,0.0072847675,0.002422269,0.000117207295,0.0006381606,0.0008699174,0.6331616],"genre_scores_gemma":[0.18279818,0.03326309,0.34045166,0.0029408936,0.004953911,0.0005926618,0.0025580295,0.0017662602,0.43067533],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9988913,0.00041223067,0.0000842268,0.00020533909,0.00036428007,0.00004273066],"domain_scores_gemma":[0.9971886,0.0020334283,0.000093085495,0.00033415505,0.00028886844,0.0000618931],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001188794,0.0010393274,0.0009092019,0.0025477333,0.0012790632,0.008900016,0.0016511219,0.0017219605,0.01972566],"category_scores_gemma":[0.005589077,0.00089461956,0.0005794722,0.0039139963,0.006913882,0.016128922,0.0011195681,0.003338909,0.009199242],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000003546235,0.0000068528498,0.000015437365,0.00005805241,0.0000020947823,0.000012171546,0.00019457394,0.00033045077,0.00007926065,0.96482354,0.012678676,0.021795347],"study_design_scores_gemma":[0.00000350343,0.0000041328235,0.000025782429,0.00006291478,0.0000033859817,0.000044116,0.000077021316,0.0013072686,0.00016421547,0.8814216,0.1168809,0.000005138386],"about_ca_topic_score_codex":0.0011357788,"about_ca_topic_score_gemma":0.0011467945,"teacher_disagreement_score":0.01972566,"about_ca_system_score_codex":0.0014572733,"about_ca_system_score_gemma":0.0009178179,"threshold_uncertainty_score":0.0659889},"labels":[],"label_agreement":null},{"id":"W4327928015","doi":"10.1109/icsc56153.2023.00025","title":"Automatic Identification of Chinese Paired Discourse Connectives","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Treebank; Natural language processing; Computer science; Artificial intelligence; Identification (biology); SIGNAL (programming language); Simple (philosophy); Relation (database); Speech recognition; Annotation; Programming language","score_opus":0.012126419459346326,"score_gpt":0.31375261518242037,"score_spread":0.30162619572307403,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4327928015","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5533243,0.0037041116,0.3944167,0.0010632866,0.000664351,0.0012479816,0.013786303,0.010129956,0.021663085],"genre_scores_gemma":[0.60289526,0.0006378293,0.37630606,0.00015140645,0.00017006809,0.0005483821,0.013593266,0.00041824477,0.0052794586],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99844354,0.00033938285,0.00013776311,0.0006727051,0.00029523092,0.000111452726],"domain_scores_gemma":[0.9963471,0.0017394257,0.00036200957,0.0002623748,0.0011243377,0.0001647878],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015027255,0.0011283015,0.0006696271,0.0061878795,0.0014717682,0.0012113936,0.0009378439,0.0005984922,0.0057616066],"category_scores_gemma":[0.004710346,0.00045778765,0.0004182188,0.0024315661,0.00079247076,0.0023865122,0.0016942314,0.0009409061,0.0016533219],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008771774,0.00023202281,0.022809597,0.002901987,0.000085475425,0.0027758323,0.0065015308,0.0038957747,0.25581148,0.02744131,0.026845045,0.6498228],"study_design_scores_gemma":[0.00031872588,0.0003773801,0.06963486,0.00059480564,0.00041328408,0.003149692,0.00801832,0.34242433,0.37127352,0.025724838,0.1777961,0.00027417674],"about_ca_topic_score_codex":0.0066588214,"about_ca_topic_score_gemma":0.008178803,"teacher_disagreement_score":0.0066588214,"about_ca_system_score_codex":0.001013683,"about_ca_system_score_gemma":0.0027537397,"threshold_uncertainty_score":0.019274473},"labels":[],"label_agreement":null},{"id":"W4360986317","doi":"","title":"Proceedings of the Workshop on Interactions between Data Mining and Natural Language Processing 2017co-located with the European Conference on Machine Learning and Principles and Practice of Knowledge Discovery in Databases (ECML-PKDD 2017)","year":2017,"lang":"en","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Computer science; Knowledge extraction; Data science; Natural (archaeology); Information retrieval; Data mining; Geography; Archaeology","score_opus":0.05595412313270085,"score_gpt":0.322213632174657,"score_spread":0.2662595090419561,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4360986317","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013420256,0.03818107,0.809364,0.07181603,0.026106002,0.00054043,0.003008045,0.0032551102,0.034309026],"genre_scores_gemma":[0.10130983,0.027093993,0.6978237,0.009054223,0.014609096,0.0008293565,0.015499215,0.004416271,0.12936437],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.98625016,0.007316921,0.0011914975,0.00187089,0.0028207318,0.0005497871],"domain_scores_gemma":[0.9625309,0.020625025,0.0006986431,0.006430631,0.006952852,0.0027619],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.023907717,0.0014731986,0.0029610924,0.002640727,0.0015405598,0.014453353,0.003491249,0.003078756,0.02762648],"category_scores_gemma":[0.033876944,0.0011794256,0.0019002733,0.0032040577,0.002371289,0.012891077,0.007710981,0.008442345,0.010585684],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00084306166,0.0007953946,0.0017055897,0.0013520662,0.00037503333,0.00045118504,0.0025857824,0.0036928589,0.008610943,0.08481699,0.34744254,0.54732853],"study_design_scores_gemma":[0.00011511698,0.00015859636,0.0014071504,0.0006152971,0.00016008664,0.00052467885,0.00086689705,0.029239926,0.0077829342,0.13848978,0.8205462,0.00009330551],"about_ca_topic_score_codex":0.0022465887,"about_ca_topic_score_gemma":0.0032694999,"teacher_disagreement_score":0.02762648,"about_ca_system_score_codex":0.0022114473,"about_ca_system_score_gemma":0.0039150002,"threshold_uncertainty_score":0.1264376},"labels":[],"label_agreement":null},{"id":"W4362503760","doi":"10.4000/afas.7496","title":"Les données orales en linguistique","year":2022,"lang":"fr","type":"article","venue":"Bulletin de l’AFAS","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ministère de l’Emploi et de la Solidarité Sociale (Québec)","funders":"","keywords":"Humanities; Political science; Philosophy","score_opus":0.029096084392661606,"score_gpt":0.29585361035587315,"score_spread":0.26675752596321156,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4362503760","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.26570848,0.029472409,0.18640207,0.028757315,0.0044750124,0.002977738,0.049249608,0.003545635,0.42941183],"genre_scores_gemma":[0.601152,0.024734125,0.17111129,0.0056320406,0.0014298423,0.0051131085,0.025484482,0.0021084263,0.1632347],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.97791904,0.010039668,0.002462074,0.0024126119,0.0065189926,0.0006475418],"domain_scores_gemma":[0.9403399,0.03649478,0.0022800148,0.008375707,0.011770928,0.0007386828],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.017579783,0.00094513356,0.0012788658,0.007776212,0.0035563563,0.011589542,0.001521339,0.0019939656,0.03061666],"category_scores_gemma":[0.06900119,0.00068270124,0.0011916738,0.007868684,0.004073164,0.0068978975,0.0058531505,0.0027705138,0.009090474],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007346809,0.00023968273,0.027916694,0.0071816146,0.00032977664,0.0015035691,0.17223474,0.0012153535,0.019381773,0.05677943,0.059280343,0.65320235],"study_design_scores_gemma":[0.00004209526,0.00019944797,0.042484656,0.0036580253,0.00023035842,0.0011638808,0.05356935,0.0009513171,0.009355207,0.022431321,0.865671,0.00024336779],"about_ca_topic_score_codex":0.026445678,"about_ca_topic_score_gemma":0.027006552,"teacher_disagreement_score":0.03061666,"about_ca_system_score_codex":0.0039224415,"about_ca_system_score_gemma":0.0105304625,"threshold_uncertainty_score":0.10242295},"labels":[],"label_agreement":null},{"id":"W4362577036","doi":"10.22215/etd/2023-15442","title":"MAWT: Multi-Attention-Weight Transformers","year":2023,"lang":"en","type":"dissertation","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"Ministère de la Défense Nationale","keywords":"Transformer; Computer science; Machine translation; Artificial intelligence; Parsing; Test set; Natural language processing; Artificial neural network; Machine learning; Engineering; Voltage","score_opus":0.014191459130621064,"score_gpt":0.2986226598812279,"score_spread":0.2844312007506068,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4362577036","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019833883,0.00044692858,0.943117,0.00032776358,0.0002740145,0.00021438551,0.0008312367,0.028974442,0.00598046],"genre_scores_gemma":[0.52803296,0.00057302695,0.43856585,0.00084911427,0.00017332743,0.0005392431,0.0035208503,0.0027013142,0.025044268],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9993332,0.000118416196,0.00007073706,0.00021423434,0.00018288608,0.00008053549],"domain_scores_gemma":[0.99882776,0.00044838456,0.000092562346,0.0003354508,0.00023792559,0.00005804115],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011157353,0.0012743706,0.00061644433,0.00088662986,0.00046114365,0.0014214206,0.0029798935,0.0011277351,0.011915285],"category_scores_gemma":[0.0050347038,0.000584765,0.0012043259,0.0007823323,0.00070197904,0.0047278465,0.0017267853,0.0020719052,0.0057146978],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006932221,0.000260592,0.0039001948,0.00036885843,0.0001871488,0.00050835364,0.00031445475,0.08145485,0.033056248,0.049580496,0.03695402,0.79272157],"study_design_scores_gemma":[0.0000716227,0.00023595797,0.0005776642,0.000042589345,0.0001142632,0.0003710211,0.000052866897,0.86957884,0.03582012,0.0731248,0.019963203,0.00004706305],"about_ca_topic_score_codex":0.003012604,"about_ca_topic_score_gemma":0.0050516487,"teacher_disagreement_score":0.011915285,"about_ca_system_score_codex":0.000834378,"about_ca_system_score_gemma":0.001139737,"threshold_uncertainty_score":0.039860666},"labels":[],"label_agreement":null},{"id":"W4362698607","doi":"10.5430/wjel.v13n5p319","title":"A New Computer Science Academic Word List","year":2023,"lang":"en","type":"article","venue":"World Journal of English Language","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Vocabulary; Computer science; Syllabus; Reading (process); Meaning (existential); Mathematics education; Word (group theory); Focus (optics); English for academic purposes; Linguistics; Psychology","score_opus":0.012123159517566992,"score_gpt":0.2896774405306479,"score_spread":0.27755428101308094,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4362698607","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.60007536,0.005318834,0.13210137,0.004723977,0.0021513994,0.026930742,0.14602722,0.003041376,0.07962977],"genre_scores_gemma":[0.3179312,0.0026887164,0.4361836,0.0013030302,0.00035041664,0.053582143,0.16871758,0.001120462,0.018122798],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.996123,0.0007039302,0.0012496806,0.00064850826,0.0011500308,0.00012472678],"domain_scores_gemma":[0.9813983,0.006474751,0.0013745717,0.0008278316,0.009318974,0.0006056141],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003930995,0.00059720414,0.00092864624,0.018893316,0.0020622802,0.0034177806,0.0011899655,0.0007455755,0.00906392],"category_scores_gemma":[0.021944376,0.0005027357,0.00069849665,0.013868623,0.0010584076,0.0062472094,0.0034822545,0.0016441566,0.0036028465],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005117054,0.00080106925,0.055235386,0.013435224,0.00011763242,0.0014739464,0.058319703,0.0015720857,0.036399387,0.049904857,0.122471936,0.65975696],"study_design_scores_gemma":[0.00029816106,0.0005144161,0.0836535,0.0034463604,0.00029467276,0.0013481964,0.045686904,0.0053236913,0.012325817,0.012881608,0.8338853,0.00034142434],"about_ca_topic_score_codex":0.005707547,"about_ca_topic_score_gemma":0.010453073,"teacher_disagreement_score":0.018893316,"about_ca_system_score_codex":0.0030062518,"about_ca_system_score_gemma":0.0070699714,"threshold_uncertainty_score":0.030321836},"labels":[],"label_agreement":null},{"id":"W4365211599","doi":"10.1145/3591269","title":"flap: A Deterministic Parser with Fused Lexing","year":2023,"lang":"en","type":"article","venue":"Proceedings of the ACM on Programming Languages","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"European Research Council; Horizon 2020 Framework Programme; Isaac Newton Trust; European Commission","keywords":"Computer science; Security token; Parsing; Parser combinator; Programming language; Context (archaeology); LR parser; Modularity (biology); Ambiguity; Interface (matter); Theoretical computer science; Parallel computing; Operating system","score_opus":0.017519603306589217,"score_gpt":0.2787272433611151,"score_spread":0.2612076400545259,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4365211599","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0045206537,0.00011292368,0.87935346,0.00020597798,0.00012140642,0.00012518791,0.0010545393,0.1114556,0.0030502484],"genre_scores_gemma":[0.10537627,0.0002209116,0.85088015,0.0006213465,0.00009173248,0.00031360166,0.0034045589,0.028096719,0.010994631],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9970113,0.00044303006,0.00033245306,0.0009353338,0.0009775606,0.00030026239],"domain_scores_gemma":[0.9960998,0.0013152666,0.00023757626,0.0015095255,0.0007141988,0.00012346233],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025996591,0.0016754374,0.001245899,0.0017450957,0.0010084924,0.00343561,0.0042353706,0.002172055,0.015142094],"category_scores_gemma":[0.008863979,0.002150722,0.002998195,0.0018351524,0.0031316604,0.007132077,0.005786314,0.0032547752,0.009170741],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011042119,0.00029323483,0.004604423,0.0010154933,0.00023298303,0.0012446071,0.0012495107,0.056154888,0.047796972,0.27890658,0.10749509,0.49990204],"study_design_scores_gemma":[0.0003138926,0.00020957629,0.0009339934,0.00019376948,0.0001971974,0.001503949,0.00024547067,0.39919233,0.15275015,0.28608614,0.1579397,0.00043379454],"about_ca_topic_score_codex":0.0054035443,"about_ca_topic_score_gemma":0.00646093,"teacher_disagreement_score":0.015142094,"about_ca_system_score_codex":0.0020204922,"about_ca_system_score_gemma":0.004815814,"threshold_uncertainty_score":0.050655305},"labels":[],"label_agreement":null},{"id":"W4367047466","doi":"10.1145/3543507.3583191","title":"Word Sense Disambiguation by Refining Target Word Embedding","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Word (group theory); Word embedding; SemEval; Word-sense disambiguation; Embedding; Linguistics; WordNet","score_opus":0.017438785269707276,"score_gpt":0.3011148116189133,"score_spread":0.283676026349206,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4367047466","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.093795605,0.0045641605,0.8789894,0.000686239,0.00062978955,0.00033716476,0.0031185485,0.01211649,0.0057626064],"genre_scores_gemma":[0.5219203,0.0023674408,0.451919,0.00069884915,0.00027834086,0.00024408927,0.012718151,0.0011646481,0.008689194],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990761,0.00013395876,0.00010681539,0.00050252146,0.0001236014,0.000056890134],"domain_scores_gemma":[0.9991762,0.00026788647,0.00008017342,0.0002265694,0.00020877869,0.000040409803],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010452286,0.002206697,0.0011604378,0.003099766,0.00058914255,0.0012684176,0.0014754892,0.0010222119,0.002741071],"category_scores_gemma":[0.0026482039,0.0005749481,0.001553596,0.0020980362,0.00078397046,0.004148776,0.0024728302,0.0016480868,0.0033954808],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005101168,0.00027977285,0.0053505916,0.00083178957,0.00026763612,0.0006382641,0.0008706886,0.061104834,0.053320896,0.010823598,0.025989782,0.84001213],"study_design_scores_gemma":[0.000092236456,0.00019417454,0.0034346385,0.00014459471,0.00025777926,0.00064417685,0.00040693974,0.88005614,0.032289345,0.048293345,0.034087423,0.00009922763],"about_ca_topic_score_codex":0.0035025799,"about_ca_topic_score_gemma":0.008328312,"teacher_disagreement_score":0.0035025799,"about_ca_system_score_codex":0.00047940618,"about_ca_system_score_gemma":0.0011338718,"threshold_uncertainty_score":0.009169817},"labels":[],"label_agreement":null},{"id":"W4367183982","doi":"10.31234/osf.io/4grqf","title":"Neural Networks Can Learn Patterns of Island-insensitivity in Norwegian","year":2023,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Norwegian; Recurrent neural network; Focus (optics); Computer science; Artificial intelligence; Subject (documents); Distribution (mathematics); Deep neural networks; Artificial neural network; Linguistics; Mathematics; Philosophy","score_opus":0.021704997018107332,"score_gpt":0.2736134344101312,"score_spread":0.25190843739202384,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4367183982","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8830786,0.00022276683,0.101832576,0.00037827896,0.000055600165,0.000031974105,0.0010536162,0.0010880831,0.012258576],"genre_scores_gemma":[0.9806042,0.00009404851,0.015781308,0.00006737272,0.000006102855,0.000021568474,0.0012706089,0.0001193622,0.002035535],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99948174,0.0001494694,0.000041385003,0.00021056584,0.00006067009,0.00005627432],"domain_scores_gemma":[0.99739265,0.0017846605,0.0002592171,0.00033763092,0.00017235533,0.000053483447],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001203389,0.00075099175,0.00036361054,0.00035690088,0.0003190387,0.0007990314,0.00049814634,0.00041028316,0.0029238975],"category_scores_gemma":[0.007959465,0.00048088448,0.00063053484,0.00030843393,0.0010836858,0.0025160764,0.001114295,0.0010064816,0.00051039463],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020299316,0.00026972385,0.072415724,0.0010693902,0.0003546478,0.003101653,0.01103015,0.3568075,0.22557777,0.03766794,0.008139215,0.2815364],"study_design_scores_gemma":[0.00008008789,0.00040260618,0.063082434,0.00017510528,0.00023105787,0.0006844223,0.0027727846,0.7674849,0.05170874,0.10086394,0.01234354,0.00017038244],"about_ca_topic_score_codex":0.018396204,"about_ca_topic_score_gemma":0.025992978,"teacher_disagreement_score":0.018396204,"about_ca_system_score_codex":0.00083357294,"about_ca_system_score_gemma":0.00042867713,"threshold_uncertainty_score":0.036578238},"labels":[],"label_agreement":null},{"id":"W4367189822","doi":"10.48550/arxiv.2304.13292","title":"Zero-Shot Slot and Intent Detection in Low-Resource Languages","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Compute Canada; Advanced Micro Devices","keywords":"Computer science; Dialog box; Margin (machine learning); Task (project management); Zero (linguistics); Encoder; Generalization; Natural language understanding; Natural language processing; Baseline (sea); Language model; Natural language; Artificial intelligence; Resource (disambiguation); Speech recognition; Machine learning; Linguistics; World Wide Web; Mathematics; Engineering","score_opus":0.056118803737671676,"score_gpt":0.21425296166809754,"score_spread":0.15813415793042587,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4367189822","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.40522903,0.0014073267,0.5701142,0.0008105463,0.00032495835,0.00020358094,0.0020470235,0.013208447,0.0066548293],"genre_scores_gemma":[0.8307395,0.00020310005,0.15879555,0.00032712866,0.00011098533,0.00013513092,0.0038632436,0.00062663166,0.0051986733],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99784684,0.00091385096,0.000085913336,0.000672696,0.0002665186,0.00021418127],"domain_scores_gemma":[0.9940037,0.004028366,0.0002518363,0.0009477971,0.00051148183,0.00025680065],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027748474,0.0014475549,0.0011745817,0.0008303282,0.0007976974,0.0017513069,0.0018929141,0.0017228788,0.003614921],"category_scores_gemma":[0.010461126,0.0005283633,0.0010220781,0.0005408569,0.0009987287,0.004335926,0.0027987543,0.0026977437,0.0026019635],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0029815733,0.0010916663,0.013130902,0.0012083777,0.00021544186,0.0014105751,0.0030864463,0.07907279,0.10162291,0.017205615,0.028895982,0.7500778],"study_design_scores_gemma":[0.0000839674,0.0004540157,0.0043811463,0.00006138085,0.00005121456,0.0006877953,0.00086253026,0.92092204,0.035195787,0.030021092,0.00716171,0.00011736631],"about_ca_topic_score_codex":0.003589715,"about_ca_topic_score_gemma":0.0050971843,"teacher_disagreement_score":0.003614921,"about_ca_system_score_codex":0.000608567,"about_ca_system_score_gemma":0.0010218921,"threshold_uncertainty_score":0.014674962},"labels":[],"label_agreement":null},{"id":"W4367322989","doi":"10.3138/jsp-2022-0039","title":"Exploring the Process and Strategies of Chinese–English Abstract Writing Using Machine Translation Tools","year":2023,"lang":"en","type":"article","venue":"Journal of Scholarly Publishing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Machine translation; Computer science; Think aloud protocol; Quality (philosophy); Session (web analytics); Process (computing); Writing process; Mathematics education; Academic writing; Psychology; Natural language processing; World Wide Web; Human–computer interaction","score_opus":0.10197991004231756,"score_gpt":0.3235418565291722,"score_spread":0.22156194648685468,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4367322989","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9529227,0.0006436526,0.026217956,0.0009116954,0.000040506275,0.0003572398,0.000039005954,0.00012455197,0.018742656],"genre_scores_gemma":[0.96968126,0.0005916808,0.024010176,0.00020646222,0.00001686471,0.00018772284,0.000058794838,0.000056036548,0.005191069],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9882383,0.007966525,0.0005656946,0.0007112397,0.0020291035,0.0004891478],"domain_scores_gemma":[0.9729111,0.019472644,0.0024589226,0.0013517718,0.0029515275,0.0008539848],"candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.01137158,0.00089011306,0.00055961363,0.0020521814,0.0026887,0.0061872173,0.0014136632,0.0011482598,0.0016242748],"category_scores_gemma":[0.03567365,0.0003575297,0.00042257496,0.0018072022,0.0039365664,0.0038536973,0.0024539032,0.001649725,0.00061460753],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000093514,0.00022812831,0.015271734,0.0006448658,0.000029110723,0.001787763,0.835704,0.0004474282,0.014395978,0.0071325777,0.00088542514,0.12337945],"study_design_scores_gemma":[0.00011284246,0.00079355336,0.05042179,0.0012925231,0.00014290468,0.0047576665,0.7937666,0.011992519,0.030120337,0.01639103,0.089936614,0.00027163865],"about_ca_topic_score_codex":0.0028272602,"about_ca_topic_score_gemma":0.0032708566,"teacher_disagreement_score":0.9938128,"about_ca_system_score_codex":0.002076147,"about_ca_system_score_gemma":0.005509758,"threshold_uncertainty_score":0.060139418},"labels":[],"label_agreement":null},{"id":"W4375841709","doi":"10.25189/2675-4916.2023.v4.n1.id666.a","title":"Resposta dos Autores: OS ACERVOS E A DOCUMENTAÇÃO LINGUÍSTICA","year":2023,"lang":"pt","type":"peer-review","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Documentation; Indigenous; Resource (disambiguation); Cultural heritage; Relevance (law); Work (physics); Latin Americans; Library science; Political science; Sociology; Public relations; Computer science; Engineering; Law; Ecology","score_opus":0.04172355721918462,"score_gpt":0.3671238500021626,"score_spread":0.32540029278297794,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4375841709","genre_codex":"other","genre_gemma":"commentary","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"commentary","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17735417,0.049419392,0.027692804,0.21312706,0.005815858,0.0003976658,0.0003735671,0.0005983829,0.52522117],"genre_scores_gemma":[0.89560854,0.017370302,0.008547943,0.0053163716,0.0017769147,0.00015798699,0.00010977514,0.00045031894,0.07066188],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.97410405,0.014760903,0.001127206,0.0018180779,0.0074089067,0.0007807898],"domain_scores_gemma":[0.923717,0.03567593,0.007047099,0.008847383,0.021829927,0.0028826485],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.024697378,0.00028385926,0.00043107558,0.0033960848,0.005884415,0.01404891,0.0012562979,0.0017896097,0.0076008392],"category_scores_gemma":[0.11341774,0.000292245,0.00023664362,0.003185813,0.011927463,0.011781256,0.0059253285,0.0030433699,0.0018813269],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006059071,0.000074296564,0.014137708,0.0011998855,0.00003792346,0.0021056968,0.25852138,0.00035991552,0.0019197188,0.26333284,0.059269924,0.39898014],"study_design_scores_gemma":[0.0000083740415,0.00003990387,0.0062381644,0.0013260897,0.00002741321,0.0015665725,0.06840544,0.00042193837,0.00083827315,0.02640465,0.8946815,0.00004161861],"about_ca_topic_score_codex":0.0150692165,"about_ca_topic_score_gemma":0.025630165,"teacher_disagreement_score":0.024697378,"about_ca_system_score_codex":0.004713819,"about_ca_system_score_gemma":0.014086987,"threshold_uncertainty_score":0.1306138},"labels":[],"label_agreement":null},{"id":"W4375949401","doi":"10.1145/3591208","title":"Editorial for the Special Issue on Computational Linguistics Processing in Low-Resource Indigenous Languages","year":2023,"lang":"en","type":"article","venue":"ACM Transactions on Asian and Low-Resource Language Information Processing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Brandon University","funders":"","keywords":"Indigenous; Citation; Library science; Resource (disambiguation); Computer science; History; Media studies; Linguistics; Sociology; Philosophy","score_opus":0.006931384925798461,"score_gpt":0.2695800335462915,"score_spread":0.26264864862049303,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4375949401","genre_codex":"editorial","genre_gemma":"editorial","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"editorial","genre_consensus":"editorial","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00007725993,0.001486748,0.00043386163,0.02485135,0.96783495,0.00003610352,0.00018233414,0.00014973016,0.0049476237],"genre_scores_gemma":[0.00077989773,0.003250552,0.00046086215,0.011086048,0.94402534,0.000052087056,0.00029608415,0.0003576083,0.03969152],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9963875,0.0004304699,0.00043957512,0.0005589951,0.0018580956,0.00032542198],"domain_scores_gemma":[0.97578377,0.0049108453,0.0009302303,0.0008947844,0.013935143,0.003545202],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00430154,0.0019294057,0.0021619094,0.003105465,0.0026096883,0.008520373,0.0021588872,0.0046295356,0.11819914],"category_scores_gemma":[0.019330889,0.000718132,0.002230681,0.0015946939,0.0015216917,0.005633582,0.0018195638,0.0078731375,0.062038574],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000013643531,0.000006352714,0.000024396888,0.00009980474,0.0000062869713,0.000030891628,0.000012294193,0.000013249187,0.00006588869,0.00027723433,0.99541074,0.004039177],"study_design_scores_gemma":[0.00001301836,0.000014050493,0.00017781799,0.00017605671,0.000014826909,0.00009249716,0.000056352084,0.00008978549,0.00009770458,0.0008079111,0.9984491,0.000010938027],"about_ca_topic_score_codex":0.0011154903,"about_ca_topic_score_gemma":0.0028108922,"teacher_disagreement_score":0.11819914,"about_ca_system_score_codex":0.002082961,"about_ca_system_score_gemma":0.0038697936,"threshold_uncertainty_score":0.39541554},"labels":[],"label_agreement":null},{"id":"W4376167025","doi":"10.48550/arxiv.2305.05858","title":"Vārta: A Large-Scale Headline-Generation Dataset for Indic Languages","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Headline; Computer science; Natural language processing; Scale (ratio); Artificial intelligence; Variety (cybernetics); Quality (philosophy); Information retrieval; Data science; Linguistics; Geography","score_opus":0.09930359967213902,"score_gpt":0.25590151813775097,"score_spread":0.15659791846561194,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4376167025","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029354215,0.0014190577,0.0059556095,0.0007532377,0.00043690403,0.0004077399,0.9328522,0.016212324,0.012608782],"genre_scores_gemma":[0.01165772,0.0002025554,0.009795621,0.00014844588,0.00006416416,0.00027310822,0.9752456,0.000576035,0.0020368695],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9985868,0.00032612035,0.00021012966,0.00034052902,0.00040194657,0.00013453196],"domain_scores_gemma":[0.99539846,0.0016200728,0.00040015872,0.00089063565,0.0012942147,0.00039649737],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001223708,0.0018336393,0.0007765479,0.006540702,0.0015487663,0.0017124369,0.0018324214,0.001864029,0.016752053],"category_scores_gemma":[0.007390967,0.000481081,0.0010786051,0.0054608956,0.00066471717,0.0024045806,0.0018316339,0.0018162662,0.020971097],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00031277214,0.00027166653,0.0047748196,0.0018937811,0.00009598857,0.0006980651,0.00054155866,0.0016122495,0.0047070556,0.0016368151,0.9412285,0.04222672],"study_design_scores_gemma":[0.00060194597,0.00020328982,0.019940425,0.00032265385,0.00014943277,0.0014359214,0.0014817758,0.019854758,0.013117763,0.0031445827,0.9395694,0.0001779471],"about_ca_topic_score_codex":0.011831955,"about_ca_topic_score_gemma":0.029927352,"teacher_disagreement_score":0.016752053,"about_ca_system_score_codex":0.0009998659,"about_ca_system_score_gemma":0.0019870724,"threshold_uncertainty_score":0.05604124},"labels":[],"label_agreement":null},{"id":"W4376460726","doi":"10.1007/978-3-031-29937-7_9","title":"Language Corpora and Principal Components Analysis","year":2023,"lang":"en","type":"book-chapter","venue":"Studies in big data","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dawson College; Cegep de Thetford; Cegep de Trois-Rivieres; College Ahuntsic; Université du Québec à Montréal; Memorial University of Newfoundland","funders":"","keywords":"Principal (computer security); Computer science; Natural language processing; Linguistics; Principal component analysis; Artificial intelligence; History; Philosophy; Computer security","score_opus":0.2423355936847015,"score_gpt":0.37504984459207,"score_spread":0.13271425090736852,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4376460726","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0031006858,0.06775218,0.73982847,0.006445093,0.0031160235,0.00016738276,0.0035508152,0.005604528,0.17043485],"genre_scores_gemma":[0.054691974,0.06493549,0.6701337,0.0014780774,0.004667364,0.0009221291,0.00956269,0.0047510257,0.18885753],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9984515,0.0006938508,0.00008825296,0.00017501888,0.000549884,0.000041570707],"domain_scores_gemma":[0.9955037,0.0031973505,0.00013977397,0.0005383386,0.00057010486,0.00005069892],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018314046,0.0011501384,0.0009261122,0.004024152,0.0009692273,0.0036165614,0.0012170845,0.0009035462,0.024476714],"category_scores_gemma":[0.009469366,0.00087836,0.00050551654,0.010942797,0.0017427024,0.004048512,0.0013420538,0.0017732658,0.011756378],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000023067543,0.000035942943,0.0003040685,0.000554255,0.00004222028,0.000080356614,0.0002912974,0.0035699834,0.0006012446,0.2740839,0.17708535,0.5433283],"study_design_scores_gemma":[0.000007894937,0.000011496643,0.0014562159,0.00028677518,0.000023446846,0.0002043044,0.00025013307,0.013691432,0.0014558686,0.5331755,0.44939992,0.00003699854],"about_ca_topic_score_codex":0.0025000991,"about_ca_topic_score_gemma":0.0041665845,"teacher_disagreement_score":0.024476714,"about_ca_system_score_codex":0.0008433215,"about_ca_system_score_gemma":0.0010631983,"threshold_uncertainty_score":0.081882715},"labels":[],"label_agreement":null},{"id":"W4376460846","doi":"10.1007/978-3-031-29937-7_13","title":"Pedagogical and Future Implications for the Training of Data Translators","year":2023,"lang":"en","type":"book-chapter","venue":"Studies in big data","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University; University of Toronto","funders":"","keywords":"Training (meteorology); Computer science; Psychology; Geography; Meteorology","score_opus":0.659167278204882,"score_gpt":0.4777002430008827,"score_spread":0.18146703520399926,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4376460846","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0053412663,0.019077972,0.059237342,0.7630442,0.007894287,0.00016570895,0.00033377155,0.000741016,0.14416456],"genre_scores_gemma":[0.1698521,0.04458943,0.37430286,0.09087923,0.008837052,0.0017817921,0.0010723668,0.0011352698,0.3075499],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9926845,0.0052587446,0.00020303782,0.0005186521,0.00080613163,0.0005288929],"domain_scores_gemma":[0.9348833,0.044207685,0.0011014128,0.003068486,0.007954699,0.008784329],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.018738344,0.0006563647,0.0005196698,0.0013840075,0.0040004244,0.009562275,0.004154746,0.0062548695,0.06084508],"category_scores_gemma":[0.03900592,0.0004680962,0.00054815545,0.0022154471,0.005270056,0.015504441,0.005386371,0.006925075,0.010857606],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014914446,0.0010726462,0.0015260738,0.0009598702,0.000010039744,0.00047168287,0.008693549,0.0007564259,0.0013765407,0.3657211,0.28868377,0.33057916],"study_design_scores_gemma":[0.00008042799,0.00014810546,0.0012882403,0.002211506,0.000015885993,0.0006055957,0.021481885,0.0018280954,0.0017263663,0.37602407,0.59453446,0.000055344863],"about_ca_topic_score_codex":0.0040104417,"about_ca_topic_score_gemma":0.0077206413,"teacher_disagreement_score":0.06084508,"about_ca_system_score_codex":0.003819988,"about_ca_system_score_gemma":0.011145428,"threshold_uncertainty_score":0.20354706},"labels":[],"label_agreement":null},{"id":"W4376652178","doi":"10.18653/v1/2023.mwe-1.13","title":"Are Frequent Phrases Directly Retrieved like Idioms? An Investigation with Self-Paced Reading and Language Models","year":2023,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Hong Kong Polytechnic University; Agence Nationale de la Recherche; Aix-Marseille Université","keywords":"Computer science; Natural language processing; Reading (process); Set (abstract data type); Artificial intelligence; Lexicon; Word (group theory); Reading comprehension; Comprehension; Speech recognition; Linguistics","score_opus":0.03403718456442428,"score_gpt":0.27943491006868953,"score_spread":0.24539772550426525,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4376652178","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.97060174,0.00015776443,0.026023366,0.00015924704,0.000023626342,0.000068823065,0.00014055792,0.00027695706,0.0025479696],"genre_scores_gemma":[0.9857071,0.00008736311,0.013216704,0.000047604386,0.000014763546,0.00003655114,0.00016206442,0.00011582477,0.0006120224],"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99809307,0.00085228356,0.00010206283,0.0006035825,0.0002865061,0.00006248551],"domain_scores_gemma":[0.9593071,0.0324648,0.00313858,0.0034048532,0.001319595,0.00036513418],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004292924,0.0005276901,0.00060895574,0.00074834883,0.0002337669,0.0020553234,0.00075115956,0.0007623992,0.0026384399],"category_scores_gemma":[0.049193013,0.0005673917,0.0005522278,0.0007105854,0.0009570455,0.004612457,0.0007869239,0.0011142301,0.0005897558],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0039545093,0.0014365654,0.15248086,0.0015105766,0.00068221596,0.0017080408,0.027745679,0.034054082,0.4471252,0.024328906,0.0022816584,0.30269173],"study_design_scores_gemma":[0.00027977864,0.002156321,0.26215997,0.00010519572,0.00037541203,0.003224278,0.0062173475,0.5304709,0.13636997,0.053530592,0.00478356,0.0003266539],"about_ca_topic_score_codex":0.00076036516,"about_ca_topic_score_gemma":0.0005247636,"teacher_disagreement_score":0.004292924,"about_ca_system_score_codex":0.00036288815,"about_ca_system_score_gemma":0.00028996228,"threshold_uncertainty_score":0.02270341},"labels":[],"label_agreement":null},{"id":"W4376854014","doi":"10.1007/978-3-031-31476-6_6","title":"Grammar Induction for Under-Resourced Languages: The Case of Ch’ol","year":2023,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Programming language; Grammar; Natural language processing; Artificial intelligence; Linguistics; Philosophy","score_opus":0.02621857760473584,"score_gpt":0.29315493188021957,"score_spread":0.26693635427548373,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4376854014","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.043640167,0.00036694374,0.9023417,0.0017931013,0.000119297445,0.000099618126,0.0004272373,0.004164973,0.047046863],"genre_scores_gemma":[0.5660883,0.0003493432,0.40307948,0.0008299144,0.00016854529,0.00012431401,0.0012560216,0.0029748636,0.025129208],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9977652,0.00074136956,0.00012589233,0.00045056138,0.000566405,0.00035063297],"domain_scores_gemma":[0.9871903,0.008704921,0.00035423238,0.0024134086,0.0010550644,0.0002820919],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026604407,0.00039456828,0.0008187063,0.001033877,0.001358099,0.0022588305,0.002246079,0.000924231,0.009391468],"category_scores_gemma":[0.0080940705,0.00071513874,0.0011573789,0.001346111,0.0039411834,0.0057372763,0.005175496,0.0044370964,0.002643103],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015140328,0.0000824818,0.0016402464,0.00035524965,0.000039491013,0.0014084299,0.001962065,0.008954947,0.013401179,0.8369295,0.010642678,0.12443226],"study_design_scores_gemma":[0.000027647744,0.000018951887,0.0003939535,0.000048383106,0.00003839124,0.00050764915,0.00049241347,0.058615897,0.010616907,0.90948427,0.019723922,0.00003172957],"about_ca_topic_score_codex":0.002973774,"about_ca_topic_score_gemma":0.004654423,"teacher_disagreement_score":0.009391468,"about_ca_system_score_codex":0.0014004309,"about_ca_system_score_gemma":0.002096605,"threshold_uncertainty_score":0.03141755},"labels":[],"label_agreement":null},{"id":"W4377220848","doi":"10.5334/joc.278","title":"A New Corpus of Lexical Substitution and Word Blend Errors: Probing the Semantic Structure of Lemma Access Failures","year":2023,"lang":"en","type":"article","venue":"Journal of Cognition","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Simon Fraser University","keywords":"Lemma (botany); Substitution (logic); Computer science; Natural language processing; Sentence; Artificial intelligence; Word (group theory); Set (abstract data type); Part of speech; Speech production; Selection (genetic algorithm); Linguistics; Speech recognition","score_opus":0.024519116892829273,"score_gpt":0.299843421022039,"score_spread":0.2753243041292097,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4377220848","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9673939,0.00090699823,0.0076512145,0.0003053694,0.00009272333,0.00020290335,0.017547827,0.0003083835,0.0055907136],"genre_scores_gemma":[0.92713654,0.00083104597,0.01850299,0.00020304183,0.00012104992,0.00067926326,0.048123244,0.00029814887,0.004104735],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9981431,0.0005116043,0.00031979158,0.00047228835,0.00047451816,0.00007861322],"domain_scores_gemma":[0.96461064,0.022180762,0.0027367927,0.0061282334,0.0033539655,0.000989709],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014421173,0.00041716083,0.00057227135,0.0032681387,0.0010915908,0.0011802422,0.0008362728,0.0008297402,0.003493189],"category_scores_gemma":[0.013731933,0.00029654487,0.00028074687,0.0032436794,0.001858749,0.0016940451,0.0023932583,0.0011745543,0.0013861115],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0024004318,0.0026217988,0.23430948,0.005445632,0.00051193015,0.009893249,0.07704973,0.0031031473,0.14784782,0.0098729655,0.056352925,0.45059082],"study_design_scores_gemma":[0.00025235693,0.00060321396,0.78121245,0.0003958608,0.00029671376,0.013888239,0.016252391,0.0075551933,0.038411725,0.005585597,0.13528769,0.000258617],"about_ca_topic_score_codex":0.0030895788,"about_ca_topic_score_gemma":0.00732089,"teacher_disagreement_score":0.003493189,"about_ca_system_score_codex":0.0003702475,"about_ca_system_score_gemma":0.00070497714,"threshold_uncertainty_score":0.011685908},"labels":[],"label_agreement":null},{"id":"W4377990179","doi":"10.59168/wtnf8464","title":"THE INTERNAL RULES OF THE EXAMPLE DATABASE DESIGN","year":2023,"lang":"en","type":"article","venue":"Professional Communication and Translation Studies","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Computer science; Representation (politics); Translation (biology); Romanian; Natural language processing; Verb; Quality (philosophy); Rule-based machine translation; Artificial intelligence; Linguistics; Information retrieval; Epistemology","score_opus":0.16789665044190247,"score_gpt":0.39469340767184635,"score_spread":0.22679675722994388,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4377990179","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00467278,0.00019315867,0.9747269,0.0011443399,0.00010238318,0.00048805852,0.00042339068,0.0012244441,0.01702444],"genre_scores_gemma":[0.10125076,0.00035731302,0.8840333,0.0007996203,0.00012851987,0.0020869053,0.0011023012,0.0011019801,0.009139196],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9776633,0.008634165,0.0031978188,0.0035182566,0.006032883,0.00095362175],"domain_scores_gemma":[0.97783685,0.008473495,0.000683672,0.00847,0.0040669492,0.00046905284],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01575217,0.00074077415,0.0010971805,0.0020250625,0.0019791564,0.014470405,0.0043414114,0.003389772,0.010907048],"category_scores_gemma":[0.044882864,0.0021578746,0.0021601242,0.002065299,0.007151068,0.0116057275,0.006022357,0.0048014037,0.0058711637],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000112839334,0.0000977872,0.0011150797,0.00029727817,0.000055206,0.00018836325,0.0015802786,0.0030178977,0.0034689475,0.91866976,0.0061643776,0.06523208],"study_design_scores_gemma":[0.00013895876,0.00012749639,0.00082378683,0.00047136523,0.00015402147,0.0011123074,0.0006812764,0.041908793,0.023148898,0.74090433,0.19039713,0.00013160467],"about_ca_topic_score_codex":0.0019796565,"about_ca_topic_score_gemma":0.001135702,"teacher_disagreement_score":0.01575217,"about_ca_system_score_codex":0.0017972899,"about_ca_system_score_gemma":0.0031509358,"threshold_uncertainty_score":0.08330643},"labels":[],"label_agreement":null},{"id":"W4378384185","doi":"10.31234/osf.io/54z93","title":"Formalizing theories of storage versus computation with Tree-based Grammars","year":2023,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Mila - Quebec Artificial Intelligence Institute; McGill University; Canadian Institute for Advanced Research","funders":"University of California, Davis; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research; National Science Foundation","keywords":"Rule-based machine translation; Computer science; Tree (set theory); Computation; Tree-adjoining grammar; Theoretical computer science; L-attributed grammar; Artificial intelligence; Programming language; Natural language processing; Mathematics; Context-free grammar; Combinatorics","score_opus":0.031159448674858394,"score_gpt":0.2980645078517889,"score_spread":0.2669050591769305,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4378384185","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015938174,0.00051234453,0.9746312,0.0010812454,0.000065549546,0.000035010475,0.00014003561,0.0003724261,0.0072239796],"genre_scores_gemma":[0.4580284,0.0015188063,0.53218377,0.00075908523,0.0002345137,0.00031202525,0.0007228437,0.00069091085,0.0055497373],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9981329,0.00061918964,0.00014829644,0.00038453945,0.00053448486,0.0001806954],"domain_scores_gemma":[0.9932636,0.0039670775,0.00040899328,0.0016430749,0.00052709156,0.00019011556],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0038138074,0.0007685075,0.0007874986,0.0016494053,0.001220165,0.004812734,0.0029667162,0.0019163431,0.005579941],"category_scores_gemma":[0.012424686,0.0007501663,0.0022041658,0.0019512827,0.00831358,0.018662686,0.0041135144,0.0030783417,0.001005882],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000009287808,0.000007463953,0.00014611516,0.000038135397,0.0000086119235,0.00004426408,0.0002996545,0.005559184,0.0004160493,0.9882934,0.000306753,0.0048710676],"study_design_scores_gemma":[0.0000050961994,0.00000442052,0.000029668443,0.000019682002,0.000004919379,0.00003152105,0.000052494244,0.016954454,0.00032194154,0.98124576,0.0013238544,0.00000625323],"about_ca_topic_score_codex":0.0023236996,"about_ca_topic_score_gemma":0.002216824,"teacher_disagreement_score":0.005579941,"about_ca_system_score_codex":0.0019449708,"about_ca_system_score_gemma":0.0017736617,"threshold_uncertainty_score":0.020169556},"labels":[],"label_agreement":null},{"id":"W4378446370","doi":"10.48550/arxiv.2305.13658","title":"Understanding Compositional Data Augmentation in Typologically Diverse Morphological Inflection","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Spurious relationship; Morpheme; Computer science; Inflection; Generalization; Compositional data; Natural language processing; Phonology; Phonotactics; Artificial intelligence; Selection (genetic algorithm); Linguistics; Machine learning; Mathematics","score_opus":0.4774692118771993,"score_gpt":0.284890306832388,"score_spread":0.19257890504481134,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4378446370","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.35482118,0.0014756785,0.6339263,0.0010527307,0.00012226484,0.00016699627,0.0016681975,0.0036146885,0.0031519893],"genre_scores_gemma":[0.63356274,0.00035468102,0.3591388,0.0002806053,0.00006428264,0.00027233615,0.005004337,0.00028878084,0.001033529],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99791104,0.0010809537,0.00012913013,0.0005049063,0.00029502227,0.00007893387],"domain_scores_gemma":[0.9876223,0.0082608685,0.00055733667,0.0026334266,0.0007774898,0.00014851266],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0048598032,0.0008197918,0.0005954025,0.001467124,0.00065366726,0.0020102207,0.001518657,0.0013400798,0.0016482491],"category_scores_gemma":[0.02042377,0.000497718,0.0010588302,0.0013677712,0.0014694604,0.0032963944,0.0030068962,0.0019525217,0.0009211681],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011464745,0.00057372515,0.04392566,0.0009970985,0.0003007382,0.0006943889,0.0033882826,0.11432354,0.065059766,0.02069858,0.007345121,0.7415467],"study_design_scores_gemma":[0.00009861256,0.0003578723,0.013888972,0.0001765277,0.00011052997,0.0007229468,0.0014015444,0.84253424,0.04034338,0.084171735,0.016123679,0.00006983534],"about_ca_topic_score_codex":0.0007830301,"about_ca_topic_score_gemma":0.0023045933,"teacher_disagreement_score":0.0048598032,"about_ca_system_score_codex":0.00041512717,"about_ca_system_score_gemma":0.00074894953,"threshold_uncertainty_score":0.025701404},"labels":[],"label_agreement":null},{"id":"W4378508602","doi":"10.48550/arxiv.2305.13989","title":"MasakhaPOS: Part-of-Speech Tagging for Typologically Diverse African Languages","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"DeepMind; International Development Research Centre; Rockefeller Foundation","keywords":"Computer science; Conditional random field; Natural language processing; Artificial intelligence; Field (mathematics); Transfer (computing); Baseline (sea); Mathematics","score_opus":0.11606710875848353,"score_gpt":0.24572287669611795,"score_spread":0.12965576793763442,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4378508602","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.38945475,0.0020997438,0.053728327,0.0012246459,0.00085409556,0.00060001644,0.50886285,0.01967593,0.023499567],"genre_scores_gemma":[0.24696438,0.0005151677,0.07336514,0.0004152351,0.00011001763,0.0011097048,0.670683,0.0013237878,0.005513563],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990792,0.00024323007,0.00010472445,0.0003080798,0.00014229884,0.00012244954],"domain_scores_gemma":[0.9985697,0.00047660441,0.00017321791,0.00042443656,0.00021348555,0.00014256332],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012567551,0.0009539683,0.00047104748,0.00316393,0.0015377582,0.00095886894,0.0009843056,0.0008444014,0.007711974],"category_scores_gemma":[0.0030226805,0.0003957768,0.0005854791,0.0026797631,0.00054580625,0.002146319,0.0028906653,0.0010527701,0.0055043804],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00355805,0.0006234563,0.08546638,0.004516135,0.0004109147,0.0029894398,0.00624775,0.008364357,0.08542023,0.016396549,0.41416746,0.37183928],"study_design_scores_gemma":[0.00048382595,0.00037642944,0.13258953,0.00058210193,0.00021618215,0.0030330506,0.007379992,0.03582497,0.052894592,0.02361453,0.74274087,0.00026387273],"about_ca_topic_score_codex":0.00575401,"about_ca_topic_score_gemma":0.011213626,"teacher_disagreement_score":0.007711974,"about_ca_system_score_codex":0.0005502832,"about_ca_system_score_gemma":0.001247899,"threshold_uncertainty_score":0.025799155},"labels":[],"label_agreement":null},{"id":"W4378603662","doi":"10.1007/978-3-031-33231-9_2","title":"A Parsing Tool for Short Linguistic Constructions","year":2023,"lang":"en","type":"book-chapter","venue":"Communications in computer and information science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of New Brunswick","funders":"","keywords":"Parsing; Linguistics; Computer science; Natural language processing; Artificial intelligence; Philosophy","score_opus":0.04977500912787741,"score_gpt":0.33312790076413007,"score_spread":0.28335289163625266,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4378603662","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0024614604,0.00042752235,0.8857943,0.0003684243,0.00027220833,0.00015103963,0.0051599224,0.08861632,0.016748875],"genre_scores_gemma":[0.046853635,0.0009564436,0.8576975,0.0005372108,0.00028276237,0.0006094796,0.021816209,0.036586486,0.03466037],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9990477,0.00021600613,0.0001404346,0.00022878373,0.00029586433,0.000071278424],"domain_scores_gemma":[0.9969297,0.0019382306,0.000117143696,0.0004547338,0.00049212895,0.000067983514],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010643323,0.0021563494,0.0015808496,0.0040279645,0.0015561596,0.0035846992,0.0022397884,0.0015385901,0.067725],"category_scores_gemma":[0.0045304205,0.0019312229,0.0017885538,0.0052614515,0.0010736719,0.00655142,0.003057254,0.0026670534,0.029387072],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001959767,0.00009185129,0.00045121263,0.00081055117,0.00006413146,0.00055269885,0.0009430964,0.0022503503,0.0122857755,0.13704711,0.22469093,0.6206163],"study_design_scores_gemma":[0.000102963786,0.00006369858,0.00057641923,0.0004978104,0.00017581505,0.0014631462,0.00039709976,0.07257588,0.04083334,0.27159998,0.61153626,0.00017763117],"about_ca_topic_score_codex":0.001426957,"about_ca_topic_score_gemma":0.0016221907,"teacher_disagreement_score":0.067725,"about_ca_system_score_codex":0.00073487323,"about_ca_system_score_gemma":0.0013243607,"threshold_uncertainty_score":0.22656268},"labels":[],"label_agreement":null},{"id":"W4378619943","doi":"10.1017/s0305000923000272","title":"Realistic and broad-scope learning simulations: first results and challenges","year":2023,"lang":"en","type":"article","venue":"Journal of Child Language","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Agence de l'innovation de Défense; Grand Équipement National De Calcul Intensif; Agence Nationale de la Recherche; École des Hautes Etudes en Sciences Sociales; Canadian Institute for Advanced Research","keywords":"Language acquisition; Scope (computer science); Second-language acquisition; Computer science; Strengths and weaknesses; Fragmentation (computing); Cognitive science; Psychology; Linguistics; Mathematics education","score_opus":0.0178879742079047,"score_gpt":0.2774399470724877,"score_spread":0.259551972864583,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4378619943","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5965569,0.008859532,0.3061828,0.010648912,0.0005255235,0.0012982944,0.0017696975,0.0019341992,0.07222414],"genre_scores_gemma":[0.93151164,0.0025962044,0.059517175,0.0005933894,0.00008210327,0.00078909285,0.00085862994,0.0003834576,0.0036684205],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9975036,0.001625496,0.0001524255,0.00018362385,0.00041614796,0.00011864029],"domain_scores_gemma":[0.95023143,0.04362273,0.0007967802,0.0026857748,0.0019036606,0.0007596168],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035791819,0.0008300585,0.0011575833,0.00053018145,0.0007739172,0.002173916,0.0018909762,0.002258249,0.009349674],"category_scores_gemma":[0.04769327,0.000482297,0.0006663602,0.0007977988,0.002019578,0.005381052,0.0033629013,0.0027014408,0.0011408905],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013304859,0.002396474,0.009812301,0.0019143261,0.00024543435,0.0003212603,0.0040559536,0.7489975,0.004414093,0.13972133,0.010057967,0.07673288],"study_design_scores_gemma":[0.0002231441,0.0009016973,0.0019188386,0.00062780385,0.00007735881,0.00013285145,0.0020348344,0.8024905,0.006431156,0.17157607,0.0134604955,0.0001253095],"about_ca_topic_score_codex":0.0053729634,"about_ca_topic_score_gemma":0.0031348893,"teacher_disagreement_score":0.009349674,"about_ca_system_score_codex":0.0014456889,"about_ca_system_score_gemma":0.0010404349,"threshold_uncertainty_score":0.031277776},"labels":[],"label_agreement":null},{"id":"W4378776239","doi":"10.5281/zenodo.7990925","title":"VITALISE D5.2 Ethical application documents for JRA1","year":2022,"lang":"en","type":"report","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"Horizon 2020 Framework Programme","keywords":"Computer science","score_opus":0.04009720314079905,"score_gpt":0.3211434082122016,"score_spread":0.28104620507140254,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4378776239","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013330526,0.0022185773,0.07699323,0.06810151,0.010592333,0.03597334,0.039123926,0.006965664,0.74670094],"genre_scores_gemma":[0.051984366,0.0013764403,0.09128829,0.03671447,0.0019747159,0.058783103,0.026549343,0.0038834745,0.7274457],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.8861372,0.04635182,0.009867846,0.005996944,0.042997256,0.008648902],"domain_scores_gemma":[0.8057896,0.0694038,0.0059384913,0.026317123,0.08667028,0.0058807833],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0933077,0.0015873731,0.0012227712,0.0041086804,0.007085229,0.016464846,0.0044687595,0.021506362,0.08766863],"category_scores_gemma":[0.19010063,0.0017988896,0.001865996,0.0030208835,0.005072131,0.004648912,0.00615303,0.011093619,0.06045002],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00042704554,0.00036704604,0.0011382197,0.0005651511,0.000025434552,0.0006930008,0.002977339,0.0006243687,0.0035706083,0.123989455,0.8360183,0.029603856],"study_design_scores_gemma":[0.00012394048,0.00012633436,0.0017704249,0.00080304313,0.000016045522,0.00050906074,0.0015271669,0.0005560043,0.002566258,0.010182793,0.9817478,0.00007100632],"about_ca_topic_score_codex":0.02043775,"about_ca_topic_score_gemma":0.021960218,"teacher_disagreement_score":0.0933077,"about_ca_system_score_codex":0.011266849,"about_ca_system_score_gemma":0.052325938,"threshold_uncertainty_score":0.49346417},"labels":[],"label_agreement":null},{"id":"W4379055835","doi":"10.3765/amp.v10i0.5429","title":"The Calibrated Error-Driven Ranking Algorithm as a Solution to Oscillation in Antagonistic Constraints: A Necessary Bias for Algorithmic Learning of Kihnu Estonian","year":2023,"lang":"en","type":"article","venue":"Proceedings of the Annual Meetings on Phonology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Estonian; Computer science; Artificial intelligence; Vowel harmony; Harmony (color); Constraint (computer-aided design); Vowel; Natural language processing; Speech recognition; Machine learning; Mathematics; Linguistics","score_opus":0.018145551282050077,"score_gpt":0.28594924965403995,"score_spread":0.26780369837198986,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4379055835","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.079923145,0.00010377447,0.91469467,0.0005390276,0.000038655653,0.00006664966,0.00005500672,0.00038316246,0.004195984],"genre_scores_gemma":[0.7369382,0.00007296699,0.25818065,0.0004216987,0.00005539502,0.00023021617,0.00020357875,0.00018134441,0.00371591],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99814355,0.0007170427,0.00010531281,0.00051926234,0.0003707191,0.00014413317],"domain_scores_gemma":[0.99212694,0.0050315424,0.000702662,0.0010912806,0.00077884697,0.00026872702],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0039410824,0.0006231379,0.00096849026,0.0006055676,0.00066262734,0.0015971161,0.0025950924,0.0016569053,0.003285924],"category_scores_gemma":[0.022476794,0.0006263124,0.0005211701,0.0006135491,0.0016530842,0.0022632603,0.0025873862,0.0030216805,0.00038977884],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024589035,0.00017606844,0.010054451,0.00016806384,0.00014814605,0.00025302792,0.0006423601,0.5312603,0.0081599355,0.23289937,0.0029797049,0.21301268],"study_design_scores_gemma":[0.0000460197,0.00011055609,0.0012070294,0.000021538166,0.000015111435,0.00007202252,0.000042947715,0.9209829,0.0018012116,0.07441285,0.0012606715,0.000027059623],"about_ca_topic_score_codex":0.0019447524,"about_ca_topic_score_gemma":0.0035733415,"teacher_disagreement_score":0.0039410824,"about_ca_system_score_codex":0.0010344549,"about_ca_system_score_gemma":0.00167222,"threshold_uncertainty_score":0.020842731},"labels":[],"label_agreement":null},{"id":"W4379514227","doi":"10.1007/978-3-031-28819-7_40","title":"Deep Dive Machine Translation","year":2023,"lang":"en","type":"book-chapter","venue":"Cognitive technologies","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"HORIZON EUROPE Framework Programme; European Commission; Institute for Catastrophic Loss Reduction","keywords":"Machine translation; Computer science; Context (archaeology); Field (mathematics); State (computer science); Data science; Point (geometry); Knowledge management; Political science; Artificial intelligence; Geography","score_opus":0.035813533350221645,"score_gpt":0.276183911044384,"score_spread":0.24037037769416236,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4379514227","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010794356,0.020625316,0.64240813,0.003623142,0.0018091891,0.00013958488,0.0024157227,0.008508981,0.3096755],"genre_scores_gemma":[0.20979644,0.015760263,0.41411114,0.0021843514,0.0008480242,0.0002154711,0.009578133,0.0020587845,0.3454474],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9995454,0.000098623015,0.000026690848,0.000120322125,0.00016102413,0.00004800533],"domain_scores_gemma":[0.9993358,0.00026669135,0.000024217481,0.0001715359,0.00016411148,0.000037586324],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00051526324,0.0007550037,0.0005691675,0.0010823569,0.00084764895,0.0032883876,0.0011656194,0.0012814044,0.038568597],"category_scores_gemma":[0.0017608695,0.00039437838,0.00057787885,0.0016738536,0.0008813299,0.0031284154,0.0027327635,0.0019245021,0.02275344],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005870822,0.00005968121,0.00031414104,0.0006515942,0.000037534865,0.00020130139,0.00029192067,0.009385511,0.008721964,0.11337176,0.08454038,0.78236556],"study_design_scores_gemma":[0.000015583886,0.000060317976,0.00053065014,0.00032310357,0.000028665614,0.00061649084,0.00021778271,0.09127197,0.0131001435,0.21685258,0.676945,0.000037779588],"about_ca_topic_score_codex":0.0019045713,"about_ca_topic_score_gemma":0.004512536,"teacher_disagreement_score":0.038568597,"about_ca_system_score_codex":0.0010379654,"about_ca_system_score_gemma":0.0011385904,"threshold_uncertainty_score":0.1290248},"labels":[],"label_agreement":null},{"id":"W4379534564","doi":"10.21428/594757db.132dae7d","title":"RISC: Generating Realistic Synthetic Bilingual Insurance Contract","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Université Laval","funders":"Fonds de recherche du Québec – Nature et technologies; Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Automobile insurance; Natural language processing; Python (programming language); Artificial intelligence; Insurance policy; Class (philosophy); Programming language; Finance; Actuarial science; Business","score_opus":0.018825961230074666,"score_gpt":0.2935253874547359,"score_spread":0.2746994262246612,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4379534564","genre_codex":"dataset","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.109997995,0.000796025,0.101025246,0.0012139332,0.0006022584,0.0012529384,0.69954515,0.05792201,0.027644485],"genre_scores_gemma":[0.10362578,0.00019792565,0.120272376,0.00031821278,0.000041742624,0.001182129,0.7648602,0.0026487692,0.006852906],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99920815,0.00021859641,0.000055131008,0.00017227003,0.00027048533,0.00007537412],"domain_scores_gemma":[0.99826556,0.0006849181,0.00007526705,0.00034562274,0.00052037585,0.00010827453],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00090486184,0.001139026,0.00043665394,0.0016410789,0.00068445405,0.0006439632,0.0015585137,0.0010366337,0.011785038],"category_scores_gemma":[0.0042326404,0.000373396,0.00075836026,0.0018214892,0.00044951978,0.0007315397,0.0009476289,0.00091751583,0.005850429],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00077192555,0.0005490934,0.010992683,0.0016453177,0.0001787584,0.0016308496,0.0006380727,0.077348515,0.012798597,0.00975709,0.7443834,0.13930568],"study_design_scores_gemma":[0.00063363626,0.00023337512,0.011901274,0.00020465998,0.00006374489,0.0009603008,0.0010710314,0.39794347,0.03268265,0.015315909,0.5388334,0.00015653607],"about_ca_topic_score_codex":0.03704761,"about_ca_topic_score_gemma":0.065347016,"teacher_disagreement_score":0.03704761,"about_ca_system_score_codex":0.0015732873,"about_ca_system_score_gemma":0.002110248,"threshold_uncertainty_score":0.07366395},"labels":[],"label_agreement":null},{"id":"W4380051257","doi":"10.1075/kl.22003.par","title":"A role of functional morphemes in Korean categorial grammars","year":2023,"lang":"en","type":"article","venue":"Korean Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Morpheme; Rule-based machine translation; Linguistics; Natural language processing; Lexicon; Agglutinative language; Computer science; Part of speech; Focus (optics); Artificial intelligence","score_opus":0.014772063284333283,"score_gpt":0.2575518908485008,"score_spread":0.2427798275641675,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4380051257","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7477266,0.00083866745,0.19095077,0.000756867,0.00009315257,0.00012726877,0.0004418298,0.0007709871,0.058293827],"genre_scores_gemma":[0.9804817,0.00010032448,0.017734013,0.000049638795,0.000008209384,0.000018367791,0.00012534618,0.00010918384,0.0013732748],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99929726,0.00024632536,0.00009923869,0.00016873266,0.00010943452,0.00007894438],"domain_scores_gemma":[0.99780613,0.0009918956,0.00022055508,0.00048345787,0.00041792815,0.00008006684],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009622927,0.00030007173,0.00019437396,0.001285396,0.00076031004,0.0018137824,0.00058169843,0.00034194888,0.0020965233],"category_scores_gemma":[0.0021051173,0.00038115148,0.0005875553,0.0008321905,0.0021606805,0.0030253571,0.0011638469,0.00073658966,0.0003111321],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012792101,0.000046491292,0.024028234,0.00030366407,0.00005346483,0.0015902015,0.011955979,0.002372386,0.056309447,0.82741344,0.0009399048,0.074858874],"study_design_scores_gemma":[0.00005369556,0.0001997507,0.057413492,0.00028453727,0.00037481473,0.006334651,0.016333498,0.03795297,0.06728895,0.73346406,0.08000225,0.00029735162],"about_ca_topic_score_codex":0.0010038356,"about_ca_topic_score_gemma":0.001267257,"teacher_disagreement_score":0.0020965233,"about_ca_system_score_codex":0.00072867237,"about_ca_system_score_gemma":0.0005570801,"threshold_uncertainty_score":0.007013619},"labels":[],"label_agreement":null},{"id":"W4380482005","doi":"10.6000/1929-4409.2020.09.298","title":"Russian Size Adjectives Analyses by Corpus Linguistics Methods","year":2022,"lang":"en","type":"article","venue":"International Journal of Criminology and Sociology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"Kazan Federal University","keywords":"Adjective; Noun; Linguistics; Mathematics; Numeral system; Noun phrase; Parametric statistics; Semantics (computer science); Characterization (materials science); Natural language processing; Statistics; Computer science; Philosophy; Physics; Arithmetic","score_opus":0.0708498737276486,"score_gpt":0.43334887447594367,"score_spread":0.36249900074829505,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4380482005","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6519463,0.005232428,0.17802587,0.0014676773,0.00042920132,0.0014279204,0.03110214,0.0022483307,0.1281201],"genre_scores_gemma":[0.8162495,0.0020156929,0.14239483,0.00018574746,0.0001531881,0.0022653092,0.019985313,0.0007211125,0.016029388],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9969862,0.0014662216,0.00038261758,0.0006007488,0.00046903815,0.000095214025],"domain_scores_gemma":[0.9924074,0.004842731,0.0007302779,0.00060123036,0.0013388979,0.00007958739],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024819756,0.00046955622,0.00041131862,0.009666632,0.0019138432,0.0021429993,0.00047562303,0.00030331995,0.0070849],"category_scores_gemma":[0.007380847,0.0003435608,0.0004789259,0.0076542213,0.0012398555,0.001780043,0.0015288842,0.0006657512,0.0016789001],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004909945,0.00024275898,0.06310829,0.0060057817,0.0002636325,0.0024106158,0.14948009,0.0028126182,0.04746842,0.1601207,0.05049904,0.517097],"study_design_scores_gemma":[0.000068180736,0.00021385498,0.17590186,0.0009649735,0.0004119785,0.002591419,0.061913136,0.018807378,0.023142492,0.03376113,0.68204033,0.00018322137],"about_ca_topic_score_codex":0.005467397,"about_ca_topic_score_gemma":0.00759862,"teacher_disagreement_score":0.009666632,"about_ca_system_score_codex":0.0015043318,"about_ca_system_score_gemma":0.0016746242,"threshold_uncertainty_score":0.02370137},"labels":[],"label_agreement":null},{"id":"W4380627553","doi":"10.29173/bluejay6353","title":"Magpies Roost at Saskatoon in Record Numbers","year":2023,"lang":"en","type":"article","venue":"Blue Jay","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Geography","score_opus":0.013397112585110662,"score_gpt":0.26637907313798626,"score_spread":0.2529819605528756,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4380627553","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.39782187,0.0005764514,0.0037256337,0.009052431,0.004177507,0.00020891488,0.014100946,0.0022214858,0.5681147],"genre_scores_gemma":[0.23583648,0.00026511188,0.0031838864,0.00090229855,0.000072283045,0.00004308387,0.0039230376,0.00020051413,0.75557333],"study_design_codex":"not_applicable","study_design_gemma":"observational","domain_scores_codex":[0.9998964,0.000009673786,0.0000030539916,0.000021699841,0.000028381337,0.000040697483],"domain_scores_gemma":[0.99968815,0.000029361105,0.000026847898,0.00002077357,0.000102920465,0.00013191036],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00015657827,0.00023844575,0.00014624151,0.00062137534,0.0028018488,0.0008345171,0.00037095128,0.0004835784,0.07640521],"category_scores_gemma":[0.00053833134,0.00013193852,0.000102704325,0.00070815545,0.00046341016,0.0004944319,0.0006976061,0.00075083395,0.01144701],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010319888,0.00011106273,0.073735885,0.00027221077,0.000048260823,0.006828421,0.012339765,0.000576198,0.024908092,0.011463443,0.6305009,0.23818369],"study_design_scores_gemma":[0.000011653383,0.00005953615,0.044923224,0.00007075731,0.000010597986,0.00043811288,0.012261973,0.00027061265,0.0020599794,0.0004180425,0.9394586,0.000016822754],"about_ca_topic_score_codex":0.19130535,"about_ca_topic_score_gemma":0.7178245,"teacher_disagreement_score":0.80869466,"about_ca_system_score_codex":0.0018416712,"about_ca_system_score_gemma":0.0020365366,"threshold_uncertainty_score":0.3803836},"labels":[],"label_agreement":null},{"id":"W4380905397","doi":"10.1007/978-3-031-35254-6_13","title":"Introducing Prolog in Language-Informed Ways","year":2023,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Prolog; Computer science; Grammar; Programming language; Subject (documents); Artificial intelligence; Position paper; Natural language processing; Cognitive science; Linguistics; World Wide Web; Psychology","score_opus":0.015853041232005247,"score_gpt":0.2721283126612512,"score_spread":0.25627527142924594,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4380905397","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0010994249,0.0019935267,0.9381631,0.0030372995,0.0008035342,0.000077722456,0.00014587199,0.0014307064,0.053248826],"genre_scores_gemma":[0.050577722,0.00532038,0.88613033,0.0024261172,0.00069409364,0.0002465222,0.00048140835,0.0018047676,0.05231872],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99836177,0.0008281774,0.00012275923,0.00019255745,0.00037984218,0.00011494918],"domain_scores_gemma":[0.99849546,0.000960822,0.00007385565,0.00023233694,0.0001609182,0.00007663822],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020293987,0.0010215738,0.00056925777,0.0010991519,0.0011218704,0.0066773896,0.0016421622,0.001250628,0.0155453775],"category_scores_gemma":[0.005138383,0.0011463845,0.0011091343,0.001322385,0.002991691,0.011083966,0.003854402,0.0062496285,0.00646345],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000031735362,0.00003214303,0.000074235126,0.00033425487,0.000017814169,0.00016193595,0.0012022399,0.0022048992,0.0013584617,0.9012966,0.014149249,0.079136476],"study_design_scores_gemma":[0.000013018746,0.000016811317,0.00002828105,0.00019365428,0.000015304428,0.00017089995,0.00023165194,0.004250454,0.0016087188,0.61292434,0.38052508,0.000021839567],"about_ca_topic_score_codex":0.0010126574,"about_ca_topic_score_gemma":0.0017844101,"teacher_disagreement_score":0.0155453775,"about_ca_system_score_codex":0.0011451844,"about_ca_system_score_gemma":0.0012358587,"threshold_uncertainty_score":0.052004457},"labels":[],"label_agreement":null},{"id":"W4381325804","doi":"10.7551/mitpress/7941.003.0016","title":"Semisupervised Learning for Machine Translation","year":2008,"lang":"en","type":"book-chapter","venue":"The MIT Press eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Advanced Research Projects Agency; Defense Advanced Research Projects Agency","keywords":"Translation (biology); Computer science; Artificial intelligence; Natural language processing; Machine translation; Machine learning; Chemistry","score_opus":0.04462626676855649,"score_gpt":0.2629182203512099,"score_spread":0.21829195358265338,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4381325804","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0011386299,0.0029152373,0.98617506,0.00050774775,0.00024620246,0.000038474176,0.0004195526,0.004106226,0.004452858],"genre_scores_gemma":[0.051906344,0.0037133237,0.90502715,0.0003576486,0.0007442465,0.0004925455,0.0053265784,0.00211284,0.030319193],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9979456,0.0009970488,0.00014092679,0.00040555248,0.00044924938,0.00006163609],"domain_scores_gemma":[0.9953956,0.0026225538,0.0001803981,0.0011455931,0.00060881895,0.000047003457],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017536551,0.001484335,0.0019995854,0.0013379278,0.0008016247,0.0019106632,0.0022341465,0.0017581299,0.01451708],"category_scores_gemma":[0.0063830228,0.0010788846,0.0011131836,0.0022678075,0.0010175558,0.004253577,0.0023209983,0.0027980315,0.013033088],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007732573,0.00009301166,0.00019622147,0.000498583,0.0000898669,0.00010257221,0.00010824738,0.023691898,0.0026135915,0.053990394,0.08420798,0.8343304],"study_design_scores_gemma":[0.000027699884,0.000048533835,0.00035748965,0.00011592627,0.00003486878,0.00031150755,0.000049936167,0.5778485,0.0074676373,0.3459326,0.06775588,0.00004941727],"about_ca_topic_score_codex":0.0013733304,"about_ca_topic_score_gemma":0.0026938196,"teacher_disagreement_score":0.01451708,"about_ca_system_score_codex":0.000756168,"about_ca_system_score_gemma":0.00092787325,"threshold_uncertainty_score":0.048564434},"labels":[],"label_agreement":null},{"id":"W4381488486","doi":"10.21248/jlcl.36.2023.243","title":"Kencorpus: A Kenyan Language Corpus of Swahili, Dholuo and Luhya for Natural Language Processing Tasks","year":2023,"lang":"en","type":"article","venue":"LDV-Forum/Journal for language technology and computational linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"International Development Research Centre; Rockefeller Foundation","keywords":"Swahili; Computer science; Natural language processing; Artificial intelligence; Machine translation; Question answering; Linguistics","score_opus":0.008866185083867504,"score_gpt":0.29924421999120293,"score_spread":0.29037803490733544,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4381488486","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1440764,0.0022672478,0.0042050406,0.00084871316,0.00036585925,0.0013184801,0.82031715,0.0015789879,0.025022069],"genre_scores_gemma":[0.08477946,0.0006837235,0.014977725,0.00028742861,0.000054617994,0.0027446556,0.8888048,0.0003857023,0.0072819134],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9991762,0.0001955206,0.00011472336,0.00023962562,0.00015898715,0.00011485911],"domain_scores_gemma":[0.99861157,0.0005480606,0.00013201912,0.00016486818,0.00040348643,0.00013999068],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00072349457,0.0009770859,0.0007365338,0.0035903018,0.0028434086,0.0011878755,0.0010270723,0.001260876,0.016605733],"category_scores_gemma":[0.0030812523,0.0005280144,0.00042456336,0.00418424,0.00093500176,0.0017482418,0.0019627102,0.0010465388,0.007846489],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00146657,0.0006637657,0.025109356,0.011705385,0.00023828456,0.006630379,0.019018976,0.0015632128,0.042249378,0.0077497014,0.7238275,0.15977758],"study_design_scores_gemma":[0.00023589123,0.00010022834,0.084159344,0.00082121376,0.00009326988,0.0016435534,0.0077654705,0.0017560578,0.009320053,0.0010898765,0.8928866,0.00012845638],"about_ca_topic_score_codex":0.036687646,"about_ca_topic_score_gemma":0.10713749,"teacher_disagreement_score":0.036687646,"about_ca_system_score_codex":0.0013146132,"about_ca_system_score_gemma":0.00336588,"threshold_uncertainty_score":0.07294822},"labels":[],"label_agreement":null},{"id":"W4381687028","doi":"10.1162/opmi_a_00086","title":"The Plausibility of Sampling as an Algorithmic Theory of Sentence Processing","year":2023,"lang":"en","type":"article","venue":"Open Mind","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Institute for Advanced Research; McGill University; Mila - Quebec Artificial Intelligence Institute","funders":"Alliance de recherche numérique du Canada","keywords":"Computer science; Variance (accounting); Sampling (signal processing); Context (archaeology); Reading (process); Grammaticality; Artificial intelligence; Class (philosophy); Sentence; Natural language processing; Linguistics; Grammar","score_opus":0.06435273318341708,"score_gpt":0.3757974606647919,"score_spread":0.31144472748137486,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4381687028","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.065826625,0.0010190828,0.9203325,0.0037323562,0.00014403307,0.00012376088,0.00025135794,0.0011023,0.007468055],"genre_scores_gemma":[0.7497344,0.0009385556,0.24247538,0.0014717218,0.00057006977,0.000370697,0.0004998002,0.00048372865,0.0034557176],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99468094,0.0021546497,0.00021639679,0.0014139226,0.0012522497,0.00028174705],"domain_scores_gemma":[0.9239379,0.061893806,0.0035173355,0.0077283704,0.0021324528,0.0007900854],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009266266,0.0011522252,0.001383908,0.0021238795,0.0014832894,0.0043084524,0.0028724966,0.0021622304,0.005034718],"category_scores_gemma":[0.059334572,0.0011102612,0.0018903184,0.0020054155,0.006168793,0.015029443,0.0019307119,0.005088894,0.00094512955],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00069717946,0.00039027224,0.019315429,0.0005424515,0.0003996624,0.0003361969,0.0013295761,0.19414242,0.006669917,0.66633886,0.0044467766,0.10539121],"study_design_scores_gemma":[0.00003598468,0.00016581267,0.002568196,0.000030855448,0.00005064148,0.00022277025,0.00006896975,0.39500195,0.0021716768,0.5980231,0.0016115154,0.000048524616],"about_ca_topic_score_codex":0.0025270935,"about_ca_topic_score_gemma":0.0026148504,"teacher_disagreement_score":0.009266266,"about_ca_system_score_codex":0.0026179997,"about_ca_system_score_gemma":0.0018924142,"threshold_uncertainty_score":0.04900533},"labels":[],"label_agreement":null},{"id":"W4382202654","doi":"10.1609/aaai.v37i11.26623","title":"A Graph Fusion Approach for Cross-Lingual Machine Reading Comprehension","year":2023,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Natural Sciences and Engineering Research Council of Canada; Natural Science Foundation of Guangdong Province; National Natural Science Foundation of China","keywords":"Computer science; Leverage (statistics); Syntax; Natural language processing; Artificial intelligence; Machine translation; Graph; Benchmark (surveying); Rule-based machine translation; Knowledge graph; Comprehension; Theoretical computer science; Programming language","score_opus":0.08442248157411145,"score_gpt":0.34648846138502365,"score_spread":0.2620659798109122,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4382202654","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013175587,0.0007939536,0.97307354,0.00050871744,0.000100204124,0.00015249332,0.0005725118,0.008557894,0.0030651293],"genre_scores_gemma":[0.29984742,0.0009225904,0.68142325,0.00068028976,0.00023589938,0.0004120214,0.007481789,0.0012649684,0.0077316454],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9981798,0.00059663487,0.00008257209,0.00076014997,0.00026130828,0.00011951181],"domain_scores_gemma":[0.99703634,0.0013201286,0.00017242822,0.0006770476,0.0006908109,0.00010327096],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020272348,0.002071777,0.0016165137,0.0044843596,0.0012764375,0.0013770832,0.0030236433,0.0027421687,0.0059264298],"category_scores_gemma":[0.0060994322,0.0007653072,0.0023507231,0.0036142236,0.0011446893,0.005312752,0.003237432,0.0031536585,0.0036408096],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025389154,0.00042486127,0.0016363796,0.00042011778,0.00028037542,0.0005234382,0.00086469506,0.08325958,0.020405356,0.02482302,0.020655649,0.8464527],"study_design_scores_gemma":[0.00003737815,0.00014802883,0.0011928184,0.00004033879,0.00011034781,0.0001563918,0.00027356678,0.88856,0.008306906,0.090745255,0.010377167,0.000051670515],"about_ca_topic_score_codex":0.008603383,"about_ca_topic_score_gemma":0.012245582,"teacher_disagreement_score":0.008603383,"about_ca_system_score_codex":0.0015454405,"about_ca_system_score_gemma":0.0016839156,"threshold_uncertainty_score":0.019825876},"labels":[],"label_agreement":null},{"id":"W4382463985","doi":"10.1609/aaai.v37i13.27068","title":"NL2LTL – a Python Package for Converting Natural Language (NL) Instructions to Linear Temporal Logic (LTL) Formulas","year":2023,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":37,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Computer science; Programming language; Python (programming language); Natural language; IBM; Linear temporal logic; Temporal logic; Extensibility; Natural language processing","score_opus":0.0672773486059246,"score_gpt":0.34221820759602817,"score_spread":0.27494085899010356,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4382463985","genre_codex":"software","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":"software","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.001251842,0.00015063421,0.4082437,0.0006633846,0.00029576343,0.0003098788,0.030228304,0.53656125,0.022295322],"genre_scores_gemma":[0.042971775,0.0006527401,0.5507982,0.0024097515,0.00023163106,0.0020395669,0.09633682,0.26462352,0.039936084],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99877983,0.00018622148,0.00013496538,0.00022479144,0.00051122083,0.00016283132],"domain_scores_gemma":[0.99826366,0.0006551216,0.00016639382,0.00035743206,0.00043479362,0.00012258398],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013379934,0.0016682325,0.0008883063,0.001587959,0.00084274873,0.002436793,0.003223037,0.0012151079,0.13830474],"category_scores_gemma":[0.006684301,0.0014945199,0.0023802288,0.0011629643,0.0011493794,0.0041272757,0.0032979602,0.0043596923,0.07990703],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00031683812,0.0001482295,0.0014051693,0.0015293424,0.00011134132,0.00050131,0.00048411786,0.0062636393,0.008838912,0.04810813,0.8066413,0.12565166],"study_design_scores_gemma":[0.0002252279,0.00005359451,0.0011437417,0.0002932999,0.00005211429,0.00061479793,0.00011452856,0.08700969,0.019044364,0.0919988,0.7992631,0.00018669396],"about_ca_topic_score_codex":0.005897383,"about_ca_topic_score_gemma":0.007933215,"teacher_disagreement_score":0.13830474,"about_ca_system_score_codex":0.0014362412,"about_ca_system_score_gemma":0.0032346915,"threshold_uncertainty_score":0.46267545},"labels":[],"label_agreement":null},{"id":"W4383334972","doi":"10.2139/ssrn.4487105","title":"A Scoping Review to Identify Candidate Quality Indicators of Tools that Support the Practice of Knowledge Translation","year":2023,"lang":"en","type":"review","venue":"SSRN Electronic Journal","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia; University of Calgary; University of Ottawa; University of Toronto; University Health Network; McMaster University","funders":"","keywords":"Knowledge translation; Quality (philosophy); Translation (biology); Knowledge management; Computer science; Data science; Process management; Business; Epistemology; Biology","score_opus":0.10695718393119552,"score_gpt":0.47188268754573154,"score_spread":0.364925503614536,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4383334972","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0027430812,0.9754556,0.0074181072,0.00339976,0.0009360858,0.0054400982,0.0014058284,0.0001233244,0.0030781715],"genre_scores_gemma":[0.027944395,0.9266489,0.033071794,0.0016976446,0.00031505892,0.006967376,0.0025574225,0.00007207861,0.0007252276],"study_design_codex":"systematic_review","study_design_gemma":"systematic_review","domain_scores_codex":[0.9269385,0.020760762,0.030942582,0.002845178,0.01749649,0.0010164792],"domain_scores_gemma":[0.6805856,0.19485301,0.05068261,0.008222713,0.062624365,0.0030316715],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.10324871,0.0028303002,0.009626108,0.04967161,0.0029264078,0.010077259,0.0047509572,0.00456947,0.005109459],"category_scores_gemma":[0.2866933,0.0020657394,0.0095187845,0.039066747,0.0029489768,0.009968452,0.0064922906,0.0030158195,0.0012731914],"study_design_candidate":"systematic_review","study_design_consensus":"systematic_review","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005233702,0.00016001517,0.002754482,0.5774331,0.0050397036,0.00019596376,0.001264841,0.0004703139,0.0007211765,0.0036743504,0.00819364,0.39956903],"study_design_scores_gemma":[0.00021747836,0.00036517982,0.00420643,0.9226328,0.021187278,0.00023144248,0.00078264513,0.00043976356,0.0007920135,0.0023877935,0.046676464,0.00008074343],"about_ca_topic_score_codex":0.012931478,"about_ca_topic_score_gemma":0.029251166,"teacher_disagreement_score":0.8967513,"about_ca_system_score_codex":0.011962755,"about_ca_system_score_gemma":0.070737585,"threshold_uncertainty_score":0.5460379},"labels":[],"label_agreement":null},{"id":"W4384027748","doi":"10.1007/978-981-99-3604-5_4","title":"The Old Church Slavonic Corpora and Their Use in Language Studies at the University","year":2023,"lang":"en","type":"book-chapter","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Slavic languages; Computer science; Point (geometry); Process (computing); Set (abstract data type); Linguistics; Natural language processing; Artificial intelligence; Corpus linguistics; Programming language; Philosophy; Mathematics","score_opus":0.039166723903777526,"score_gpt":0.2549387203660978,"score_spread":0.21577199646232031,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4384027748","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04822316,0.043555796,0.051241897,0.01361599,0.0030852004,0.00021671278,0.009781497,0.0017533227,0.8285263],"genre_scores_gemma":[0.40706578,0.03328245,0.13679712,0.0020745448,0.0014314188,0.0005361006,0.01613864,0.0052122874,0.3974616],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9989231,0.0005137963,0.00007365511,0.00013932872,0.00029596203,0.00005406503],"domain_scores_gemma":[0.9956162,0.0027430851,0.00014974544,0.0006841473,0.0006362727,0.00017051777],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003085698,0.00030706637,0.00031426415,0.005778266,0.0029146885,0.0053913086,0.00081170344,0.0006526368,0.016946817],"category_scores_gemma":[0.007896425,0.00062571577,0.00015401097,0.009979339,0.0026780614,0.004605828,0.0017963295,0.0018864494,0.0030662885],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009013352,0.00006377749,0.0013600718,0.0004544366,0.000009420278,0.00019706074,0.008975463,0.0006723428,0.002603381,0.41277257,0.14664982,0.42615148],"study_design_scores_gemma":[0.00001202783,0.000016543016,0.003526324,0.0004236353,0.000007916757,0.00014022339,0.0020147716,0.0006647,0.0021116596,0.03510366,0.9559571,0.000021320846],"about_ca_topic_score_codex":0.014525104,"about_ca_topic_score_gemma":0.05262826,"teacher_disagreement_score":0.016946817,"about_ca_system_score_codex":0.0028514697,"about_ca_system_score_gemma":0.0024620444,"threshold_uncertainty_score":0.05669278},"labels":[],"label_agreement":null},{"id":"W4384664207","doi":"10.3758/s13428-023-02170-w","title":"LaDEP: A large database of English pseudo-compounds","year":2023,"lang":"en","type":"article","venue":"Behavior Research Methods","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Yorkville University; University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Database; Programming language; Information retrieval; Natural language processing; World Wide Web","score_opus":0.2067183506943114,"score_gpt":0.5583345911829365,"score_spread":0.35161624048862505,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4384664207","genre_codex":"empirical","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.53962475,0.0038626483,0.013947647,0.00027313692,0.0001498087,0.0012936629,0.42556912,0.002330936,0.012948263],"genre_scores_gemma":[0.20904088,0.0017239655,0.03695416,0.00033497738,0.00008267158,0.0020890683,0.742004,0.0006477303,0.0071226326],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99869305,0.0002601543,0.00042458714,0.00026044424,0.00029088504,0.00007101758],"domain_scores_gemma":[0.9949692,0.0021533326,0.00044438607,0.00093317055,0.0011985159,0.00030133067],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008996624,0.0010606732,0.0008736182,0.004707865,0.0007026929,0.0013300596,0.0014097905,0.0012124781,0.0128449295],"category_scores_gemma":[0.005451878,0.00042799278,0.00065582164,0.0048269196,0.00043555145,0.0022502749,0.0017913865,0.00069394696,0.009234064],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0060236566,0.0024403895,0.110394776,0.014069782,0.0008249609,0.016136523,0.0046544285,0.0024230466,0.07304693,0.003883352,0.1778705,0.5882316],"study_design_scores_gemma":[0.00064574246,0.0012387824,0.35906407,0.0006588919,0.00045264297,0.014101582,0.0044493214,0.005004043,0.03469329,0.0023205697,0.57700706,0.0003640305],"about_ca_topic_score_codex":0.001665786,"about_ca_topic_score_gemma":0.0027172782,"teacher_disagreement_score":0.0128449295,"about_ca_system_score_codex":0.00026525636,"about_ca_system_score_gemma":0.00075426913,"threshold_uncertainty_score":0.04297054},"labels":[],"label_agreement":null},{"id":"W4384928505","doi":"10.1075/jerpp.00013.hab","title":"AI-mediated English for research publication purposes","year":2023,"lang":"en","type":"article","venue":"Journal of English for Research Publication Purposes","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Computer science; Title page; World Wide Web; Library science; Information retrieval","score_opus":0.1410917601999685,"score_gpt":0.4604054938852358,"score_spread":0.3193137336852673,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4384928505","genre_codex":"editorial","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0010271624,0.026367046,0.017902154,0.21636745,0.5995282,0.00029701882,0.0007304005,0.0012356067,0.13654496],"genre_scores_gemma":[0.04076854,0.041846115,0.016837997,0.11295499,0.5099925,0.0010760752,0.0011996463,0.0030002159,0.27232403],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9368703,0.031856894,0.009657674,0.0031754507,0.016547672,0.0018920053],"domain_scores_gemma":[0.55387634,0.30471954,0.021857837,0.037801296,0.07061303,0.011131843],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04478059,0.0012534079,0.0015602346,0.0063105654,0.003597424,0.019085526,0.0038654779,0.007183656,0.104651116],"category_scores_gemma":[0.22620165,0.0006392953,0.0012255562,0.0037145885,0.008392639,0.01743961,0.007472453,0.009806168,0.038557302],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000092031725,0.00007731481,0.00011759155,0.0036761959,0.000057912748,0.00039006854,0.0021020565,0.000083163395,0.00210335,0.10390685,0.79409593,0.093297526],"study_design_scores_gemma":[0.000013018668,0.000022337266,0.00017534554,0.00089488545,0.00001305262,0.00016870713,0.0003114238,0.00013880461,0.0005110558,0.007783094,0.98994887,0.000019473253],"about_ca_topic_score_codex":0.00067162054,"about_ca_topic_score_gemma":0.0011292095,"teacher_disagreement_score":0.104651116,"about_ca_system_score_codex":0.003019518,"about_ca_system_score_gemma":0.0065416633,"threshold_uncertainty_score":0.3500929},"labels":[],"label_agreement":null},{"id":"W4385362683","doi":"10.1142/9789811234378_0054","title":"<i>Word Ways</i> Selections","year":2023,"lang":"he","type":"book-chapter","venue":"WORLD SCIENTIFIC eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Word (group theory); Linguistics; Computer science; Natural language processing; Philosophy","score_opus":0.04543467307747142,"score_gpt":0.2666365240787056,"score_spread":0.22120185100123416,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385362683","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00035131202,0.002298415,0.004231499,0.0014316767,0.0082575455,0.00007664839,0.0016242341,0.0012402507,0.9804883],"genre_scores_gemma":[0.0014416763,0.0011473657,0.0015187267,0.0006250431,0.0012349657,0.00005233916,0.0008617956,0.0015324027,0.9915856],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99972814,0.0000354024,0.0000144125315,0.000042448057,0.00015215795,0.000027391368],"domain_scores_gemma":[0.9992587,0.00018195168,0.000025695355,0.000071420116,0.00036144612,0.000100751655],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.00028765513,0.0015393368,0.0010535185,0.0029316633,0.0016196604,0.004690674,0.0010617747,0.00064824877,0.52309865],"category_scores_gemma":[0.0015119461,0.00045030378,0.0005336895,0.0036260744,0.00056008744,0.0038968264,0.0011671842,0.0016342883,0.35839722],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000025729132,0.000009545107,0.000017985894,0.00011435568,0.000001761135,0.000033704586,0.00007081408,0.000025629386,0.0006644946,0.0129963765,0.9294535,0.056586217],"study_design_scores_gemma":[0.0000028932502,0.0000062770405,0.000048620088,0.000048786365,0.0000017566316,0.000043257874,0.000050621155,0.000042111584,0.0002719334,0.0026600622,0.9968189,0.000004725036],"about_ca_topic_score_codex":0.0019155043,"about_ca_topic_score_gemma":0.005603595,"teacher_disagreement_score":0.52309865,"about_ca_system_score_codex":0.0008196461,"about_ca_system_score_gemma":0.00048513338,"threshold_uncertainty_score":0.6802419},"labels":[],"label_agreement":null},{"id":"W4385386391","doi":"10.18280/ria.370310","title":"Deep Neural Networks for Part-of-Speech Tagging in Under-Resourced Amazigh","year":2023,"lang":"en","type":"article","venue":"Revue d intelligence artificielle","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Artificial neural network; Artificial intelligence; Speech recognition","score_opus":0.04500213278044095,"score_gpt":0.3038118357281671,"score_spread":0.25880970294772615,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385386391","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.38346976,0.0024200308,0.5907564,0.00083140103,0.00027316698,0.00006838582,0.004054546,0.005704697,0.0124215605],"genre_scores_gemma":[0.8766582,0.0006241055,0.10994282,0.00022710953,0.00003870917,0.00006370887,0.005645071,0.00020827945,0.006591972],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9996824,0.00008662509,0.00002277552,0.00010454841,0.000044349716,0.000059283233],"domain_scores_gemma":[0.99925214,0.00039460076,0.00008529515,0.000109839966,0.00013494038,0.000023147879],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007809864,0.0005596093,0.0003063117,0.00072797656,0.00037896886,0.00069916894,0.0006489182,0.00046522726,0.0018417622],"category_scores_gemma":[0.002320931,0.00019783493,0.00032997903,0.0010244264,0.00034378422,0.0017074302,0.0011269952,0.0008418735,0.0012125693],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00080046494,0.00016946254,0.019185802,0.0004220862,0.0001488707,0.0008743569,0.0010014328,0.15608269,0.032421846,0.0179271,0.017515812,0.75345004],"study_design_scores_gemma":[0.00001645192,0.0000800692,0.0047337837,0.00007022644,0.000053325683,0.00017219193,0.00048163589,0.945918,0.015838455,0.021683808,0.010911662,0.00004044296],"about_ca_topic_score_codex":0.008531898,"about_ca_topic_score_gemma":0.020447724,"teacher_disagreement_score":0.008531898,"about_ca_system_score_codex":0.000499361,"about_ca_system_score_gemma":0.0006402017,"threshold_uncertainty_score":0.016964495},"labels":[],"label_agreement":null},{"id":"W4385387051","doi":"10.18280/ijsse.130315","title":"A Proxy-Based and Collusion Resistant Multi-Authority Revocable CPABE Framework with Efficient User and Attribute-Level Revocation (PCMR-CPABE)","year":2023,"lang":"en","type":"article","venue":"International Journal of Safety and Security Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Revocation; Collusion; Proxy (statistics); Computer security; Computer science; Computer network; Business; Operating system","score_opus":0.012461940221023303,"score_gpt":0.25925700640349403,"score_spread":0.2467950661824707,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385387051","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02297001,0.0009216666,0.97071236,0.00029008923,0.000112320784,0.0003200332,0.00009881751,0.0011892201,0.003385457],"genre_scores_gemma":[0.800056,0.0005792585,0.19427167,0.00017187357,0.000082887746,0.00023233856,0.00021985019,0.00012117371,0.0042649833],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99674207,0.0010499422,0.00028109745,0.0005777062,0.0009610328,0.00038816038],"domain_scores_gemma":[0.9968155,0.000497276,0.00047390242,0.0012807794,0.0006217833,0.00031074477],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021819698,0.00051191304,0.0012077953,0.00051961927,0.00092888036,0.0017563032,0.0021266828,0.001070921,0.0014350661],"category_scores_gemma":[0.004458756,0.00029630013,0.00083828653,0.000726346,0.0012190825,0.003809775,0.0032985557,0.0017049374,0.0006802976],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010646898,0.00036026028,0.0034974315,0.0010173987,0.00023867998,0.0028819875,0.001679949,0.19269946,0.07976706,0.46587688,0.010337072,0.24057917],"study_design_scores_gemma":[0.00011802729,0.0006062724,0.0010652567,0.00013326753,0.00012824993,0.00357299,0.00045237516,0.8240153,0.02962843,0.072645284,0.06737968,0.0002547958],"about_ca_topic_score_codex":0.0014065412,"about_ca_topic_score_gemma":0.0009888187,"teacher_disagreement_score":0.0021819698,"about_ca_system_score_codex":0.00070648943,"about_ca_system_score_gemma":0.0019229596,"threshold_uncertainty_score":0.011539519},"labels":[],"label_agreement":null},{"id":"W4385388948","doi":"10.18280/ria.370324","title":"Towards Amazigh Word Embedding: Corpus Creation and Word2Vec Models Evaluations","year":2023,"lang":"en","type":"article","venue":"Revue d intelligence artificielle","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Word2vec; Computer science; Word embedding; Natural language processing; Artificial intelligence; Word (group theory); Embedding; Linguistics; Philosophy","score_opus":0.0634468073586783,"score_gpt":0.3522641990459674,"score_spread":0.2888173916872891,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385388948","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.77322155,0.0027186938,0.17552021,0.0009464293,0.0008286507,0.0009825117,0.019659724,0.012061859,0.014060315],"genre_scores_gemma":[0.7049836,0.0011837248,0.21756315,0.00023981201,0.00013481258,0.0012266979,0.064780064,0.00070250157,0.0091857435],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9979577,0.00083612086,0.00021987261,0.00038144397,0.00047455766,0.00013037311],"domain_scores_gemma":[0.9964101,0.00103399,0.000105305415,0.0007502201,0.0015145906,0.00018577965],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020288927,0.0010048045,0.0006360045,0.0021451958,0.0007237356,0.0010811497,0.0009339094,0.0006624641,0.0032833137],"category_scores_gemma":[0.0066243545,0.00024706722,0.0005317828,0.0022577313,0.0006040014,0.0026865199,0.0019333251,0.001001178,0.0017581065],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009297431,0.0015989672,0.02358283,0.0018967636,0.0003503726,0.0008681628,0.001642794,0.033832207,0.030441077,0.007970225,0.083714105,0.8131727],"study_design_scores_gemma":[0.0003127469,0.0021139903,0.049558204,0.000418832,0.000335208,0.0024181097,0.0063120336,0.6832712,0.14500222,0.009798633,0.10019572,0.00026311274],"about_ca_topic_score_codex":0.008776582,"about_ca_topic_score_gemma":0.011426567,"teacher_disagreement_score":0.008776582,"about_ca_system_score_codex":0.00058724964,"about_ca_system_score_gemma":0.0009061802,"threshold_uncertainty_score":0.017450988},"labels":[],"label_agreement":null},{"id":"W4385389006","doi":"10.18280/ria.370322","title":"Fine-Tuned IndoBERT Based Model and Data Augmentation for Indonesian Language Paraphrase Identification","year":2023,"lang":"en","type":"article","venue":"Revue d intelligence artificielle","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Paraphrase; Indonesian; Identification (biology); Natural language processing; Computer science; Artificial intelligence; Linguistics; Philosophy; Biology","score_opus":0.06767984215524922,"score_gpt":0.35368987376408795,"score_spread":0.28601003160883876,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385389006","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2660651,0.0029763489,0.68767756,0.0010976135,0.00058556465,0.0006241066,0.0055406033,0.024799017,0.010634104],"genre_scores_gemma":[0.69064337,0.000721276,0.28137842,0.0006235465,0.00012715587,0.0006797241,0.01230531,0.0006439658,0.012877216],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994823,0.0000994627,0.000048532806,0.00022269867,0.000095519616,0.000051483774],"domain_scores_gemma":[0.99898285,0.00031553337,0.00006133986,0.00030553594,0.0002889533,0.00004579302],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00084264437,0.0013252237,0.0009167551,0.0008251632,0.0003861245,0.00093346054,0.002143549,0.0010633153,0.003403832],"category_scores_gemma":[0.00318224,0.00042809316,0.0014704445,0.00071055524,0.00033739046,0.0021299548,0.0010423944,0.002228974,0.002327447],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00096919603,0.0009700514,0.0060328185,0.00047911925,0.00030169974,0.0005878112,0.0002818338,0.24213295,0.040444974,0.002555901,0.019453986,0.6857896],"study_design_scores_gemma":[0.000031245705,0.0001931814,0.0011928454,0.00002866942,0.00005428512,0.00018365627,0.000045637255,0.97698885,0.014821003,0.0015217011,0.0049074097,0.000031600728],"about_ca_topic_score_codex":0.008720703,"about_ca_topic_score_gemma":0.012753008,"teacher_disagreement_score":0.008720703,"about_ca_system_score_codex":0.00080671755,"about_ca_system_score_gemma":0.0009225726,"threshold_uncertainty_score":0.017339885},"labels":[],"label_agreement":null},{"id":"W4385512068","doi":"10.1515/9780228000143-011","title":"Metis (Style) Fiddling","year":2020,"lang":"en","type":"book-chapter","venue":"McGill-Queen's University Press eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Metis; Style (visual arts); History; Archaeology; Computer science; World Wide Web","score_opus":0.0163454143632165,"score_gpt":0.21305723581654057,"score_spread":0.19671182145332405,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385512068","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0005294654,0.0023566324,0.019491179,0.0011445223,0.0024592516,0.00007462262,0.0023584468,0.0044281017,0.96715784],"genre_scores_gemma":[0.0027185208,0.00092016417,0.0050939173,0.00029789974,0.0002409265,0.000043627628,0.0015140316,0.0018011988,0.9873697],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99969614,0.000029946206,0.000015116603,0.000057240966,0.0001699465,0.000031584343],"domain_scores_gemma":[0.9997553,0.000043302876,0.000009019601,0.00005399481,0.00010751748,0.00003076443],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.00029792666,0.0009990628,0.00064596283,0.0024264245,0.0011546083,0.0031591556,0.0012496404,0.00077001535,0.4250922],"category_scores_gemma":[0.0009652767,0.00041736529,0.00066307874,0.0025241855,0.0005868432,0.002670668,0.0015406617,0.0011930862,0.27238357],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000042293403,0.000012262376,0.00005150232,0.00017044746,0.00000408651,0.000038530587,0.00011381752,0.00014874988,0.0007416621,0.03142384,0.7213719,0.24588083],"study_design_scores_gemma":[0.0000024588837,0.0000043119126,0.00007664807,0.000031891945,0.0000018060194,0.000060730203,0.00002496909,0.00010635059,0.0002620353,0.0024846038,0.9969405,0.0000036700626],"about_ca_topic_score_codex":0.0095345825,"about_ca_topic_score_gemma":0.020054862,"teacher_disagreement_score":0.9904654,"about_ca_system_score_codex":0.0014298251,"about_ca_system_score_gemma":0.0010304864,"threshold_uncertainty_score":0.8200362},"labels":[],"label_agreement":null},{"id":"W4385565328","doi":"10.18653/v1/2023.americasnlp-1.15","title":"Finding words that aren’t there: Using word embeddings to improve dictionary search for low-resource languages","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Indigenous; Word (group theory); Natural language processing; Resource (disambiguation); Artificial intelligence; Linguistics; Bilingual dictionary; Natural language; Biology; Philosophy; Ecology","score_opus":0.033113266886195696,"score_gpt":0.3380253507598908,"score_spread":0.3049120838736951,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385565328","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.39234865,0.009499157,0.5631265,0.002241693,0.001612509,0.00036120653,0.0056906687,0.015622137,0.009497481],"genre_scores_gemma":[0.5455276,0.0019775566,0.42600682,0.0010500491,0.00030521126,0.00019173251,0.016731488,0.0019206809,0.006288973],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987558,0.0003429695,0.00019773326,0.00036564612,0.00020783799,0.00012993262],"domain_scores_gemma":[0.9961659,0.001798945,0.0002996778,0.00078838784,0.0007557358,0.00019132737],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011314454,0.0013723616,0.0016577038,0.0031078823,0.0010724769,0.0018726973,0.0017667406,0.0014831645,0.0040113637],"category_scores_gemma":[0.0070392825,0.0007422454,0.0009315916,0.0035661452,0.000852105,0.007859856,0.002811527,0.0015555937,0.0037736096],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012695323,0.00054099824,0.013092465,0.0009091321,0.00029778137,0.0006746175,0.0011859852,0.011448096,0.0282033,0.008725562,0.047337927,0.8863145],"study_design_scores_gemma":[0.00068278157,0.0013428217,0.0061517307,0.000445677,0.00065198616,0.002101848,0.0070618168,0.76195556,0.031837862,0.12234938,0.06515003,0.0002685093],"about_ca_topic_score_codex":0.004073185,"about_ca_topic_score_gemma":0.011670504,"teacher_disagreement_score":0.004073185,"about_ca_system_score_codex":0.00038510642,"about_ca_system_score_gemma":0.0009833772,"threshold_uncertainty_score":0.01341933},"labels":[],"label_agreement":null},{"id":"W4385567069","doi":"10.18653/v1/2022.findings-emnlp.388","title":"Cross-lingual Text-to-SQL Semantic Parsing with Representation Mixup","year":2022,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Compute Canada","keywords":"Computer science; Parsing; Natural language processing; Artificial intelligence; SQL; Task (project management); Programming language; Information retrieval","score_opus":0.0164783422746464,"score_gpt":0.31992541769673727,"score_spread":0.3034470754220909,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385567069","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023243073,0.00036821407,0.9018968,0.0004643254,0.00013559048,0.0003087307,0.00088924234,0.068439804,0.0042541395],"genre_scores_gemma":[0.25415197,0.00033052443,0.72612244,0.0007461127,0.000104936415,0.00033828986,0.006607064,0.005228283,0.006370336],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9964669,0.0011135982,0.00033272905,0.0012004498,0.00057011546,0.0003162675],"domain_scores_gemma":[0.9954482,0.0017681104,0.0002905805,0.001576126,0.0007584609,0.0001584396],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004125655,0.002594722,0.0015643403,0.0016463569,0.0010182492,0.0027963265,0.0032840034,0.0016763086,0.009958453],"category_scores_gemma":[0.0089809,0.001196595,0.0017473941,0.001842775,0.0013157299,0.008653222,0.0080303,0.0036610253,0.0066906423],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006945739,0.0006909196,0.0032594379,0.00091372145,0.00031308646,0.0011723575,0.0018350782,0.03298667,0.060671035,0.027758235,0.03789862,0.83180624],"study_design_scores_gemma":[0.00013574875,0.000396622,0.0014554745,0.00012759007,0.0002904276,0.0014200857,0.0011125469,0.71782374,0.1724404,0.049267028,0.055300537,0.00022987381],"about_ca_topic_score_codex":0.0035011324,"about_ca_topic_score_gemma":0.004731628,"teacher_disagreement_score":0.009958453,"about_ca_system_score_codex":0.0008904812,"about_ca_system_score_gemma":0.002714566,"threshold_uncertainty_score":0.033314347},"labels":[],"label_agreement":null},{"id":"W4385569875","doi":"10.18653/v1/2023.acl-long.310","title":"What the DAAM: Interpreting Stable Diffusion Using Cross Attention","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":98,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Vector Institute; Government of Canada; Canadian Institute for Advanced Research","keywords":"Diffusion; Computer science; Physics; Thermodynamics","score_opus":0.023973227545401102,"score_gpt":0.33085793910418754,"score_spread":0.30688471155878644,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385569875","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.068120755,0.002928855,0.87269104,0.006204307,0.0009365638,0.00013542139,0.00094315264,0.003528503,0.044511385],"genre_scores_gemma":[0.858441,0.00096090947,0.12577005,0.00050611846,0.00031138398,0.00013897786,0.00062212086,0.0016737723,0.0115755815],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9995697,0.00018128126,0.000020681175,0.0001410801,0.000047090947,0.000040064555],"domain_scores_gemma":[0.99767643,0.0013657181,0.0001521961,0.00037581593,0.0003050424,0.00012473605],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018052703,0.0005771184,0.0004487992,0.0013618045,0.0011415804,0.003992303,0.0013115641,0.0015991033,0.014799501],"category_scores_gemma":[0.016153926,0.0005666321,0.00065581163,0.0010753302,0.0013900972,0.008390858,0.0020475187,0.0014400946,0.0016369397],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004686735,0.00008442795,0.00946802,0.00057085557,0.00021894567,0.00083614286,0.005991456,0.019656904,0.012247839,0.64242274,0.03448553,0.27354848],"study_design_scores_gemma":[0.00006886272,0.000050527848,0.0031229723,0.00011718536,0.00013648298,0.00032212472,0.001289256,0.24014638,0.0052169305,0.72389555,0.025562659,0.00007097275],"about_ca_topic_score_codex":0.010776843,"about_ca_topic_score_gemma":0.008096751,"teacher_disagreement_score":0.014799501,"about_ca_system_score_codex":0.0007264152,"about_ca_system_score_gemma":0.0005735994,"threshold_uncertainty_score":0.049509287},"labels":[],"label_agreement":null},{"id":"W4385570009","doi":"10.18653/v1/2023.acl-long.693","title":"Towards Leaving No Indic Language Behind: Building Monolingual Corpora, Benchmark and Models for Indic Languages","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":43,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; Mila - Quebec Artificial Intelligence Institute","funders":"Ministry of Electronics and Information technology","keywords":"Computer science; Benchmark (surveying); Linguistics; Natural language processing; Artificial intelligence; Volume (thermodynamics); Computational linguistics; Programming language; Geography; Philosophy","score_opus":0.01818371428975249,"score_gpt":0.3075003750777553,"score_spread":0.2893166607880028,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385570009","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.26860544,0.0115324445,0.45582137,0.008316102,0.0027878296,0.0014273353,0.10473827,0.09047675,0.056294404],"genre_scores_gemma":[0.33692843,0.0023074343,0.30569547,0.0012237477,0.00035714716,0.0013623774,0.32853624,0.009684571,0.01390456],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99691147,0.0013183252,0.0002277172,0.0009641749,0.000359957,0.00021841512],"domain_scores_gemma":[0.99139684,0.003284262,0.00026998267,0.0023806447,0.0020934693,0.0005748737],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0066923914,0.0020389985,0.001256786,0.0034170826,0.0025834388,0.0047037536,0.0036207687,0.0019165138,0.0076522953],"category_scores_gemma":[0.017050639,0.001622366,0.0015904078,0.003903338,0.001325037,0.014218609,0.0054076714,0.003695731,0.012359134],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016171945,0.0015987325,0.01679928,0.00251373,0.0005465225,0.0010936373,0.0036047979,0.034963265,0.016752344,0.027525887,0.36642665,0.5265579],"study_design_scores_gemma":[0.0007125947,0.00095421117,0.014035383,0.0009622621,0.0007577823,0.0014989083,0.009223645,0.4753,0.032744016,0.056166567,0.40723613,0.00040848582],"about_ca_topic_score_codex":0.01818051,"about_ca_topic_score_gemma":0.03735367,"teacher_disagreement_score":0.01818051,"about_ca_system_score_codex":0.0015109184,"about_ca_system_score_gemma":0.0038236496,"threshold_uncertainty_score":0.036149383},"labels":[],"label_agreement":null},{"id":"W4385570106","doi":"10.18653/v1/2023.acl-long.184","title":"Word sense extension","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Word (group theory); Natural language processing; Extension (predicate logic); Artificial intelligence; Security token; Chaining; Word-sense disambiguation; Word embedding; SemEval; WordNet; Embedding; Linguistics; Psychology","score_opus":0.019868342840360407,"score_gpt":0.2869996631379112,"score_spread":0.26713132029755077,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385570106","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.23496279,0.0029184197,0.73818815,0.0009767319,0.0005577545,0.00038212253,0.0034359752,0.00414941,0.014428606],"genre_scores_gemma":[0.80575633,0.0010177813,0.18409,0.0003694148,0.00010180866,0.00016410156,0.003495774,0.0004025931,0.004602217],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99845386,0.00029599658,0.00017181164,0.00067164283,0.0003242273,0.00008252013],"domain_scores_gemma":[0.99611914,0.0017093086,0.00045484747,0.0010199285,0.00055883365,0.00013801071],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015529438,0.0010604135,0.00052804337,0.0020411925,0.0007614637,0.001691055,0.0012771277,0.0008884476,0.0055493065],"category_scores_gemma":[0.009019129,0.0005514677,0.0014839874,0.0016821537,0.0016012928,0.006306883,0.002769339,0.0014563571,0.0017202195],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009243193,0.00032081982,0.07077766,0.0012493881,0.0004031418,0.0019493746,0.003628235,0.05300616,0.051220056,0.14104751,0.01787416,0.65759915],"study_design_scores_gemma":[0.000096156655,0.00036318347,0.019491868,0.00032184803,0.00024023865,0.0042138337,0.0017840087,0.43161863,0.033398103,0.41240537,0.09585107,0.00021578232],"about_ca_topic_score_codex":0.0016237446,"about_ca_topic_score_gemma":0.0035706377,"teacher_disagreement_score":0.0055493065,"about_ca_system_score_codex":0.00062548707,"about_ca_system_score_gemma":0.00089708297,"threshold_uncertainty_score":0.018564284},"labels":[],"label_agreement":null},{"id":"W4385570158","doi":"10.18653/v1/2023.iwslt-1.32","title":"Language Model Based Target Token Importance Rescaling for Simultaneous Neural Machine Translation","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Natural Sciences and Engineering Research Council of Canada; Ministère de la Défense Nationale","keywords":"Computer science; Machine translation; Security token; Latency (audio); Language model; Artificial intelligence; Entropy (arrow of time); Translation (biology); Context (archaeology); Speech recognition; Natural language processing","score_opus":0.020878076061024497,"score_gpt":0.29888282817787937,"score_spread":0.2780047521168549,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385570158","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.044547215,0.0004945559,0.9492107,0.0001616196,0.00009950942,0.000047904592,0.00011145537,0.0035105022,0.0018165064],"genre_scores_gemma":[0.78047603,0.000256998,0.21330892,0.0001723413,0.00010272084,0.00013243059,0.00053704967,0.0005748294,0.004438671],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99933654,0.0002190377,0.000045191806,0.00017696833,0.00016284642,0.00005940356],"domain_scores_gemma":[0.99862564,0.00070356246,0.00009882203,0.00028199767,0.0002326129,0.000057376485],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011203944,0.00078566815,0.00072328607,0.00048718747,0.00032099496,0.00064056786,0.00091394637,0.00069009315,0.0030236752],"category_scores_gemma":[0.003865772,0.00039270124,0.0005093813,0.0006598724,0.0005480264,0.0018869226,0.00147165,0.001495384,0.0013890874],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007434615,0.00023281058,0.0010664071,0.0002670602,0.000105993015,0.0003645655,0.00023399341,0.33959144,0.07168922,0.011499964,0.0035447835,0.5706603],"study_design_scores_gemma":[0.000017623353,0.00011505391,0.00021264017,0.000006548054,0.000018033228,0.00007322037,0.000014279491,0.9790133,0.013204099,0.006255301,0.0010582422,0.000011644016],"about_ca_topic_score_codex":0.0022352196,"about_ca_topic_score_gemma":0.0041782404,"teacher_disagreement_score":0.0030236752,"about_ca_system_score_codex":0.00048393462,"about_ca_system_score_gemma":0.0007391935,"threshold_uncertainty_score":0.010115206},"labels":[],"label_agreement":null},{"id":"W4385570256","doi":"10.18653/v1/2023.findings-acl.623","title":"Taxonomy of Problems in Lexical Semantics","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Machine Intelligence Institute","keywords":"Soundness; Computer science; Correctness; Equivalence (formal languages); Taxonomy (biology); Natural language processing; Artificial intelligence; Semantics (computer science); Semantic equivalence; Lexical semantics; Formal semantics (linguistics); Meaning (existential); Linguistics; Programming language; Lexical item; Semantic computing; Psychology","score_opus":0.03821098176625253,"score_gpt":0.2781028607484599,"score_spread":0.23989187898220737,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385570256","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03634813,0.00849174,0.91242725,0.011874814,0.00056026294,0.0005065838,0.00064064236,0.000836806,0.028313743],"genre_scores_gemma":[0.3010756,0.0065131574,0.6768829,0.0013455364,0.0009046595,0.0011887341,0.003254199,0.00034294554,0.008492288],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9832712,0.0055625085,0.0024056048,0.002668152,0.005242102,0.00085044024],"domain_scores_gemma":[0.98112595,0.010944159,0.0009353819,0.0040120278,0.002474123,0.0005083981],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009016785,0.0016222155,0.0015145174,0.0071674497,0.004628855,0.009569333,0.0036510103,0.0042271474,0.0042876746],"category_scores_gemma":[0.032994527,0.0010932679,0.0024655259,0.0075843045,0.009555748,0.029629162,0.007023162,0.006710579,0.0014678256],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000034128196,0.0000810487,0.0011632282,0.00039026374,0.000034724733,0.00009152562,0.0008113892,0.0050249803,0.00042864252,0.9112257,0.004140479,0.07657376],"study_design_scores_gemma":[0.000005628113,0.000015877857,0.00021375663,0.0000675794,0.000007298519,0.00013837381,0.00034918217,0.008605103,0.00022684703,0.9792077,0.011148507,0.000014209736],"about_ca_topic_score_codex":0.0021249107,"about_ca_topic_score_gemma":0.0016927369,"teacher_disagreement_score":0.009569333,"about_ca_system_score_codex":0.0038500817,"about_ca_system_score_gemma":0.004921504,"threshold_uncertainty_score":0.04768586},"labels":[],"label_agreement":null},{"id":"W4385570344","doi":"10.18653/v1/2023.findings-acl.412","title":"Long to reign over us: A Case Study of Machine Translation and a New Monarch","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Machine translation; Computer science; Terminology; Artificial intelligence; Synchronous context-free grammar; Natural language processing; Ambiguity; Context (archaeology); Example-based machine translation; Machine translation software usability; Computer-assisted translation; Translation (biology); Linguistics; History; Programming language","score_opus":0.04532918360292675,"score_gpt":0.33658202260244896,"score_spread":0.2912528389995222,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385570344","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.83661014,0.002752957,0.0083244955,0.05107974,0.000505393,0.0001901331,0.00057596224,0.00028221103,0.09967891],"genre_scores_gemma":[0.9572391,0.0012041124,0.007396419,0.004407841,0.00018296765,0.00009502979,0.00029083772,0.0002650111,0.028918667],"study_design_codex":"qualitative","study_design_gemma":"not_applicable","domain_scores_codex":[0.99299663,0.0049748244,0.00020995206,0.0004220281,0.0007563026,0.0006402862],"domain_scores_gemma":[0.98862225,0.0072030835,0.0007940306,0.0011855826,0.0010131438,0.0011819394],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0057091843,0.0007131066,0.000598731,0.0015399908,0.019674685,0.004377215,0.0017604697,0.007131142,0.0062937057],"category_scores_gemma":[0.02128433,0.0003835204,0.0005703154,0.0039513195,0.007786904,0.005179128,0.0030393302,0.0053923544,0.0015428623],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00033774396,0.00048658793,0.014102153,0.00037912678,0.000047632642,0.058968924,0.72992206,0.002820583,0.0024759662,0.07240694,0.041130576,0.07692178],"study_design_scores_gemma":[0.000060630562,0.0002160078,0.01069965,0.0003832251,0.000035803125,0.015052924,0.43378645,0.004203651,0.0026031474,0.017361285,0.5154879,0.000109260814],"about_ca_topic_score_codex":0.05024839,"about_ca_topic_score_gemma":0.12528948,"teacher_disagreement_score":0.05024839,"about_ca_system_score_codex":0.0066454075,"about_ca_system_score_gemma":0.004280293,"threshold_uncertainty_score":0.09991181},"labels":[],"label_agreement":null},{"id":"W4385570347","doi":"10.18653/v1/2023.findings-acl.826","title":"Data Sampling and (In)stability in Machine Translation Evaluation","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Computer science; Machine translation; Ranking (information retrieval); Sampling (signal processing); Consistency (knowledge bases); Data mining; Sample (material); Data set; Artificial intelligence; Set (abstract data type); Data quality; Representation (politics); Stability (learning theory); Translation (biology); Machine learning; Quality (philosophy); Information retrieval; Natural language processing; Service (business)","score_opus":0.2005086114380628,"score_gpt":0.4040928662854281,"score_spread":0.2035842548473653,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385570347","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.51917654,0.0030714488,0.46752518,0.0023005602,0.0003365478,0.0008911996,0.0009371223,0.0008971048,0.004864303],"genre_scores_gemma":[0.9116709,0.00023636434,0.08341328,0.0005138833,0.0002757666,0.0008838129,0.0015292589,0.0003950858,0.0010815589],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.8313111,0.13276191,0.007835992,0.010503304,0.015968418,0.0016193261],"domain_scores_gemma":[0.45468804,0.42787823,0.029599655,0.049446277,0.035708956,0.0026787885],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.13024914,0.0009769915,0.0013860957,0.0043845633,0.0027766584,0.0039974027,0.002244624,0.0018610955,0.001346774],"category_scores_gemma":[0.40342405,0.0007863595,0.00086115213,0.0052123144,0.004112952,0.0052287425,0.0044931234,0.002876244,0.00060755224],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.008146262,0.0014965595,0.3942745,0.0014403869,0.0013509257,0.00052532053,0.014060992,0.06812516,0.027519375,0.038383823,0.009269862,0.43540683],"study_design_scores_gemma":[0.0010280713,0.0049003456,0.21469885,0.0010150454,0.0011208662,0.0017810926,0.007369083,0.5318165,0.088566884,0.1218411,0.025332864,0.000529412],"about_ca_topic_score_codex":0.0034483317,"about_ca_topic_score_gemma":0.004224519,"teacher_disagreement_score":0.86975086,"about_ca_system_score_codex":0.0027219753,"about_ca_system_score_gemma":0.0018316138,"threshold_uncertainty_score":0.68883157},"labels":[],"label_agreement":null},{"id":"W4385570381","doi":"10.18653/v1/2023.acl-long.841","title":"The KITMUS Test: Evaluating Knowledge Integration from Multiple Sources","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; Mila - Quebec Artificial Intelligence Institute","funders":"Canadian Institute for Advanced Research; Nvidia; Microsoft Research","keywords":"Test (biology); Computer science; Computational linguistics; Artificial intelligence; Geology","score_opus":0.03977971996361721,"score_gpt":0.3397854355624773,"score_spread":0.3000057155988601,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385570381","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.87121767,0.01791728,0.045658126,0.0018315894,0.0010766917,0.0017296075,0.020216173,0.01125774,0.029095225],"genre_scores_gemma":[0.8259636,0.0028417553,0.08912022,0.0006233603,0.000257407,0.0010528956,0.065911494,0.00088161434,0.013347644],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9963588,0.001237311,0.00041996018,0.00069431914,0.001176002,0.00011354826],"domain_scores_gemma":[0.98915964,0.007842279,0.0005777964,0.00083646714,0.0010856926,0.0004981354],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035795844,0.0019357316,0.0010695542,0.004310282,0.0007776275,0.0019784707,0.002676517,0.0023711359,0.007471866],"category_scores_gemma":[0.022452494,0.0004572488,0.0010470683,0.0025721448,0.00057705573,0.004410886,0.004317239,0.0011400611,0.0032714831],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.012668006,0.0028985243,0.05833944,0.005355627,0.0068037547,0.0023739422,0.0015895363,0.021260839,0.012439695,0.002635105,0.06496609,0.80866945],"study_design_scores_gemma":[0.010731996,0.016701331,0.24211396,0.0030794432,0.009795975,0.010904958,0.00758893,0.4771094,0.072358236,0.02581028,0.12282237,0.0009830658],"about_ca_topic_score_codex":0.005366194,"about_ca_topic_score_gemma":0.0074787047,"teacher_disagreement_score":0.007471866,"about_ca_system_score_codex":0.00059027364,"about_ca_system_score_gemma":0.0009052843,"threshold_uncertainty_score":0.024995863},"labels":[],"label_agreement":null},{"id":"W4385570394","doi":"10.18653/v1/2023.findings-acl.45","title":"Can Language Models Be Specific? How?","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"Office of the Vice Chancellor for Research, University of Illinois at Chicago; University of Illinois at Urbana-Champaign; Center for Cognitive Computing Systems Research; National Science Foundation","keywords":"Computer science; Benchmark (surveying); Language model; Machine learning; Artificial intelligence; Preference; Natural language processing; Measure (data warehouse); Test (biology); Data mining","score_opus":0.0321223790638185,"score_gpt":0.2750230125668324,"score_spread":0.24290063350301394,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385570394","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5070424,0.0016090039,0.45476717,0.010381285,0.0002954531,0.00023716946,0.003067994,0.0045172516,0.018082272],"genre_scores_gemma":[0.962219,0.00028888969,0.0327046,0.0009717306,0.000047212412,0.000066836365,0.002072194,0.00027287952,0.001356598],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99206233,0.0048859334,0.0003090973,0.0017168062,0.0006525789,0.00037329207],"domain_scores_gemma":[0.95944875,0.031343,0.0018466724,0.00492198,0.0016295139,0.0008101268],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008122038,0.0014460655,0.0007907113,0.0010120755,0.0005165713,0.003655695,0.0012004726,0.0019209271,0.002733026],"category_scores_gemma":[0.051026832,0.0007543011,0.0010163074,0.0007361061,0.0015925606,0.010687368,0.0018776192,0.0040243124,0.0024613685],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0022529406,0.0006078814,0.18256617,0.0016168269,0.00090985466,0.001492955,0.007137109,0.16613498,0.029348236,0.035821114,0.025264734,0.54684716],"study_design_scores_gemma":[0.000118606986,0.000350579,0.029936275,0.00039650413,0.00040286337,0.0017830216,0.0032437718,0.72219175,0.02070301,0.2020081,0.018647647,0.00021789773],"about_ca_topic_score_codex":0.0027304525,"about_ca_topic_score_gemma":0.0044379453,"teacher_disagreement_score":0.008122038,"about_ca_system_score_codex":0.0008692229,"about_ca_system_score_gemma":0.0013218832,"threshold_uncertainty_score":0.042953968},"labels":[],"label_agreement":null},{"id":"W4385570529","doi":"10.18653/v1/2023.findings-acl.421","title":"Residual Prompt Tuning: improving prompt tuning with residual reparameterization","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute; University of Toronto","funders":"","keywords":"Residual; Computer science; Initialization; Benchmark (surveying); Fine-tuning; Base (topology); Algorithm; Mathematics; Physics","score_opus":0.017217007835090593,"score_gpt":0.26368067026990516,"score_spread":0.24646366243481457,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385570529","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08100789,0.001276879,0.84706485,0.0006622523,0.0007105072,0.0002975387,0.00073823624,0.06189466,0.0063471445],"genre_scores_gemma":[0.6984284,0.00038729573,0.28060013,0.0014007225,0.00018540486,0.0005219368,0.0026202917,0.0058224616,0.010033378],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9988701,0.00031610642,0.00006401784,0.0004503872,0.0001651389,0.00013418286],"domain_scores_gemma":[0.998192,0.000864006,0.00010382522,0.00046215186,0.00024802558,0.00012995431],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001695863,0.002586627,0.0010756335,0.00075913395,0.0006386696,0.0014774025,0.0024152535,0.002182625,0.009473467],"category_scores_gemma":[0.012159901,0.0007305218,0.0009785574,0.00056522316,0.0009990323,0.0051065246,0.0028676863,0.0047751316,0.005221465],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015354537,0.0006602098,0.003649973,0.00068357517,0.00016159011,0.00057342404,0.0007441072,0.23916048,0.043578956,0.015431568,0.037634064,0.6561866],"study_design_scores_gemma":[0.00016495246,0.00031459774,0.00049125025,0.000057216337,0.000043508182,0.0001456706,0.00016531973,0.9517955,0.016271742,0.023008548,0.0074841906,0.000057495243],"about_ca_topic_score_codex":0.0019730662,"about_ca_topic_score_gemma":0.0045195757,"teacher_disagreement_score":0.009473467,"about_ca_system_score_codex":0.00086807867,"about_ca_system_score_gemma":0.0013863075,"threshold_uncertainty_score":0.03169191},"labels":[],"label_agreement":null},{"id":"W4385570592","doi":"10.18653/v1/2023.americasnlp-1.7","title":"Fine-tuning Sentence-RoBERTa to Construct Word Embeddings for Low-resource Languages from Bilingual Dictionaries","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of New Brunswick","funders":"","keywords":"Computer science; Word (group theory); Natural language processing; Sentence; Artificial intelligence; German; Construct (python library); Bilingual dictionary; Resource (disambiguation); Word embedding; Cluster analysis; Encoding (memory); Embedding; Similarity (geometry); Linguistics","score_opus":0.013557841742162334,"score_gpt":0.2914919215342536,"score_spread":0.2779340797920912,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385570592","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13394763,0.00087707036,0.83983,0.00046777303,0.0003488013,0.00035456012,0.0021565931,0.015472016,0.0065455562],"genre_scores_gemma":[0.3804068,0.00042730724,0.6003024,0.00043468855,0.00005968891,0.0006105764,0.010177138,0.0014711539,0.0061102086],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9989483,0.0003895922,0.00011317125,0.00037512157,0.00010902854,0.000064840024],"domain_scores_gemma":[0.9980696,0.0007784111,0.00010023599,0.0005867054,0.00038031832,0.00008475733],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017152636,0.0015598038,0.0007223265,0.0012376764,0.00043330074,0.0014779184,0.0012674393,0.0007081568,0.004844789],"category_scores_gemma":[0.007191929,0.00051413436,0.0011926799,0.0012353649,0.0005351084,0.005043377,0.0024101324,0.00252194,0.005478054],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004386475,0.00066901447,0.010847698,0.0007690622,0.0003686093,0.00017894035,0.00096388167,0.051912617,0.033900272,0.018278884,0.026724188,0.85494816],"study_design_scores_gemma":[0.00011985574,0.00040554238,0.0027946248,0.000104412306,0.00013196963,0.00041369302,0.0008707663,0.91473585,0.02638831,0.030784214,0.023176467,0.00007428738],"about_ca_topic_score_codex":0.0024086905,"about_ca_topic_score_gemma":0.0068608723,"teacher_disagreement_score":0.004844789,"about_ca_system_score_codex":0.00059911516,"about_ca_system_score_gemma":0.0009487899,"threshold_uncertainty_score":0.016207397},"labels":[],"label_agreement":null},{"id":"W4385570702","doi":"10.18653/v1/2023.findings-acl.580","title":"AutoMoE: Heterogeneous Mixture-of-Experts with Adaptive Computation for Efficient Neural Machine Translation","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Machine translation; Computer science; Computation; Translation (biology); Natural language processing; Artificial intelligence; Linguistics; Artificial neural network; Speech recognition; Algorithm; Philosophy; Chemistry; Messenger RNA","score_opus":0.022024061738559617,"score_gpt":0.28468806832091026,"score_spread":0.2626640065823506,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385570702","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008140333,0.00072947313,0.98178816,0.00024049667,0.00015546287,0.000078775505,0.00016147352,0.0064714425,0.0022343579],"genre_scores_gemma":[0.23126316,0.000560111,0.7506085,0.0005918404,0.00030219788,0.0005085828,0.0016945098,0.0012394309,0.013231696],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99909353,0.0003738796,0.00004265323,0.00021399879,0.00015348436,0.00012247944],"domain_scores_gemma":[0.9990331,0.00053626514,0.000035953144,0.00017959437,0.00015631088,0.00005880701],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017835431,0.0014943214,0.0019010575,0.0010107995,0.0009611717,0.0013827303,0.0028249288,0.0022920826,0.009032507],"category_scores_gemma":[0.004191389,0.00095126097,0.0013117415,0.0016025817,0.00061921426,0.002781089,0.0036909638,0.0024180354,0.0048569404],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00097030296,0.00027295522,0.00069560413,0.0001966982,0.0003671972,0.00031798353,0.0002158606,0.23326603,0.0076228185,0.013975174,0.025603805,0.7164955],"study_design_scores_gemma":[0.00004656747,0.000044480992,0.000084624895,0.0000082758315,0.00002148535,0.00003557746,0.000022935601,0.9874557,0.0018557775,0.008129827,0.0022829762,0.000011838481],"about_ca_topic_score_codex":0.0075165513,"about_ca_topic_score_gemma":0.015597962,"teacher_disagreement_score":0.009032507,"about_ca_system_score_codex":0.0007192375,"about_ca_system_score_gemma":0.0013032847,"threshold_uncertainty_score":0.030216753},"labels":[],"label_agreement":null},{"id":"W4385570706","doi":"10.18653/v1/2023.acl-industry.50","title":"Evaluating Embedding APIs for Information Retrieval","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Embedding; Volume (thermodynamics); Computer science; Library science; Information retrieval; Artificial intelligence; Physics","score_opus":0.05085128616261112,"score_gpt":0.3954791912835629,"score_spread":0.34462790512095176,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385570706","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.57774186,0.035768833,0.2680689,0.0037046946,0.0024550278,0.0020081955,0.015946805,0.035202283,0.05910336],"genre_scores_gemma":[0.7879354,0.004609993,0.16189606,0.00035484537,0.00045854933,0.00051096437,0.035510253,0.00096533954,0.007758689],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.993435,0.0020296543,0.00076894555,0.00059895165,0.0027371044,0.00043040552],"domain_scores_gemma":[0.9916333,0.0034773564,0.00044773775,0.0022536172,0.001852922,0.00033496897],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003429842,0.0011579766,0.0012062679,0.0054502287,0.00088243315,0.003176612,0.0011537881,0.001533122,0.0042112265],"category_scores_gemma":[0.01660377,0.00038197514,0.0009835777,0.003644906,0.00055301125,0.008168432,0.0029866502,0.00089942105,0.0028624008],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020567821,0.0010813429,0.012298586,0.0015814083,0.00042875565,0.00028509198,0.00034785524,0.01379496,0.012446643,0.026292041,0.08699522,0.8423914],"study_design_scores_gemma":[0.00050347415,0.0026529264,0.011217423,0.00043577794,0.00069115096,0.0011279045,0.0011097827,0.8292393,0.03990649,0.04649923,0.06646402,0.00015259912],"about_ca_topic_score_codex":0.0026158316,"about_ca_topic_score_gemma":0.0044885213,"teacher_disagreement_score":0.0054502287,"about_ca_system_score_codex":0.0010215899,"about_ca_system_score_gemma":0.0015256584,"threshold_uncertainty_score":0.018138945},"labels":[],"label_agreement":null},{"id":"W4385570797","doi":"10.18653/v1/2023.findings-acl.414","title":"K-UniMorph: Korean Universal Morphology and its Feature Schema","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"University of British Columbia; National Research Foundation of Korea; National Research Foundation","keywords":"Computer science; Natural language processing; Artificial intelligence; Inflection; Morpheme; Verb; Schema (genetic algorithms); Feature (linguistics); Linguistics; Information retrieval","score_opus":0.014780051778576456,"score_gpt":0.2621955323836276,"score_spread":0.24741548060505114,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385570797","genre_codex":"dataset","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07715174,0.0007082314,0.0090387305,0.00029293224,0.00018166768,0.00033055476,0.8936418,0.0060081296,0.012646173],"genre_scores_gemma":[0.029937113,0.00016287711,0.013544491,0.00011005629,0.000013865998,0.00040725703,0.9532249,0.00033989342,0.002259463],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99910897,0.00009428479,0.00018425751,0.00029125385,0.00020309212,0.000118130454],"domain_scores_gemma":[0.9982437,0.00020131926,0.0002064339,0.0006885916,0.0004660915,0.00019383982],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00068427506,0.0012011589,0.0006373398,0.003972917,0.0010531122,0.001483074,0.0023316802,0.0011798993,0.008673841],"category_scores_gemma":[0.0024539137,0.0004290667,0.0011225529,0.0050975257,0.0006378842,0.0022693542,0.002168088,0.0011075891,0.009081094],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011971755,0.00074475136,0.051332906,0.0042472044,0.0002208996,0.001761418,0.0014592241,0.003895705,0.027841575,0.016237114,0.68119305,0.209869],"study_design_scores_gemma":[0.00015919202,0.00017654878,0.056717865,0.00023678564,0.000084885,0.0020801192,0.0014912927,0.005695747,0.0138429245,0.0045185476,0.9148721,0.00012397308],"about_ca_topic_score_codex":0.0063889353,"about_ca_topic_score_gemma":0.014977622,"teacher_disagreement_score":0.008673841,"about_ca_system_score_codex":0.0009717668,"about_ca_system_score_gemma":0.0015421676,"threshold_uncertainty_score":0.029016912},"labels":[],"label_agreement":null},{"id":"W4385570952","doi":"10.18653/v1/2023.findings-acl.179","title":"Grounding the Lexical Substitution Task in Entailment","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Machine Intelligence Institute","keywords":"Textual entailment; Logical consequence; Substitution (logic); Computer science; Natural language processing; Sentence; Artificial intelligence; Context (archaeology); Task (project management); Relation (database); Word (group theory); Linguistics; Programming language; Data mining","score_opus":0.023238470013058673,"score_gpt":0.29845548792310733,"score_spread":0.27521701791004866,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385570952","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14349954,0.0022777584,0.8297484,0.0014657342,0.00033743697,0.000604736,0.0034381696,0.005995969,0.012632287],"genre_scores_gemma":[0.4078676,0.000515623,0.5794479,0.000407771,0.0001401747,0.0004185548,0.008300397,0.0006357838,0.0022661497],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.98883665,0.0052044997,0.0012986196,0.002118391,0.0021728962,0.0003689056],"domain_scores_gemma":[0.98376137,0.008334717,0.0009254197,0.0051717074,0.0016123038,0.0001944701],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006218716,0.0014063771,0.0021788639,0.0033769305,0.0022996902,0.0031086418,0.0030015393,0.0021837447,0.005581841],"category_scores_gemma":[0.030305741,0.0008922329,0.0016558588,0.003980956,0.002394356,0.011122035,0.007047263,0.0023670257,0.002622248],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017337099,0.00071674044,0.013010858,0.0029113407,0.00047426877,0.0014111929,0.0031035962,0.030329254,0.047182232,0.17842129,0.03098945,0.68971604],"study_design_scores_gemma":[0.00023810718,0.0005397232,0.006126823,0.00048314955,0.00042491787,0.0019877267,0.0021206199,0.44034365,0.10370999,0.36148453,0.08231746,0.00022329586],"about_ca_topic_score_codex":0.0029067863,"about_ca_topic_score_gemma":0.0069378065,"teacher_disagreement_score":0.006218716,"about_ca_system_score_codex":0.001229201,"about_ca_system_score_gemma":0.0024988875,"threshold_uncertainty_score":0.032888114},"labels":[],"label_agreement":null},{"id":"W4385571027","doi":"10.18653/v1/2023.acl-short.64","title":"Improving Automatic Quotation Attribution in Literary Novels","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University; Vector Institute; University of Toronto","funders":"","keywords":"Computer science; Attribution; Coreference; Benchmark (surveying); Task (project management); Authorship attribution; Natural language processing; Character (mathematics); Identification (biology); Artificial intelligence; Set (abstract data type); Inference; Training set; Resolution (logic); Psychology; Programming language","score_opus":0.014265566408176543,"score_gpt":0.2774959925160585,"score_spread":0.26323042610788194,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385571027","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.58412576,0.01026456,0.35318437,0.0032360258,0.001863056,0.00032696401,0.003802878,0.032401923,0.01079445],"genre_scores_gemma":[0.8771364,0.0008767675,0.10600468,0.0003882672,0.00046903113,0.000091837916,0.0075650406,0.00075173227,0.006716184],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9959454,0.0014592771,0.00030683677,0.0015006426,0.0005359638,0.00025196516],"domain_scores_gemma":[0.9868217,0.008135526,0.0009483253,0.0016557853,0.0020336597,0.00040510567],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0065127546,0.0016230359,0.0013751693,0.0034250878,0.0013577596,0.0028926134,0.002625747,0.002169483,0.0029099945],"category_scores_gemma":[0.024888048,0.0005936021,0.0011042319,0.0018696898,0.00094797753,0.0053175115,0.003098135,0.002383612,0.004335598],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009849502,0.0007362764,0.030033931,0.0013146942,0.00041194717,0.0007581946,0.0020260552,0.064472765,0.019411225,0.004130548,0.028990561,0.84672886],"study_design_scores_gemma":[0.00005576845,0.00012846115,0.00734792,0.000111378366,0.00009763764,0.00032471318,0.0009986134,0.9583852,0.015754454,0.008818736,0.007918521,0.00005855418],"about_ca_topic_score_codex":0.004468019,"about_ca_topic_score_gemma":0.006003317,"teacher_disagreement_score":0.0065127546,"about_ca_system_score_codex":0.0010197806,"about_ca_system_score_gemma":0.0015527724,"threshold_uncertainty_score":0.0344432},"labels":[],"label_agreement":null},{"id":"W4385571145","doi":"10.18653/v1/2023.acl-long.865","title":"LeXFiles and LegalLAMA: Facilitating English Multinational Legal Language Model Development","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":30,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Fonds de recherche du Québec – Nature et technologies; Innovationsfonden","keywords":"Multinational corporation; Upstream (networking); Benchmark (surveying); Computer science; Downstream (manufacturing); Best practice; Natural language processing; Work (physics); Knowledge management; Artificial intelligence; Engineering; Political science; Operations management; Law","score_opus":0.014058859617278895,"score_gpt":0.2717166111702214,"score_spread":0.2576577515529425,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385571145","genre_codex":"empirical","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5167446,0.0017455079,0.20546523,0.0044978485,0.0008771823,0.0012743401,0.038931448,0.19239293,0.038070824],"genre_scores_gemma":[0.6019179,0.0005890641,0.26070523,0.000919559,0.00007512816,0.001121614,0.11656435,0.007881297,0.010225837],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99717134,0.0013569127,0.00020050876,0.0006295155,0.00048717542,0.00015460109],"domain_scores_gemma":[0.98374915,0.010632995,0.00037911476,0.0031162614,0.0017039491,0.00041851474],"candidate_categories":["open_science"],"consensus_categories":[],"category_scores_codex":[0.005826701,0.0012741133,0.0005654818,0.0014372085,0.0006599394,0.0021481018,0.0031767495,0.0012950408,0.015480586],"category_scores_gemma":[0.03336447,0.00074576744,0.0007956204,0.0010984815,0.00079690607,0.0068135243,0.0035286022,0.003191066,0.0063390597],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010766814,0.0016895764,0.023659471,0.0015227355,0.00035415703,0.0008906933,0.0030236475,0.110821396,0.01647819,0.013937897,0.18722235,0.63932323],"study_design_scores_gemma":[0.00030139746,0.000529913,0.009581777,0.0003401523,0.00009656923,0.000518275,0.0019668087,0.8347487,0.036184277,0.010981324,0.10458173,0.00016902667],"about_ca_topic_score_codex":0.021869233,"about_ca_topic_score_gemma":0.033345018,"teacher_disagreement_score":0.99682325,"about_ca_system_score_codex":0.0016865958,"about_ca_system_score_gemma":0.0025244062,"threshold_uncertainty_score":0.051787674},"labels":[],"label_agreement":null},{"id":"W4385571292","doi":"10.18653/v1/2023.law-1.22","title":"Unified Syntactic Annotation of English in the CGEL Framework","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Humber Polytechnic","funders":"National Science Foundation","keywords":"Treebank; Computer science; Annotation; Natural language processing; Artificial intelligence; Formalism (music); Grammar; Syntax; English grammar; Linguistics","score_opus":0.016652868257825502,"score_gpt":0.2916388996318318,"score_spread":0.2749860313740063,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385571292","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019234665,0.00037657272,0.93959504,0.0012665915,0.0002682552,0.0001637489,0.0030051942,0.0057473266,0.030342601],"genre_scores_gemma":[0.22350058,0.00045045748,0.7561174,0.00071030273,0.00016090216,0.00030481478,0.006492095,0.0027727145,0.009490707],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9980848,0.0007442356,0.00016487866,0.00051505794,0.0003415274,0.0001495841],"domain_scores_gemma":[0.99644,0.0012184227,0.00021166066,0.0010386355,0.0009916128,0.000099638],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033943986,0.0008253279,0.0007740706,0.003922056,0.0015628516,0.003892391,0.0017578715,0.0009601544,0.009516178],"category_scores_gemma":[0.0068705,0.0007306899,0.0007757114,0.0038235625,0.002329966,0.007885438,0.0030442728,0.0018259377,0.0026897155],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001300802,0.00007078505,0.0008447585,0.00038098136,0.000030061876,0.00058230053,0.0036424033,0.0060067396,0.008234611,0.8822981,0.01675797,0.08102119],"study_design_scores_gemma":[0.00007637211,0.000063921594,0.0023716188,0.00045832444,0.00011539865,0.0006013967,0.0018135525,0.07235973,0.024336934,0.4366951,0.4609456,0.00016213093],"about_ca_topic_score_codex":0.007975573,"about_ca_topic_score_gemma":0.010827872,"teacher_disagreement_score":0.009516178,"about_ca_system_score_codex":0.002018649,"about_ca_system_score_gemma":0.002379975,"threshold_uncertainty_score":0.03183478},"labels":[],"label_agreement":null},{"id":"W4385571319","doi":"10.18653/v1/2023.acl-long.366","title":"Multiview Identifiers Enhanced Generative Retrieval","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":32,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"Hong Kong Polytechnic University; National Natural Science Foundation of China","keywords":"Identifier; Computer science; Information retrieval; Substring; Robustness (evolution); Unique identifier; Generative grammar; Artificial intelligence; Data mining; Data structure; Programming language","score_opus":0.021807482967770183,"score_gpt":0.3116852515179208,"score_spread":0.2898777685501506,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385571319","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04321941,0.0042999093,0.9207515,0.0010634579,0.00094827346,0.00015788781,0.002142622,0.013109081,0.014307813],"genre_scores_gemma":[0.3719399,0.0015548635,0.56784207,0.0007546853,0.00082148676,0.00011274603,0.008091181,0.0019348523,0.046948213],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9986726,0.00034653413,0.0000762834,0.00023156738,0.00048678997,0.00018622351],"domain_scores_gemma":[0.9983551,0.00043987963,0.000056624584,0.0005319631,0.00053225516,0.000084107756],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001402041,0.0007663756,0.0019538687,0.0025759155,0.0007224246,0.0026092306,0.0016669189,0.0013725536,0.01637291],"category_scores_gemma":[0.002812094,0.0005036946,0.0012782289,0.002108362,0.0006585063,0.0023345205,0.0028256678,0.0011684925,0.006871596],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00096939213,0.00025070668,0.0008062959,0.00030598624,0.0001481602,0.00036232002,0.00022380464,0.02152026,0.07197346,0.026351364,0.049333777,0.8277545],"study_design_scores_gemma":[0.00019305896,0.00026329738,0.0017713477,0.000058609512,0.0002859702,0.0013140586,0.00016294976,0.8293029,0.08127306,0.035696793,0.049558446,0.000119517455],"about_ca_topic_score_codex":0.007145163,"about_ca_topic_score_gemma":0.012029023,"teacher_disagreement_score":0.01637291,"about_ca_system_score_codex":0.0009396381,"about_ca_system_score_gemma":0.0016386939,"threshold_uncertainty_score":0.054772854},"labels":[],"label_agreement":null},{"id":"W4385571394","doi":"10.18653/v1/2023.acl-demo.45","title":"LaTeX2Solver: a Hierarchical Semantic Parsing of LaTeX Document into Code for an Assistive Optimization Modeling Application","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia; Huawei Technologies (Canada)","funders":"","keywords":"Parsing; Computer science; Zhàng; Chen; Programming language; Natural language processing; Code (set theory); Artificial intelligence; Volume (thermodynamics); Association (psychology); Philosophy; Physics; History; Biology","score_opus":0.02244437265272833,"score_gpt":0.32045620560524873,"score_spread":0.29801183295252043,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385571394","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00623376,0.00020449566,0.6658718,0.000430145,0.0001616394,0.00014253585,0.004707114,0.31408688,0.008161707],"genre_scores_gemma":[0.10136579,0.00052760204,0.7900048,0.0005738066,0.00007963268,0.0004269281,0.020661471,0.06425723,0.022102814],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99948907,0.000097537064,0.000048081936,0.0001135551,0.00021280155,0.00003887612],"domain_scores_gemma":[0.9990938,0.00041002172,0.000058111338,0.00020938733,0.00018098633,0.00004764984],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00085235666,0.0015014603,0.0005985709,0.0009942746,0.0006911235,0.0020900657,0.0021789852,0.0011822359,0.026323711],"category_scores_gemma":[0.00260426,0.0009761261,0.0013525467,0.00068092445,0.00053505,0.0027433725,0.0018785014,0.0014978902,0.011453303],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001077373,0.00042650037,0.0026563762,0.0011752106,0.00023836255,0.0009597806,0.001089845,0.026037542,0.053084794,0.047742486,0.4611873,0.40432444],"study_design_scores_gemma":[0.0003374306,0.00017128453,0.0011847386,0.00017575426,0.00011894431,0.0005699151,0.00032248805,0.48356175,0.08979181,0.031395115,0.3922055,0.00016526716],"about_ca_topic_score_codex":0.004107387,"about_ca_topic_score_gemma":0.0070644175,"teacher_disagreement_score":0.026323711,"about_ca_system_score_codex":0.000704595,"about_ca_system_score_gemma":0.0017268528,"threshold_uncertainty_score":0.08806157},"labels":[],"label_agreement":null},{"id":"W4385571527","doi":"10.18653/v1/2023.acl-short.112","title":"Exploring the Impact of Layer Normalization for Zero-shot Neural Machine Translation","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Japan Society for the Promotion of Science; Institute for Catastrophic Loss Reduction","keywords":"Normalization (sociology); Machine translation; Computer science; Speech recognition; Artificial intelligence; Artificial neural network; Computational linguistics; Natural language processing; Deep neural networks; Sociology","score_opus":0.15734077928244297,"score_gpt":0.357068152226181,"score_spread":0.19972737294373802,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385571527","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.49917322,0.009517545,0.45764694,0.0015289075,0.00072902156,0.00016632711,0.00053280714,0.007666952,0.023038309],"genre_scores_gemma":[0.9099207,0.00076547946,0.08433744,0.00018481581,0.00008482508,0.00007913507,0.00075923203,0.00051142706,0.0033569534],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988802,0.0005268896,0.0000730799,0.00018614468,0.00018905223,0.00014457008],"domain_scores_gemma":[0.9957853,0.002777618,0.00013323374,0.0005528139,0.0006572382,0.00009376042],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026031123,0.0010724314,0.0010105146,0.000589174,0.00069714425,0.0014142294,0.0017334668,0.0011187178,0.0050422386],"category_scores_gemma":[0.013025835,0.00044051997,0.00045778003,0.0008967069,0.00053882593,0.0046173413,0.0011676897,0.0011398157,0.0015016047],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001859699,0.0007885847,0.0054654954,0.0004923935,0.00027352167,0.00026722735,0.000258244,0.19882943,0.02784171,0.015552948,0.010216466,0.7381542],"study_design_scores_gemma":[0.000049395057,0.00023522045,0.0008423423,0.00002850371,0.00008949323,0.00006647123,0.00012815841,0.97261363,0.01511248,0.009540507,0.0012803528,0.0000135218725],"about_ca_topic_score_codex":0.008782985,"about_ca_topic_score_gemma":0.014072755,"teacher_disagreement_score":0.008782985,"about_ca_system_score_codex":0.0010767904,"about_ca_system_score_gemma":0.0014258573,"threshold_uncertainty_score":0.017463744},"labels":[],"label_agreement":null},{"id":"W4385571623","doi":"10.18653/v1/2023.iwslt-1.27","title":"Learning Nearest Neighbour Informed Latent Word Embeddings to Improve Zero-Shot Machine Translation","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Natural Sciences and Engineering Research Council of Canada; Ministère de la Défense Nationale","keywords":"Computer science; Machine translation; Artificial intelligence; Translation (biology); Sentence; Zero (linguistics); Word (group theory); Natural language processing; Security token; Exploit; Shot (pellet); Transfer of learning; Linguistics","score_opus":0.022378695401782757,"score_gpt":0.29631099334402705,"score_spread":0.2739322979422443,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385571623","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.076575994,0.0010852983,0.91190493,0.00019751523,0.00019647295,0.00007222736,0.00023068143,0.0060170824,0.003719771],"genre_scores_gemma":[0.6835267,0.00045947105,0.30331013,0.00031798685,0.00011558346,0.00014407892,0.0023979535,0.0005983224,0.00912962],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989023,0.0003608926,0.000068175614,0.0003353495,0.00023845173,0.00009482214],"domain_scores_gemma":[0.99877864,0.0004803326,0.00008560196,0.0003945332,0.00020881786,0.000052021027],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011178055,0.0010385123,0.0011766949,0.0007310845,0.000564105,0.00094982,0.0017393767,0.0011680329,0.0032180347],"category_scores_gemma":[0.0044546262,0.00041439317,0.00072889496,0.0010508057,0.0007588407,0.0030213988,0.0023018213,0.0017122837,0.0032032232],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00048588126,0.00059788185,0.0014996935,0.00029599303,0.00019862728,0.000330097,0.00043752423,0.13180141,0.03534968,0.01887343,0.010817651,0.79931206],"study_design_scores_gemma":[0.00005041327,0.00027714265,0.00040504007,0.000021403926,0.000042658474,0.00018496098,0.0001208784,0.95199966,0.015281382,0.028560102,0.0030213885,0.00003501903],"about_ca_topic_score_codex":0.002412211,"about_ca_topic_score_gemma":0.005578727,"teacher_disagreement_score":0.0032180347,"about_ca_system_score_codex":0.0005035611,"about_ca_system_score_gemma":0.00079634547,"threshold_uncertainty_score":0.010765374},"labels":[],"label_agreement":null},{"id":"W4385571651","doi":"10.18653/v1/2023.findings-acl.597","title":"Leveraging Synthetic Targets for Machine Translation","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Machine translation; Ground truth; Translation (biology); Synthetic data; Training set; Artificial intelligence; Machine learning; Natural language processing","score_opus":0.028863720562592322,"score_gpt":0.29049041270562476,"score_spread":0.2616266921430324,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385571651","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20358364,0.0014296719,0.76283765,0.0016440178,0.0007576568,0.00044347366,0.007178337,0.0093553,0.0127703445],"genre_scores_gemma":[0.77598906,0.0005336837,0.18394233,0.00065820315,0.00021438026,0.00083757733,0.03161454,0.0014982526,0.0047120205],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9964036,0.0022126876,0.00017589283,0.0005870284,0.00047777762,0.00014311187],"domain_scores_gemma":[0.99135005,0.005269812,0.00030828916,0.0018077167,0.0010783119,0.00018579711],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004175521,0.0016912726,0.0010777916,0.00088710734,0.00082296686,0.0018116771,0.0018561241,0.0016225089,0.004335271],"category_scores_gemma":[0.01928459,0.000691939,0.0009600152,0.0015388839,0.0010956889,0.0031254336,0.0024789446,0.0023251793,0.0037900927],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015362261,0.00068263104,0.008287156,0.0010698996,0.0003900591,0.00080722454,0.0006786494,0.736239,0.028805338,0.02162801,0.030400861,0.16947494],"study_design_scores_gemma":[0.000092888666,0.00030242314,0.0012667982,0.00008009932,0.00006441049,0.00039081526,0.00019870332,0.9342528,0.026677014,0.021019232,0.015598181,0.00005669837],"about_ca_topic_score_codex":0.0029710366,"about_ca_topic_score_gemma":0.0040429705,"teacher_disagreement_score":0.004335271,"about_ca_system_score_codex":0.000755761,"about_ca_system_score_gemma":0.0009891032,"threshold_uncertainty_score":0.022082508},"labels":[],"label_agreement":null},{"id":"W4385571741","doi":"10.18653/v1/2023.acl-short.74","title":"Gradient Ascent Post-training Enhances Language Model Generalization","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"Institute for Information and Communications Technology Promotion; Ministry of Science and ICT, South Korea; Korea Advanced Institute of Science and Technology","keywords":"Generalization; Computer science; Task (project management); Artificial intelligence; Training set; Language model; Training (meteorology); Zero (linguistics); Natural language processing; Speech recognition; Mathematics; Linguistics","score_opus":0.022953831562168726,"score_gpt":0.2952370719509481,"score_spread":0.2722832403887794,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385571741","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.25342172,0.001877186,0.70722795,0.0010134943,0.000645932,0.00020405713,0.00055713305,0.025739567,0.009313061],"genre_scores_gemma":[0.7990459,0.00036570997,0.18822235,0.000680727,0.00015566037,0.00017161909,0.0016919194,0.0011071155,0.008559028],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990295,0.00025911588,0.00004944139,0.00037168086,0.00018272575,0.00010758225],"domain_scores_gemma":[0.9976184,0.0013064991,0.000080833204,0.00047528703,0.0004097668,0.00010909709],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020235742,0.0016321944,0.0008317164,0.0005910173,0.00044455598,0.0008687178,0.0012397203,0.001417888,0.004409952],"category_scores_gemma":[0.010616629,0.0004284755,0.0005807791,0.00041217744,0.0006545887,0.0022088008,0.0013925445,0.0026877567,0.0030598247],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005228193,0.00057664944,0.0032758645,0.0003143125,0.00019829077,0.0002807852,0.0003172563,0.26934877,0.072698146,0.0023313959,0.015056899,0.6350787],"study_design_scores_gemma":[0.000030764924,0.00018568263,0.002069536,0.000018887677,0.000035474633,0.000094612195,0.000056979363,0.9652613,0.027449023,0.0019263042,0.002846615,0.00002482478],"about_ca_topic_score_codex":0.004605994,"about_ca_topic_score_gemma":0.008230888,"teacher_disagreement_score":0.004605994,"about_ca_system_score_codex":0.0005923008,"about_ca_system_score_gemma":0.0009422289,"threshold_uncertainty_score":0.014752686},"labels":[],"label_agreement":null},{"id":"W4385571887","doi":"10.18653/v1/2023.acl-long.180","title":"HistRED: A Historical Document-Level Relation Extraction Dataset","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"National Supercomputing Center, Korea Institute of Science and Technology Information; Defense Acquisition Program Administration; Korea Advanced Institute of Science and Technology; Agency for Defense Development; Korea University","keywords":"Computer science; Natural language processing; Relationship extraction; Robustness (evolution); Relation (database); Sentence; Artificial intelligence; Context (archaeology); Information retrieval; License; Information extraction; Data mining; History; Archaeology","score_opus":0.043339890609997336,"score_gpt":0.32191329201971425,"score_spread":0.2785734014097169,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385571887","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021917017,0.0016292586,0.0024182799,0.0005216435,0.00017078531,0.0001858945,0.96360105,0.004056857,0.005499244],"genre_scores_gemma":[0.009826916,0.00023031383,0.005863031,0.00012792322,0.00002539266,0.00013161704,0.9818304,0.00007927918,0.0018850295],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99888366,0.00017099828,0.00018367583,0.00037165743,0.0002785175,0.000111443485],"domain_scores_gemma":[0.9981146,0.00048762068,0.00021137256,0.00054250367,0.00048235542,0.00016163617],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00085663784,0.0015243806,0.00074663956,0.0054788715,0.0012185072,0.0011412231,0.002049844,0.0018208963,0.0082945945],"category_scores_gemma":[0.0035224361,0.00037477724,0.0010223916,0.006240895,0.00049440336,0.0017855613,0.001456751,0.0011252353,0.010777595],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00043025293,0.00042887562,0.0118740685,0.0025651364,0.00017261606,0.0011074649,0.00039861968,0.0021200615,0.0057914224,0.0024993357,0.906482,0.066130094],"study_design_scores_gemma":[0.0002668537,0.00017043573,0.038624898,0.0003291331,0.00014072412,0.0014658616,0.0010124177,0.009293002,0.0091115385,0.0022912873,0.9371782,0.000115730676],"about_ca_topic_score_codex":0.01994704,"about_ca_topic_score_gemma":0.04843967,"teacher_disagreement_score":0.01994704,"about_ca_system_score_codex":0.0014584874,"about_ca_system_score_gemma":0.0017609717,"threshold_uncertainty_score":0.039661884},"labels":[],"label_agreement":null},{"id":"W4385571975","doi":"10.18653/v1/2023.cawl-1.9","title":"Disambiguating Numeral Sequences to Decipher Ancient Accounting Corpora","year":2023,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Numeral system; Computer science; DECIPHER; Scripting language; Natural language processing; Bootstrapping (finance); Artificial intelligence; Notation; Set (abstract data type); Task (project management); Selection (genetic algorithm); Devanagari; Arithmetic; Programming language; Mathematics","score_opus":0.0499249752871117,"score_gpt":0.3279673079420105,"score_spread":0.2780423326548988,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385571975","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.46569645,0.0030914587,0.48362917,0.001377184,0.0006915689,0.00063959026,0.008782297,0.010451335,0.025640989],"genre_scores_gemma":[0.5515441,0.00061909645,0.42825213,0.0002033615,0.0001975769,0.00022611422,0.014302872,0.000997717,0.0036570122],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9975489,0.0008103268,0.0003679426,0.00060487597,0.000562423,0.00010550431],"domain_scores_gemma":[0.9887468,0.0067816176,0.0006516363,0.0021747241,0.0014569903,0.00018822956],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002264586,0.00076046336,0.00070021313,0.004701765,0.0015853959,0.002482227,0.0012789755,0.0008426789,0.0042027864],"category_scores_gemma":[0.020481063,0.00040045546,0.00038494696,0.004790091,0.0013858059,0.0043144957,0.0020873575,0.0018071325,0.0024085508],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000827417,0.00033277445,0.015541708,0.0018411872,0.00010115661,0.0020218072,0.0065545803,0.028247027,0.049995143,0.08526592,0.037281975,0.7719892],"study_design_scores_gemma":[0.00015828326,0.00023001642,0.020674825,0.000578845,0.00012532675,0.0028209023,0.0065501463,0.37709963,0.17027238,0.1408426,0.28043762,0.00020942217],"about_ca_topic_score_codex":0.0014928922,"about_ca_topic_score_gemma":0.003065787,"teacher_disagreement_score":0.004701765,"about_ca_system_score_codex":0.0008730802,"about_ca_system_score_gemma":0.001080168,"threshold_uncertainty_score":0.014059722},"labels":[],"label_agreement":null},{"id":"W4385572241","doi":"10.18653/v1/2023.findings-acl.478","title":"DiffuDetox: A Mixed Diffusion Model for Text Detoxification","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"LG Electronics","keywords":"Griffin; Detoxification (alternative medicine); Diffusion; Computer science; Geochemistry; Thermodynamics; Geology; Medicine; History; Physics; Archaeology","score_opus":0.02908380101525114,"score_gpt":0.2944964562920917,"score_spread":0.26541265527684055,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385572241","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009304974,0.0010874042,0.97759503,0.0014454854,0.00025109263,0.00022763992,0.002187301,0.0043120286,0.0035890571],"genre_scores_gemma":[0.25036952,0.0016155168,0.70505077,0.0006718782,0.00034492605,0.0014146526,0.0067667468,0.0022209198,0.03154502],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99848336,0.00088982465,0.00008694718,0.00026206818,0.00019439976,0.000083498184],"domain_scores_gemma":[0.9958171,0.0030288585,0.0001548174,0.00038344756,0.00043806463,0.00017778786],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037760804,0.0011344224,0.001419122,0.0016670062,0.0009964273,0.00323996,0.0034254228,0.0022825284,0.014128052],"category_scores_gemma":[0.012502338,0.0008538565,0.0019362955,0.0016239346,0.0007314751,0.004282633,0.0028739297,0.0026318391,0.0051449486],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011105184,0.00036376552,0.005132713,0.00046876675,0.00054124714,0.00046088584,0.00092833303,0.36878875,0.0018523621,0.27768284,0.04850852,0.29416135],"study_design_scores_gemma":[0.000051390267,0.000031895157,0.00015024561,0.000016744372,0.000020528101,0.00003263189,0.000025560794,0.9453597,0.00024157137,0.04632904,0.007722469,0.00001819213],"about_ca_topic_score_codex":0.01531926,"about_ca_topic_score_gemma":0.019063696,"teacher_disagreement_score":0.01531926,"about_ca_system_score_codex":0.001693068,"about_ca_system_score_gemma":0.0019734032,"threshold_uncertainty_score":0.047263086},"labels":[],"label_agreement":null},{"id":"W4385572461","doi":"10.18653/v1/2023.acl-long.371","title":"Towards standardizing Korean Grammatical Error Correction: Datasets and Annotation","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"Institute for Information and Communications Technology Promotion; Ministry of Science and ICT, South Korea","keywords":"Annotation; Computer science; Natural language processing; Computational linguistics; Volume (thermodynamics); Artificial intelligence; Information retrieval","score_opus":0.022818588312560038,"score_gpt":0.3183019235604153,"score_spread":0.29548333524785525,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385572461","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1813631,0.004690282,0.15529193,0.0040028146,0.0025010353,0.0035320169,0.5799575,0.053679924,0.014981398],"genre_scores_gemma":[0.065618895,0.00056116725,0.14049621,0.0007811472,0.00012933004,0.002400765,0.7836235,0.0033330603,0.0030559946],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.98257095,0.005563087,0.0036916786,0.0046421173,0.0027840436,0.0007481291],"domain_scores_gemma":[0.92205465,0.010440736,0.0058439146,0.031888235,0.027414953,0.0023575178],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.018745733,0.0028955373,0.0015276471,0.009388514,0.0030170525,0.0029227827,0.006235063,0.0043564215,0.004432401],"category_scores_gemma":[0.044071615,0.0012536168,0.001661469,0.0071873716,0.0017594629,0.005291253,0.009966641,0.0035853856,0.011562851],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014339356,0.0018151054,0.066748574,0.0044785817,0.00067406584,0.0013036027,0.0029426715,0.0073030875,0.033524606,0.007827921,0.5958317,0.27611625],"study_design_scores_gemma":[0.0015542966,0.0009504029,0.108311005,0.002196714,0.0012582749,0.0025716466,0.0077033197,0.06498696,0.110546954,0.020259904,0.6788245,0.00083606737],"about_ca_topic_score_codex":0.010162019,"about_ca_topic_score_gemma":0.019658625,"teacher_disagreement_score":0.018745733,"about_ca_system_score_codex":0.001575986,"about_ca_system_score_gemma":0.0056176996,"threshold_uncertainty_score":0.09913808},"labels":[],"label_agreement":null},{"id":"W4385572774","doi":"10.18653/v1/2022.emnlp-main.89","title":"Geographic Citation Gaps in NLP Research","year":2022,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Citation; Publication; Field (mathematics); Computer science; Diversity (politics); China; Data science; Artificial intelligence; World Wide Web; Geography; Political science","score_opus":0.040031379573968316,"score_gpt":0.3559137284766063,"score_spread":0.315882348902638,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385572774","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5359053,0.09654161,0.026874704,0.024300752,0.0013496362,0.00031599702,0.2431037,0.0031124975,0.06849578],"genre_scores_gemma":[0.7398317,0.020273415,0.021067059,0.0014982369,0.0014752893,0.00080974534,0.20797937,0.00076850486,0.006296675],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.98007226,0.006547528,0.0033482616,0.003647671,0.0056764297,0.000707887],"domain_scores_gemma":[0.84582514,0.10780207,0.024497783,0.011194903,0.009023428,0.0016566428],"candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.011099553,0.00056343037,0.001198782,0.032182883,0.0024779637,0.0067012794,0.0016113067,0.0017488095,0.006818156],"category_scores_gemma":[0.11970812,0.00054034137,0.00072968897,0.10113987,0.001820462,0.007411618,0.0055551883,0.0013445542,0.0029849198],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00064568705,0.0001990059,0.36649317,0.012798523,0.00095271174,0.0013454136,0.017385475,0.012049995,0.002098189,0.107834436,0.21496187,0.26323548],"study_design_scores_gemma":[0.0001614121,0.00006132176,0.24696429,0.0018971957,0.00029949227,0.0012029853,0.007888689,0.007848232,0.0020098994,0.0984778,0.63306004,0.00012873608],"about_ca_topic_score_codex":0.010052412,"about_ca_topic_score_gemma":0.015258175,"teacher_disagreement_score":0.9889004,"about_ca_system_score_codex":0.00266557,"about_ca_system_score_gemma":0.0028244709,"threshold_uncertainty_score":0.05870074},"labels":[],"label_agreement":null},{"id":"W4385572937","doi":"10.18653/v1/2022.emnlp-main.190","title":"ConvTrans: Transforming Web Search Sessions for Conversational Dense Retrieval","year":2022,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"Fundamental Research Funds for the Central Universities; Ministry of Education, India; Renmin University of China; National Natural Science Foundation of China","keywords":"Computer science; Session (web analytics); Information retrieval; Relevance (law); Focus (optics); Search engine; Human–computer information retrieval; Quality (philosophy); World Wide Web; Artificial intelligence","score_opus":0.025947049569399448,"score_gpt":0.3024997025863022,"score_spread":0.27655265301690274,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385572937","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15206027,0.0021630393,0.7645274,0.000664272,0.00045613883,0.0011203809,0.0061867153,0.059205025,0.013616837],"genre_scores_gemma":[0.54479325,0.00071536866,0.41184887,0.000850825,0.00019419317,0.0013927559,0.021076072,0.0025786925,0.016549973],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9989562,0.0003527944,0.00006915133,0.00030966333,0.00019406829,0.000117997326],"domain_scores_gemma":[0.99824286,0.0007257776,0.00008233292,0.000556775,0.00029349668,0.0000988289],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001405905,0.0013186153,0.0008181226,0.0010196617,0.00051762216,0.0008137845,0.0018365033,0.00090322475,0.004847882],"category_scores_gemma":[0.005615806,0.0005616773,0.001284157,0.0008344417,0.00058738,0.0026911383,0.0018885641,0.002002952,0.003799742],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012507393,0.001000998,0.003501234,0.0010669365,0.00023462447,0.00035364515,0.0012170769,0.050767384,0.0638243,0.005875485,0.073757686,0.79714984],"study_design_scores_gemma":[0.0001800401,0.0005501588,0.0022532565,0.000060250753,0.00009158685,0.00041105412,0.0004260466,0.9096379,0.04159832,0.0111829005,0.03350371,0.000104791456],"about_ca_topic_score_codex":0.008712018,"about_ca_topic_score_gemma":0.016046612,"teacher_disagreement_score":0.008712018,"about_ca_system_score_codex":0.0007101244,"about_ca_system_score_gemma":0.0013585615,"threshold_uncertainty_score":0.0173226},"labels":[],"label_agreement":null},{"id":"W4385572949","doi":"10.18653/v1/2022.wanlp-1.49","title":"On The Arabic Dialects’ Identification: Overcoming Challenges of Geographical Similarities Between Arabic dialects and Imbalanced Datasets","year":2022,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Windsor","funders":"Vector Institute; Compute Canada","keywords":"Arabic; Preprocessor; Computer science; Natural language processing; Artificial intelligence; Macro; Identification (biology); Task (project management); Entropy (arrow of time); Linguistics; Engineering","score_opus":0.02296168059963013,"score_gpt":0.2617181005923733,"score_spread":0.2387564199927432,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385572949","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.663888,0.010013338,0.29473436,0.009354462,0.001629494,0.0006421959,0.008050979,0.003446027,0.00824116],"genre_scores_gemma":[0.84130245,0.0016320398,0.13727196,0.0011105026,0.0008567667,0.00021221659,0.014291211,0.000232687,0.003090165],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9937152,0.0022572007,0.0004807068,0.0016579285,0.0014178512,0.00047107562],"domain_scores_gemma":[0.98912513,0.0047341743,0.0008569593,0.0023048518,0.002502386,0.00047652595],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006673339,0.0014211045,0.0018600834,0.004477846,0.002004574,0.0036891461,0.002218428,0.001859855,0.0008039242],"category_scores_gemma":[0.020177746,0.00036443007,0.0007908956,0.004846173,0.000996838,0.004630406,0.004648727,0.0018047943,0.0010455367],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018296463,0.0009806274,0.06689723,0.0007865318,0.0007377783,0.0011535881,0.0028183819,0.034199625,0.01749421,0.0102444105,0.06233197,0.800526],"study_design_scores_gemma":[0.00023709759,0.0003236672,0.058056157,0.00024031948,0.00042286486,0.0012113779,0.012713026,0.8090285,0.016225241,0.04637772,0.05502341,0.00014059938],"about_ca_topic_score_codex":0.009067715,"about_ca_topic_score_gemma":0.012664251,"teacher_disagreement_score":0.009067715,"about_ca_system_score_codex":0.0008412564,"about_ca_system_score_gemma":0.0019475012,"threshold_uncertainty_score":0.035292387},"labels":[],"label_agreement":null},{"id":"W4385573074","doi":"10.18653/v1/2022.emnlp-demos.32","title":"Camelira: An Arabic Multi-Dialect Morphological Disambiguator","year":2022,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"York University; New York University Abu Dhabi","keywords":"Arabic; Computer science; Component (thermodynamics); Modern Standard Arabic; Natural language processing; Identification (biology); Interface (matter); Artificial intelligence; Linguistics","score_opus":0.029024596473310016,"score_gpt":0.2951499694393067,"score_spread":0.2661253729659967,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385573074","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029318519,0.0021520483,0.48521385,0.0006456869,0.001096615,0.0004364692,0.016659126,0.43750986,0.026967846],"genre_scores_gemma":[0.11321948,0.00063808507,0.81327784,0.0012892478,0.0003229898,0.00036826136,0.033441305,0.015940787,0.021502012],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99930143,0.00007863423,0.00008887299,0.00027011847,0.00020887973,0.000052136536],"domain_scores_gemma":[0.99924374,0.0001706158,0.000085802305,0.00015477471,0.00025078014,0.000094376956],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007778714,0.0014695416,0.0008517003,0.0028412084,0.0010554533,0.0018748894,0.0018926599,0.0011770042,0.020486914],"category_scores_gemma":[0.0022783123,0.00073296565,0.0010494486,0.0008970961,0.000452312,0.0031622762,0.0032469435,0.0010485202,0.02202927],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015122384,0.00024109129,0.005076025,0.0013353409,0.0002577114,0.0019101333,0.0012946974,0.0015711727,0.11884053,0.010951673,0.261666,0.5953434],"study_design_scores_gemma":[0.00030558615,0.00024618243,0.0053558797,0.00026842157,0.00022298028,0.0048311,0.00059482164,0.039575826,0.15341777,0.011043006,0.7838035,0.00033488087],"about_ca_topic_score_codex":0.0012467989,"about_ca_topic_score_gemma":0.0019030754,"teacher_disagreement_score":0.020486914,"about_ca_system_score_codex":0.00046461166,"about_ca_system_score_gemma":0.0006763106,"threshold_uncertainty_score":0.06853557},"labels":[],"label_agreement":null},{"id":"W4385573366","doi":"10.18653/v1/2022.wanlp-1.9","title":"NADI 2022: The Third Nuanced Arabic Dialect Identification Shared Task","year":2022,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":54,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Alliance de recherche numérique du Canada; Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs; Georgetown University","keywords":"Arabic; Task (project management); Identification (biology); Context (archaeology); Computer science; Natural language processing; Linguistics; Artificial intelligence; History; Engineering","score_opus":0.010988478313284169,"score_gpt":0.25395594272305166,"score_spread":0.2429674644097675,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385573366","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.54481214,0.0045361747,0.14848371,0.008771071,0.0071392744,0.0099348165,0.15260938,0.0312929,0.0924205],"genre_scores_gemma":[0.44274256,0.00042390235,0.24280933,0.0034274878,0.0009944995,0.009538806,0.2573059,0.0036646598,0.039092854],"study_design_codex":"not_applicable","study_design_gemma":"observational","domain_scores_codex":[0.9888205,0.004890483,0.0006504374,0.002550235,0.0020244329,0.00106394],"domain_scores_gemma":[0.982679,0.005723555,0.00055149157,0.0040409327,0.0035624637,0.0034425338],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012006971,0.0026178018,0.0022420464,0.0024167912,0.003394597,0.004306696,0.003190388,0.0032337445,0.011652978],"category_scores_gemma":[0.024207924,0.0005925428,0.0019340944,0.0016736002,0.0011700301,0.0041879355,0.012788821,0.00488618,0.012196505],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004243729,0.0038024182,0.029166244,0.0032560057,0.0008375367,0.0019787196,0.0133792935,0.009555506,0.054750897,0.005971174,0.531178,0.34188047],"study_design_scores_gemma":[0.0022121554,0.0031550026,0.07259126,0.00061718514,0.00045670656,0.003137667,0.019559024,0.093057744,0.04481161,0.024701715,0.7349071,0.00079280644],"about_ca_topic_score_codex":0.010712243,"about_ca_topic_score_gemma":0.01844407,"teacher_disagreement_score":0.012006971,"about_ca_system_score_codex":0.0018763347,"about_ca_system_score_gemma":0.00479653,"threshold_uncertainty_score":0.06349969},"labels":[],"label_agreement":null},{"id":"W4385573420","doi":"10.18653/v1/2022.findings-emnlp.326","title":"The Curious Case of Absolute Position Embeddings","year":2022,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Canadian Institute for Advanced Research","keywords":"Computer science; Word order; Sentence; ENCODE; Position (finance); Natural language processing; Artificial intelligence; Absolute (philosophy); Question answering","score_opus":0.006132223697991322,"score_gpt":0.2647748818964915,"score_spread":0.2586426581985002,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385573420","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08539641,0.0011735826,0.89126515,0.0033696766,0.0006153007,0.000064719075,0.0011530691,0.0034500388,0.01351208],"genre_scores_gemma":[0.8312183,0.0008430423,0.15573168,0.00095555495,0.00023782377,0.00009153355,0.0014806645,0.0011530508,0.008288268],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9981495,0.00063592766,0.00013365582,0.0007400104,0.00023351485,0.00010748964],"domain_scores_gemma":[0.99055827,0.004258562,0.00046053543,0.0039262255,0.00056954595,0.00022693806],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028882446,0.0013917966,0.00079419557,0.00051503466,0.00062295893,0.0030025481,0.0018096209,0.0018689273,0.006475952],"category_scores_gemma":[0.022974636,0.0009246392,0.0007284755,0.00062248704,0.0022054098,0.012718929,0.0029229731,0.005812438,0.0027603237],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010196657,0.00023043729,0.013286943,0.0012019784,0.0003474711,0.001353423,0.0025178527,0.109091915,0.052286852,0.35253793,0.02419611,0.44192943],"study_design_scores_gemma":[0.000034596083,0.0002931375,0.0023447492,0.00013444865,0.000084097555,0.0010739518,0.0004215727,0.3996017,0.016820623,0.55383974,0.025261205,0.000090223075],"about_ca_topic_score_codex":0.0015517961,"about_ca_topic_score_gemma":0.0024867533,"teacher_disagreement_score":0.006475952,"about_ca_system_score_codex":0.00046255573,"about_ca_system_score_gemma":0.00075044774,"threshold_uncertainty_score":0.021664202},"labels":[],"label_agreement":null},{"id":"W4385573565","doi":"10.18653/v1/2022.blackboxnlp-1.22","title":"On the Compositional Generalization Gap of In-Context Learning","year":2022,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Generalization; Computer science; Parsing; Artificial intelligence; Task (project management); Context (archaeology); Generative grammar; Natural language processing; Machine learning; Benchmark (surveying); Mathematics","score_opus":0.016434348996075168,"score_gpt":0.2584125213705757,"score_spread":0.24197817237450053,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385573565","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5644477,0.009669134,0.3927682,0.008473645,0.00035624005,0.00019645883,0.0008430863,0.006406487,0.01683906],"genre_scores_gemma":[0.9429053,0.0011746171,0.05107097,0.0011886115,0.00017267422,0.00016941904,0.0010794472,0.00060155074,0.001637409],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9940156,0.0031573647,0.0002854467,0.0012732743,0.0009144178,0.00035388453],"domain_scores_gemma":[0.9514384,0.036886644,0.0010844223,0.00792301,0.0017595122,0.0009080544],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.016954271,0.0020023924,0.0017459606,0.0012576676,0.0010958115,0.0022759095,0.0020989918,0.0025630568,0.0028819004],"category_scores_gemma":[0.05273409,0.0008127642,0.001092,0.0009658962,0.0030511315,0.009847098,0.005189442,0.005809372,0.00097597553],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017601512,0.00082334125,0.024816373,0.000642326,0.0005625113,0.00046969988,0.001444976,0.5438079,0.009897571,0.03847812,0.011818323,0.36547863],"study_design_scores_gemma":[0.000078239194,0.00045016207,0.0028795882,0.000087350585,0.00009417893,0.00019605116,0.00022021717,0.9306833,0.0049927365,0.058090463,0.00218685,0.000040935472],"about_ca_topic_score_codex":0.004539289,"about_ca_topic_score_gemma":0.005519936,"teacher_disagreement_score":0.016954271,"about_ca_system_score_codex":0.0015517961,"about_ca_system_score_gemma":0.0016142842,"threshold_uncertainty_score":0.08966386},"labels":[],"label_agreement":null},{"id":"W4385573577","doi":"10.18653/v1/2022.emnlp-main.680","title":"Specializing Multi-domain NMT via Penalizing Low Mutual Information","year":2022,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"Institute for Information and Communications Technology Promotion; Ministry of Science and ICT, South Korea; Korea Advanced Institute of Science and Technology; National Research Foundation of Korea; National Research Foundation","keywords":"Computer science; Domain (mathematical analysis); Mutual information; Artificial intelligence; Machine translation; Domain model; Task (project management); Machine learning; Domain knowledge; Mathematics; Engineering","score_opus":0.010981153505409704,"score_gpt":0.24912075379717785,"score_spread":0.23813960029176814,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385573577","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03219521,0.0004667734,0.9636688,0.0003836107,0.00008474762,0.000047959824,0.000053114185,0.00079661666,0.002303284],"genre_scores_gemma":[0.6951377,0.0003473779,0.29742783,0.000624242,0.00024656192,0.00020503567,0.00052505295,0.0005270366,0.0049592666],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9989091,0.00043776672,0.000069810136,0.00026655968,0.00022785168,0.00008897199],"domain_scores_gemma":[0.99742895,0.0014396524,0.00029599603,0.00036954155,0.0003337387,0.00013210533],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002391979,0.0013876712,0.0011874003,0.00073500635,0.00067869097,0.0010395715,0.0013650735,0.0019212709,0.0018090964],"category_scores_gemma":[0.006378352,0.00047505918,0.000880764,0.0007810393,0.0011052239,0.002369503,0.00237811,0.0021773113,0.00072344876],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002642021,0.0001756029,0.0016393855,0.000253643,0.00021107546,0.0003562318,0.00018318312,0.78206235,0.020215217,0.025760356,0.005510614,0.16336815],"study_design_scores_gemma":[0.0000075751786,0.00003150788,0.000114923634,0.0000056632384,0.000012190482,0.000050665283,0.000009400646,0.9926549,0.0019221114,0.0045130677,0.0006707695,0.0000071865156],"about_ca_topic_score_codex":0.001983356,"about_ca_topic_score_gemma":0.003382891,"teacher_disagreement_score":0.002391979,"about_ca_system_score_codex":0.00086919847,"about_ca_system_score_gemma":0.001218838,"threshold_uncertainty_score":0.012650132},"labels":[],"label_agreement":null},{"id":"W4385573611","doi":"10.18653/v1/2022.emnlp-main.620","title":"Sequence Models for Document Structure Identification in an Undeciphered Script","year":2022,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada; Ministère de la Défense Nationale","keywords":"Computer science; Header; Natural language processing; Domain (mathematical analysis); Identification (biology); Artificial intelligence; Meaning (existential); Feature (linguistics); Information retrieval; Sequence (biology); Linguistics; Psychology","score_opus":0.037019790925109375,"score_gpt":0.3200339211502506,"score_spread":0.2830141302251412,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385573611","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1402631,0.0007817711,0.84926087,0.0010407191,0.000085130334,0.00017360543,0.001803649,0.001784735,0.0048064687],"genre_scores_gemma":[0.7552058,0.00057511433,0.22384089,0.0002963847,0.00016949045,0.000475476,0.004935931,0.000408706,0.014092088],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9992106,0.000332015,0.00004907797,0.00025345478,0.0000964259,0.000058300437],"domain_scores_gemma":[0.99179596,0.006283554,0.0007021696,0.00048508216,0.0005908465,0.00014249588],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021831868,0.0006412984,0.0005665087,0.0022867967,0.00061471626,0.0015338631,0.0012019475,0.0012755012,0.003728704],"category_scores_gemma":[0.009909154,0.0004814021,0.0009358088,0.0016322518,0.00085455243,0.002923985,0.00069400214,0.0018953425,0.0019790782],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004820045,0.00022338235,0.010488038,0.00027444572,0.00012143303,0.00041968474,0.0013563089,0.72372955,0.0064012078,0.106548704,0.0057198643,0.14423539],"study_design_scores_gemma":[0.000006134265,0.000014539751,0.00047906744,0.000010816119,0.000005858265,0.000030158139,0.00003031518,0.97396445,0.00044124798,0.02420513,0.00080504885,0.000007202505],"about_ca_topic_score_codex":0.009691208,"about_ca_topic_score_gemma":0.015677903,"teacher_disagreement_score":0.009691208,"about_ca_system_score_codex":0.0018174553,"about_ca_system_score_gemma":0.0011176973,"threshold_uncertainty_score":0.019269586},"labels":[],"label_agreement":null},{"id":"W4385573697","doi":"10.18653/v1/2022.emnlp-demos.22","title":"MIC: A Multi-task Interactive Curation Tool","year":2022,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Thomson Reuters (Canada)","funders":"","keywords":"Computer science; Task (project management); Annotation; Natural language processing; Sentence; Process (computing); Information retrieval; Data curation; Artificial intelligence; World Wide Web; Programming language","score_opus":0.014054830763769155,"score_gpt":0.28933937967780554,"score_spread":0.27528454891403636,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385573697","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0029721104,0.000573714,0.7714364,0.00040065366,0.00030838003,0.0004022847,0.004848495,0.21047194,0.008585938],"genre_scores_gemma":[0.043932877,0.00047534358,0.89107996,0.00074125506,0.00023922246,0.001528784,0.013009016,0.031157488,0.017836032],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9961622,0.0009173342,0.0002605325,0.00096947525,0.0014467784,0.0002436893],"domain_scores_gemma":[0.9908471,0.0056007844,0.00034941782,0.0012042035,0.00138884,0.00060966145],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003221288,0.0033018559,0.0019442211,0.0044729556,0.0017823305,0.0029348135,0.004253139,0.0027917535,0.04693813],"category_scores_gemma":[0.014123006,0.0011345991,0.0015253563,0.0019659447,0.0009821684,0.004646861,0.007029044,0.0022103777,0.02003994],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014165582,0.00022521031,0.001297293,0.0025663571,0.0003081535,0.0019826272,0.0029281157,0.009346307,0.042798366,0.013990067,0.4560911,0.46704987],"study_design_scores_gemma":[0.00033082397,0.00019886208,0.0018230511,0.00042634935,0.00010544597,0.001745903,0.0008758919,0.13547675,0.03422802,0.028882954,0.79546094,0.0004449228],"about_ca_topic_score_codex":0.002999448,"about_ca_topic_score_gemma":0.0063683665,"teacher_disagreement_score":0.04693813,"about_ca_system_score_codex":0.0008131085,"about_ca_system_score_gemma":0.0020304297,"threshold_uncertainty_score":0.15702367},"labels":[],"label_agreement":null},{"id":"W4385573952","doi":"10.18653/v1/2022.findings-emnlp.331","title":"Improving HowNet-Based Chinese Word Sense Disambiguation with Translations","year":2022,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Machine Intelligence Institute","keywords":"Computer science; WordNet; Natural language processing; Artificial intelligence; Word-sense disambiguation; Task (project management); Machine translation; Context (archaeology); Word (group theory); Knowledge base; Set (abstract data type); SemEval; Translation (biology); Linguistics","score_opus":0.007067461936243107,"score_gpt":0.23759734869048355,"score_spread":0.23052988675424044,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385573952","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.39443192,0.0032204983,0.54820997,0.0008455588,0.0010584203,0.000497114,0.0040565254,0.024889015,0.022791052],"genre_scores_gemma":[0.6293685,0.00091780594,0.3449295,0.0006219081,0.00012909503,0.00021816356,0.0110934125,0.00095980713,0.0117618535],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99870145,0.00023248175,0.00016135347,0.0005527028,0.0002384736,0.00011356748],"domain_scores_gemma":[0.9986174,0.00035804376,0.000084728024,0.0003175697,0.00054896617,0.00007329684],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015365916,0.0018520189,0.001415568,0.0031150186,0.0015403535,0.0016206793,0.0011483966,0.0007919438,0.003693032],"category_scores_gemma":[0.0033375653,0.0004956668,0.0010828839,0.0025659439,0.0008340239,0.004990614,0.002168868,0.0010215485,0.0024985643],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00054652034,0.00048768093,0.013205093,0.0007842976,0.00040470652,0.0008502942,0.0014762623,0.043537043,0.049260944,0.015075369,0.034352243,0.8400195],"study_design_scores_gemma":[0.00019790034,0.00027800305,0.0082569085,0.000105531544,0.0003615328,0.0007543172,0.0016374588,0.83496594,0.08673234,0.01655813,0.049985405,0.00016652487],"about_ca_topic_score_codex":0.019869646,"about_ca_topic_score_gemma":0.0380167,"teacher_disagreement_score":0.019869646,"about_ca_system_score_codex":0.00094119314,"about_ca_system_score_gemma":0.0032563119,"threshold_uncertainty_score":0.039507985},"labels":[],"label_agreement":null},{"id":"W4385574228","doi":"10.18653/v1/2022.sustainlp-1.11","title":"AfroLM: A Self-Active Learning-based Multilingual Pretrained Language Model for 23 African Languages","year":2022,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; Mila - Quebec Artificial Intelligence Institute","funders":"","keywords":"Computer science; Natural language processing; Simple (philosophy); Artificial intelligence; Natural language; Linguistics; Programming language; Philosophy","score_opus":0.012595687143594991,"score_gpt":0.28956684033213964,"score_spread":0.27697115318854465,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385574228","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16519931,0.0027147725,0.7840517,0.0009156737,0.00052887487,0.00031498563,0.004424274,0.034460995,0.007389341],"genre_scores_gemma":[0.64497995,0.0007928271,0.31833556,0.00070155965,0.00013761866,0.0005884704,0.015599377,0.0018627556,0.017001849],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99966633,0.000091734,0.000022682556,0.00013159802,0.00003980381,0.000047855483],"domain_scores_gemma":[0.99957496,0.00022144221,0.000018673312,0.000047371497,0.00009889313,0.00003856811],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009547158,0.0011993931,0.000697797,0.0007901788,0.0006587549,0.0009562652,0.0020397904,0.0009573847,0.00454398],"category_scores_gemma":[0.0013753243,0.00052746705,0.0011823081,0.00037958246,0.00027598997,0.0024216012,0.0016681618,0.0023306634,0.0026666818],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010151726,0.00057056453,0.003756777,0.00023798645,0.00036161643,0.00030694535,0.00033318764,0.12115776,0.0130493855,0.0031494503,0.022906473,0.8331548],"study_design_scores_gemma":[0.00006608034,0.00015886931,0.0006577801,0.00003351303,0.000077675366,0.00009923089,0.00013083185,0.98003685,0.0082562035,0.0033246258,0.007130449,0.000027976415],"about_ca_topic_score_codex":0.011426949,"about_ca_topic_score_gemma":0.020889282,"teacher_disagreement_score":0.011426949,"about_ca_system_score_codex":0.0007253748,"about_ca_system_score_gemma":0.0010923445,"threshold_uncertainty_score":0.022720873},"labels":[],"label_agreement":null},{"id":"W4385574268","doi":"10.18653/v1/2022.findings-emnlp.143","title":"How sensitive are translation systems to extra contexts? Mitigating gender bias in Neural Machine Translation models through relevant contexts.","year":2022,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; Mila - Quebec Artificial Intelligence Institute","funders":"","keywords":"Machine translation; Computer science; Inference; Metric (unit); Natural language processing; Artificial intelligence; Machine learning; Transformer; Translation (biology)","score_opus":0.08517361896420383,"score_gpt":0.2946986273082353,"score_spread":0.20952500834403148,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385574268","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7521416,0.0071926885,0.2155033,0.0029775742,0.00059608644,0.00013585949,0.0017090146,0.008735985,0.011007904],"genre_scores_gemma":[0.95131856,0.0004953597,0.043532763,0.0005580628,0.000080477126,0.00007657389,0.0016705587,0.0006953558,0.0015723131],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9938949,0.0038977545,0.00032891665,0.0009918717,0.00063130766,0.00025529915],"domain_scores_gemma":[0.9850118,0.008957896,0.0009378073,0.0032420433,0.0015690959,0.0002812719],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0064214305,0.0010853263,0.0007100188,0.00079338613,0.00071668235,0.001719333,0.0009949778,0.0011914223,0.0017344276],"category_scores_gemma":[0.039681464,0.00047972985,0.00050928124,0.00084813085,0.0009764731,0.0032881894,0.0019456272,0.0018124906,0.0012807839],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0023262657,0.00028422958,0.075331815,0.0012194893,0.00091232895,0.0006849816,0.001701319,0.15047863,0.08185249,0.011610228,0.012036457,0.66156185],"study_design_scores_gemma":[0.0002182324,0.0010530384,0.022912515,0.00027063073,0.00059214834,0.0010308633,0.00094050413,0.7718422,0.12658353,0.054083463,0.020302854,0.00017012854],"about_ca_topic_score_codex":0.0037816456,"about_ca_topic_score_gemma":0.0072198976,"teacher_disagreement_score":0.0064214305,"about_ca_system_score_codex":0.00077839335,"about_ca_system_score_gemma":0.0011772094,"threshold_uncertainty_score":0.033960164},"labels":[],"label_agreement":null},{"id":"W4385574346","doi":"10.18653/v1/2022.emnlp-main.366","title":"Rethinking Style Transformer with Energy-based Interpretation: Adversarial Unsupervised Style Transfer using a Pretrained Model","year":2022,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"Institute for Information and Communications Technology Promotion; Ministry of Science and ICT, South Korea; Korea Advanced Institute of Science and Technology; National Research Foundation of Korea; National Research Foundation","keywords":"Transformer; Computer science; Adversarial system; Style (visual arts); Energy transfer; Natural language processing; Artificial intelligence; Interpretation (philosophy); Engineering; Programming language; Art; Electrical engineering; Visual arts; Voltage","score_opus":0.017652765574851282,"score_gpt":0.24009498604147933,"score_spread":0.22244222046662804,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385574346","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.043624252,0.001385436,0.9436243,0.00055140984,0.0005541329,0.0001000892,0.00027675103,0.0041640503,0.0057196184],"genre_scores_gemma":[0.7894262,0.0007595497,0.1880907,0.001053493,0.00036478852,0.00014700761,0.0011188784,0.00090989895,0.018129427],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99949586,0.00016088999,0.000023604964,0.00016345293,0.00010272831,0.00005345218],"domain_scores_gemma":[0.9988569,0.00054049486,0.000056150715,0.00035521557,0.0001234021,0.000067782385],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012996304,0.0012971688,0.0010851723,0.00049246394,0.00037285197,0.0010988776,0.0017928759,0.0014391101,0.004353583],"category_scores_gemma":[0.0028169292,0.0005615464,0.0012843205,0.0004904958,0.0007544743,0.00182803,0.0017823889,0.0028934153,0.0025402422],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00048005322,0.0003158299,0.0015779822,0.00015437702,0.00028249982,0.00025120817,0.00021041838,0.36482662,0.023440287,0.010967712,0.014893139,0.5825999],"study_design_scores_gemma":[0.000013632196,0.000045725345,0.00011523747,0.000008881965,0.000015582924,0.000042782012,0.000009580495,0.9914302,0.0020511274,0.00552469,0.0007337346,0.000008849886],"about_ca_topic_score_codex":0.0019274061,"about_ca_topic_score_gemma":0.004085417,"teacher_disagreement_score":0.004353583,"about_ca_system_score_codex":0.00046113724,"about_ca_system_score_gemma":0.0005205545,"threshold_uncertainty_score":0.0145641565},"labels":[],"label_agreement":null},{"id":"W4385574362","doi":"10.18653/v1/2022.blackboxnlp-1.27","title":"Using Roark-Hollingshead Distance to Probe BERT’s Syntactic Competence","year":2022,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute; University of Toronto","funders":"","keywords":"Computer science; Syntax; Natural language processing; Artificial intelligence; Language model; Encoder; Conjecture; Syntactic structure; Baseline (sea); Linguistics; Philosophy; Mathematics; Discrete mathematics","score_opus":0.028982076533094547,"score_gpt":0.294284269398719,"score_spread":0.26530219286562445,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385574362","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.51505446,0.00031860024,0.45494798,0.0010197235,0.00006448841,0.00006664854,0.0010971852,0.0020260466,0.025404783],"genre_scores_gemma":[0.9318601,0.00006793864,0.06389815,0.0001999284,0.000014840163,0.000056488094,0.0010992686,0.00036657494,0.0024366498],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9990206,0.00028644712,0.00005642468,0.0003494787,0.0001922572,0.000094817],"domain_scores_gemma":[0.9930864,0.0044408846,0.00051529746,0.0011820209,0.00047433187,0.00030102272],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019015771,0.0007639825,0.00056708755,0.001040035,0.00069426844,0.0019623814,0.0010245319,0.0013011899,0.006243001],"category_scores_gemma":[0.017011771,0.0003503464,0.0005023564,0.00068326335,0.0015961895,0.0046022576,0.002427093,0.0022040755,0.0017285651],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016200133,0.00029221663,0.06824749,0.00061701355,0.00020066333,0.00073309144,0.006657172,0.08270904,0.08085704,0.24738069,0.012847059,0.49783853],"study_design_scores_gemma":[0.00006552483,0.00043263155,0.038382,0.00009159099,0.00007893656,0.00074077456,0.0016150482,0.5488057,0.026233597,0.3712003,0.012161256,0.00019275778],"about_ca_topic_score_codex":0.0026897436,"about_ca_topic_score_gemma":0.004371311,"teacher_disagreement_score":0.006243001,"about_ca_system_score_codex":0.0008073804,"about_ca_system_score_gemma":0.0007600316,"threshold_uncertainty_score":0.020884871},"labels":[],"label_agreement":null},{"id":"W4385574389","doi":"10.18653/v1/2022.findings-emnlp.91","title":"Translating Hanja Historical Documents to Contemporary Korean and English","year":2022,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"Samsung; Ministry of Science and ICT, South Korea; National Research Foundation of Korea; National Research Foundation; Samsung Advanced Institute of Technology; National Science Foundation","keywords":"Literal translation; Computer science; Natural language processing; Machine translation; Artificial intelligence; Annals; Transformer; Pace; Linguistics; History; Source text; Classics; Geography","score_opus":0.017247610474269624,"score_gpt":0.25569130156386993,"score_spread":0.2384436910896003,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385574389","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8687728,0.0031649345,0.08927678,0.0009391813,0.00090748136,0.00042894087,0.006973591,0.011615993,0.017920325],"genre_scores_gemma":[0.77574897,0.0023996104,0.15479171,0.0006016374,0.00014318545,0.00027431382,0.045891915,0.0006094136,0.019539347],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9995716,0.000094712814,0.00004616009,0.00019253412,0.00005885139,0.000036202906],"domain_scores_gemma":[0.9994149,0.00016861787,0.00004024583,0.00019748896,0.00013666712,0.000042069365],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006126977,0.00087683514,0.00041234444,0.0007299936,0.0005641991,0.0009288897,0.00078284467,0.00065644324,0.0025386396],"category_scores_gemma":[0.0022808448,0.00028428124,0.000572471,0.000790111,0.00032271573,0.0018342808,0.0008491587,0.0011453294,0.0019571707],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00062537385,0.00097096467,0.011397412,0.0010728254,0.00022400996,0.0011779143,0.0016350517,0.047490597,0.06543847,0.0031086635,0.038295753,0.8285629],"study_design_scores_gemma":[0.0003451834,0.0011222924,0.03625909,0.00037891095,0.00034121325,0.0024563891,0.0042803898,0.6252524,0.17355055,0.005817725,0.149975,0.00022080792],"about_ca_topic_score_codex":0.006877036,"about_ca_topic_score_gemma":0.020995133,"teacher_disagreement_score":0.006877036,"about_ca_system_score_codex":0.0005242727,"about_ca_system_score_gemma":0.0010039088,"threshold_uncertainty_score":0.013674021},"labels":[],"label_agreement":null},{"id":"W4385584413","doi":"10.3233/faia230162","title":"Chapter 30. Weakly Supervised Reasoning by Neuro-Symbolic Approaches","year":2023,"lang":"en","type":"book-chapter","venue":"Frontiers in artificial intelligence and applications","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta; Canadian Institute for Advanced Research","funders":"Alliance de recherche numérique du Canada; DeepMind; Natural Sciences and Engineering Research Council of Canada; National Natural Science Foundation of China; Canadian Institute for Advanced Research","keywords":"Computer science; Artificial intelligence; Connectionism; Natural language processing; Task (project management); Artificial neural network","score_opus":0.05537922583434473,"score_gpt":0.2647449643928741,"score_spread":0.20936573855852936,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385584413","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.002684533,0.03999489,0.7198606,0.0058553885,0.0010930958,0.00019179171,0.0007383162,0.0022748716,0.2273065],"genre_scores_gemma":[0.090432175,0.047248997,0.6202421,0.002467251,0.0022598726,0.0005218767,0.002559897,0.0011513364,0.23311645],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.999602,0.000081426864,0.000024000192,0.000099414865,0.00017250926,0.000020614674],"domain_scores_gemma":[0.99956685,0.0002605543,0.000015110468,0.00006656694,0.00007277518,0.000018092884],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000571261,0.00086116936,0.00044165543,0.00072199886,0.00051071437,0.0021048547,0.0011498503,0.0008620392,0.01604647],"category_scores_gemma":[0.001539246,0.00042658445,0.000711766,0.0010665165,0.0012565802,0.0028925913,0.0012186541,0.0022308212,0.007497409],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000044617056,0.000102249716,0.00018122305,0.00072304247,0.000045431534,0.000098350225,0.00024620123,0.015442621,0.0032177605,0.5008917,0.08443179,0.39457503],"study_design_scores_gemma":[0.000009462723,0.00001440381,0.00020486463,0.00025393232,0.000014918248,0.00013465996,0.000045286077,0.043817848,0.002884215,0.6551504,0.29745188,0.00001819873],"about_ca_topic_score_codex":0.0015696132,"about_ca_topic_score_gemma":0.0021026114,"teacher_disagreement_score":0.01604647,"about_ca_system_score_codex":0.0015485589,"about_ca_system_score_gemma":0.0010000675,"threshold_uncertainty_score":0.053680778},"labels":[],"label_agreement":null},{"id":"W4385599112","doi":"10.59962/9780774837804-002","title":"A Note on Romanization","year":2018,"lang":"en","type":"book-chapter","venue":"University of British Columbia Press eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Romanization; Linguistics; History; Philosophy","score_opus":0.010709681204783704,"score_gpt":0.19212114089044352,"score_spread":0.1814114596856598,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385599112","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0013763495,0.03482131,0.009851149,0.0433642,0.027731707,0.000055385557,0.00013064475,0.00039397346,0.88227534],"genre_scores_gemma":[0.0966026,0.039579242,0.015301014,0.041666944,0.034527086,0.00034904206,0.0005194055,0.0022426038,0.769212],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.991839,0.0040269853,0.00042923036,0.0010303803,0.001995074,0.0006793954],"domain_scores_gemma":[0.99682236,0.0013547508,0.00017544898,0.00045785564,0.0010444409,0.0001452081],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0049402067,0.0012869145,0.0011263363,0.0019832896,0.006682765,0.0076332935,0.0016157174,0.0022491224,0.020960117],"category_scores_gemma":[0.010987347,0.00059266365,0.0009838598,0.002444132,0.012666548,0.012200648,0.004548069,0.011702591,0.018562911],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000022794606,0.000010581659,0.000082185856,0.00009003017,0.0000027295662,0.000091177804,0.0020227341,0.00006411821,0.00010696906,0.74584997,0.22129731,0.03035948],"study_design_scores_gemma":[0.000002646982,0.000008940451,0.0000644615,0.00011105767,0.0000013859774,0.00009316937,0.00031263044,0.00005249476,0.00009162673,0.022468602,0.976785,0.000007860142],"about_ca_topic_score_codex":0.0038460495,"about_ca_topic_score_gemma":0.0038948616,"teacher_disagreement_score":0.020960117,"about_ca_system_score_codex":0.0042470307,"about_ca_system_score_gemma":0.0040095756,"threshold_uncertainty_score":0.070118606},"labels":[],"label_agreement":null},{"id":"W4385718143","doi":"10.18653/v1/2023.sigmorphon-1.13","title":"SIGMORPHON–UniMorph 2023 Shared Task 0: Typologically Diverse Morphological Inflection","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Stony Brook University","keywords":"Phonology; Inflection; Computer science; Morphology (biology); Task (project management); Unimorph; Linguistics; Natural language processing; Geology; Artificial intelligence; Engineering; Philosophy; Paleontology; Systems engineering","score_opus":0.03238077149547515,"score_gpt":0.28962444397953435,"score_spread":0.2572436724840592,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385718143","genre_codex":"dataset","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.28385374,0.007577279,0.049578357,0.005373128,0.008077118,0.0034515697,0.4493999,0.07863061,0.114058256],"genre_scores_gemma":[0.20693259,0.0007215441,0.05125014,0.0022558216,0.00063738413,0.0033609057,0.68183875,0.009644763,0.04335816],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9956928,0.0013947301,0.00039192138,0.00134283,0.0007024174,0.00047525537],"domain_scores_gemma":[0.9922321,0.002503337,0.00029158368,0.0032124424,0.00096388627,0.00079661375],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0051136394,0.006482926,0.0034558398,0.0024604858,0.003386138,0.005852671,0.0059694606,0.0061637554,0.06855005],"category_scores_gemma":[0.014058121,0.0016364143,0.002769082,0.0018184141,0.0017627673,0.010857924,0.017177908,0.0044200127,0.057475917],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0046278443,0.0013483166,0.0043702205,0.0022864374,0.00046877016,0.001823479,0.001104832,0.0017538785,0.02347496,0.0030205941,0.75147647,0.20424417],"study_design_scores_gemma":[0.005001546,0.0018523996,0.03324541,0.00078266294,0.00062413106,0.0073090894,0.0059770136,0.05250498,0.055727106,0.04985546,0.7863451,0.0007750488],"about_ca_topic_score_codex":0.010035385,"about_ca_topic_score_gemma":0.018577954,"teacher_disagreement_score":0.06855005,"about_ca_system_score_codex":0.0018298337,"about_ca_system_score_gemma":0.003514767,"threshold_uncertainty_score":0.22932279},"labels":[],"label_agreement":null},{"id":"W4385718182","doi":"10.18653/v1/2023.sigmorphon-1.24","title":"Glossy Bytes: Neural Glossing using Subword Encoding","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Phonology; Encoding (memory); Computer science; Byte; Programming language; Speech recognition; Phonetics; Natural language processing; Artificial intelligence; Linguistics; Cognitive science; Philosophy; Psychology","score_opus":0.043141374138962484,"score_gpt":0.32077019445424204,"score_spread":0.2776288203152796,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385718182","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19127223,0.008749006,0.67262894,0.0015355137,0.0039656484,0.0002978379,0.007975459,0.07149555,0.042079818],"genre_scores_gemma":[0.5783387,0.0036964284,0.35782087,0.00062689185,0.00049482146,0.0002957358,0.016269311,0.005336224,0.037121028],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9998573,0.000016938304,0.000016003525,0.0000547014,0.000035544,0.000019598396],"domain_scores_gemma":[0.9995648,0.00011792337,0.000023978617,0.00017951337,0.00008043593,0.00003335403],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00029313407,0.0009433284,0.0009219865,0.0008225474,0.00046333575,0.0015563857,0.0019869222,0.00093415985,0.020037794],"category_scores_gemma":[0.0013895207,0.00053280185,0.0006418879,0.0009771287,0.00042989312,0.0036653765,0.0019909202,0.0014718419,0.006255622],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00042655112,0.00011430608,0.00065611786,0.00031261778,0.00008977242,0.00020484501,0.00017504154,0.006785496,0.051743466,0.012864519,0.045041587,0.8815857],"study_design_scores_gemma":[0.00033721342,0.00051730225,0.0033283595,0.00021069267,0.00035675731,0.0006171066,0.00057304144,0.683888,0.13704544,0.097482584,0.075466886,0.00017664555],"about_ca_topic_score_codex":0.0028376211,"about_ca_topic_score_gemma":0.0057783173,"teacher_disagreement_score":0.020037794,"about_ca_system_score_codex":0.00037538636,"about_ca_system_score_gemma":0.0003922083,"threshold_uncertainty_score":0.06703311},"labels":[],"label_agreement":null},{"id":"W4385732132","doi":"10.2139/ssrn.4526253","title":"AI Regulations in the Context of Natural Language Processing Research","year":2023,"lang":"en","type":"article","venue":"SSRN Electronic Journal","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Context (archaeology); Natural (archaeology); Linguistics; Computer science; Natural language processing; History; Philosophy","score_opus":0.021360563099075483,"score_gpt":0.3608205395995328,"score_spread":0.33945997650045734,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385732132","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029520258,0.0173834,0.17578992,0.15299574,0.0039494983,0.00025035584,0.00065795187,0.0011100599,0.6183428],"genre_scores_gemma":[0.7719826,0.013575307,0.10396222,0.029888079,0.0055964026,0.0011679226,0.000753491,0.00073169236,0.0723423],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.97535026,0.011650436,0.0021329208,0.0030494463,0.0068111196,0.0010058274],"domain_scores_gemma":[0.8728909,0.087047,0.006737835,0.01273759,0.018315904,0.002270798],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.022566315,0.0003277533,0.00079667295,0.0031125927,0.004123749,0.013845635,0.002776335,0.004959515,0.008867968],"category_scores_gemma":[0.053131614,0.00059264444,0.0006673326,0.0047225608,0.017732203,0.009307127,0.0023890736,0.007987753,0.0026490262],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000011197031,0.0000152132425,0.00019394147,0.000054484557,0.000004731613,0.00004284045,0.0005424469,0.00023066727,0.00034925566,0.9890771,0.0033032412,0.0061749727],"study_design_scores_gemma":[0.000017413377,0.000024968811,0.0007757945,0.00020054156,0.00001782952,0.00011247281,0.00076206005,0.0016912292,0.0016241466,0.8555552,0.13919066,0.000027714786],"about_ca_topic_score_codex":0.007812547,"about_ca_topic_score_gemma":0.004250312,"teacher_disagreement_score":0.022566315,"about_ca_system_score_codex":0.004274219,"about_ca_system_score_gemma":0.010201064,"threshold_uncertainty_score":0.11934352},"labels":[],"label_agreement":null},{"id":"W4385855213","doi":"10.59962/9780774850971","title":"Musqueam Reference Grammar","year":2007,"lang":"ru","type":"book","venue":"University of British Columbia Press eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Grammar; Computer science; Linguistics; Natural language processing; Philosophy","score_opus":0.019009318614704056,"score_gpt":0.21267687684794068,"score_spread":0.19366755823323661,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385855213","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015104535,0.0034939898,0.012721016,0.007441252,0.0010606742,0.00007491315,0.0007971848,0.0006460899,0.95866036],"genre_scores_gemma":[0.5438595,0.0026679488,0.0185251,0.0032455511,0.000670821,0.00025213553,0.0019145485,0.0011265465,0.4277379],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99877363,0.00036426357,0.00006794149,0.00025103366,0.00038456896,0.00015849505],"domain_scores_gemma":[0.99904424,0.0002564658,0.00005209285,0.00014167248,0.00045864691,0.000046943453],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001012587,0.00045628697,0.00037679716,0.0013668707,0.0042805555,0.003315519,0.0016681126,0.0011680145,0.026333937],"category_scores_gemma":[0.0029093316,0.00025906783,0.0002718144,0.0017641758,0.0030986478,0.003508161,0.0020807676,0.0017944374,0.004415879],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000017743672,0.000008096922,0.00044371898,0.00009042048,0.000004508641,0.0001680095,0.012662632,0.000101273574,0.00039502216,0.8609712,0.07821817,0.046919182],"study_design_scores_gemma":[0.000002480638,0.0000074098234,0.00060200377,0.00006626713,0.0000026843563,0.000161197,0.0017782253,0.00019577442,0.0002152451,0.023445634,0.97351325,0.000009830229],"about_ca_topic_score_codex":0.04398385,"about_ca_topic_score_gemma":0.07001736,"teacher_disagreement_score":0.04398385,"about_ca_system_score_codex":0.004966223,"about_ca_system_score_gemma":0.0025406892,"threshold_uncertainty_score":0.088095784},"labels":[],"label_agreement":null},{"id":"W4385981993","doi":"10.7860/jcdr/2023/61063.18267","title":"Roles and Responsibility of the Retriever Renalogist: An Insight","year":2023,"lang":"en","type":"article","venue":"JOURNAL OF CLINICAL AND DIAGNOSTIC RESEARCH","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Labrador Retriever; Business; Medicine; Surgery","score_opus":0.171169429833047,"score_gpt":0.49758526832080435,"score_spread":0.32641583848775735,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385981993","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.054046825,0.010583209,0.086552694,0.31583777,0.0023937956,0.000112862275,0.00016863759,0.00012880196,0.53017545],"genre_scores_gemma":[0.9213803,0.0043800147,0.015087586,0.014759941,0.0012831243,0.000099891804,0.000063638945,0.000097591204,0.04284788],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99592954,0.0019751615,0.00012889365,0.0005038461,0.0006597501,0.00080270873],"domain_scores_gemma":[0.9954568,0.0023194281,0.0002902188,0.00027724795,0.00079138624,0.0008648545],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0064416383,0.0005161104,0.0003692394,0.001533755,0.004346654,0.009529463,0.0013747246,0.0046770275,0.0067775697],"category_scores_gemma":[0.0052701156,0.00047027585,0.00043407961,0.0006249368,0.017055107,0.017266518,0.005380619,0.0057931347,0.0016043633],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003244325,0.000038470484,0.001359758,0.00003485915,0.0000037106925,0.00058603584,0.02373991,0.00020950666,0.00027583656,0.9557418,0.008251252,0.009726324],"study_design_scores_gemma":[0.00003736599,0.0000615839,0.0031242464,0.00041926382,0.0000149850375,0.0023044462,0.045721643,0.0025342987,0.00040667126,0.7126928,0.23262,0.00006265042],"about_ca_topic_score_codex":0.0092030065,"about_ca_topic_score_gemma":0.006288552,"teacher_disagreement_score":0.009529463,"about_ca_system_score_codex":0.0053983703,"about_ca_system_score_gemma":0.0059854886,"threshold_uncertainty_score":0.03916818},"labels":[],"label_agreement":null},{"id":"W4386427483","doi":"10.1016/j.softx.2023.101508","title":"CoTranslate: A web-based tool for crowdsourcing high-quality sentence pair corpora","year":2023,"lang":"en","type":"article","venue":"SoftwareX","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Crowdsourcing; Computer science; Sentence; World Wide Web; Natural language processing; Quality (philosophy); Web application; Artificial intelligence","score_opus":0.027959253072589032,"score_gpt":0.3009481217690872,"score_spread":0.2729888686964982,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386427483","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017407026,0.0011582382,0.68409294,0.00163911,0.0019738104,0.0045974227,0.07722454,0.17176136,0.040145513],"genre_scores_gemma":[0.076647356,0.0005554031,0.70575917,0.0012180255,0.0006975134,0.008915515,0.14932197,0.025934031,0.030951122],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9911878,0.0037184586,0.0007760084,0.0014415617,0.0025954458,0.0002806489],"domain_scores_gemma":[0.9688917,0.013694184,0.0015006735,0.0076340195,0.0068985373,0.0013808995],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0117366165,0.0021666195,0.0013299881,0.007495178,0.0026082762,0.0028248536,0.002801837,0.0019621078,0.047830183],"category_scores_gemma":[0.04283631,0.001281651,0.0011192998,0.004523486,0.0013625199,0.0036417006,0.008786818,0.0025190893,0.034111094],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015417058,0.0005130944,0.002347175,0.0038210717,0.00048843713,0.0011565855,0.004292174,0.007644097,0.058454573,0.014867251,0.56911933,0.33575454],"study_design_scores_gemma":[0.001050708,0.00046600905,0.0051990016,0.00063143205,0.00013930362,0.00087920186,0.0023788125,0.09342148,0.052666042,0.04786336,0.79474556,0.00055903196],"about_ca_topic_score_codex":0.0043037953,"about_ca_topic_score_gemma":0.011236055,"teacher_disagreement_score":0.047830183,"about_ca_system_score_codex":0.0012288861,"about_ca_system_score_gemma":0.004339381,"threshold_uncertainty_score":0.16000795},"labels":[],"label_agreement":null},{"id":"W4386441883","doi":"10.1503/cmaj.230949-f","title":"Nouvelle politique sur l’utilisation des outils d’intelligence artificielle dans les manuscrits soumis au<i>JAMC</i>","year":2023,"lang":"fr","type":"editorial","venue":"Canadian Medical Association Journal","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Philosophy","score_opus":0.0260718365813335,"score_gpt":0.2712125235011003,"score_spread":0.24514068691976681,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386441883","genre_codex":"editorial","genre_gemma":"editorial","domain_codex":null,"domain_gemma":"reporting","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"editorial","genre_consensus":"editorial","domain_candidate":"reporting","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.000097379074,0.05638107,0.0017931008,0.16364451,0.7695567,0.000025490312,0.000085674226,0.0001258014,0.008290238],"genre_scores_gemma":[0.0019108306,0.033722833,0.0010378415,0.034913346,0.9127913,0.000042034204,0.00004687716,0.0001505633,0.015384259],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9921524,0.0018067375,0.00075458013,0.0007750118,0.004160727,0.00035044985],"domain_scores_gemma":[0.9470707,0.03965168,0.0012328139,0.000997735,0.009644058,0.0014028593],"candidate_categories":["metaresearch","research_integrity"],"consensus_categories":[],"category_scores_codex":[0.009848869,0.002024522,0.0021073953,0.0034249101,0.0031860117,0.009587612,0.0030584838,0.015462028,0.008170464],"category_scores_gemma":[0.04171089,0.00086429477,0.0019140764,0.0017828514,0.006430283,0.006422821,0.0015660997,0.026992666,0.0064824345],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000464249,0.0000069510443,0.000024045397,0.00030486708,0.000023549508,0.00007402303,0.0000527694,0.00007851206,0.00007621619,0.0087861065,0.97437954,0.016146962],"study_design_scores_gemma":[0.000024558785,0.000012718653,0.000084729865,0.00037171674,0.000027785629,0.00010814804,0.00003065065,0.00019850915,0.000113261674,0.005341528,0.9936668,0.000019551833],"about_ca_topic_score_codex":0.007969275,"about_ca_topic_score_gemma":0.017433334,"teacher_disagreement_score":0.9901511,"about_ca_system_score_codex":0.005419634,"about_ca_system_score_gemma":0.0058278115,"threshold_uncertainty_score":0.052086413},"labels":[],"label_agreement":null},{"id":"W4386488973","doi":"10.1162/tacl_a_00595","title":"<b>MIRACL</b>: A Multilingual Retrieval Dataset Covering 18 Diverse Languages","year":2023,"lang":"en","type":"article","venue":"Transactions of the Association for Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":40,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada); University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Annotation; Relevance (law); Natural language processing; Information retrieval; Process (computing); Quality (philosophy); Artificial intelligence; Resource (disambiguation); World Wide Web","score_opus":0.027441409683995673,"score_gpt":0.33102321209940083,"score_spread":0.30358180241540517,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386488973","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019041667,0.0013930508,0.0040184115,0.0005867735,0.00025876393,0.0004399314,0.95332134,0.012171822,0.008768313],"genre_scores_gemma":[0.009179716,0.00009346559,0.0065516345,0.00013320408,0.000028569024,0.00026119073,0.9817632,0.00033840246,0.0016506384],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99573225,0.0010675262,0.00074373366,0.00097341056,0.0010699106,0.00041309907],"domain_scores_gemma":[0.9931892,0.0015683466,0.00047018778,0.0013182411,0.002697692,0.00075638946],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002839488,0.0023430488,0.0015584885,0.008977263,0.0022430175,0.0031396241,0.0026531566,0.0029793864,0.023030382],"category_scores_gemma":[0.011861596,0.0007697366,0.001334622,0.0068334173,0.0009814847,0.0034484563,0.0036414978,0.001993055,0.043897513],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00050338526,0.00025965858,0.003980467,0.002816411,0.00016221908,0.00032845465,0.0004677125,0.0009957766,0.0079573365,0.0012197614,0.9553984,0.025910493],"study_design_scores_gemma":[0.00081470335,0.0003360078,0.026161341,0.0005403364,0.0001891932,0.0013124172,0.0015134165,0.011822992,0.012136758,0.0021710065,0.94268405,0.0003178157],"about_ca_topic_score_codex":0.03527757,"about_ca_topic_score_gemma":0.057656307,"teacher_disagreement_score":0.03527757,"about_ca_system_score_codex":0.0017853057,"about_ca_system_score_gemma":0.00242796,"threshold_uncertainty_score":0.07704425},"labels":[],"label_agreement":null},{"id":"W4386544688","doi":"10.1007/s11042-023-16615-z","title":"The hypergeometric test performs comparably to TF-IDF on standard text analysis tasks","year":2023,"lang":"en","type":"article","venue":"Multimedia Tools and Applications","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Prince Edward Island","funders":"","keywords":"Computer science; Test (biology); Natural language processing; Information retrieval; Artificial intelligence","score_opus":0.021239142534183048,"score_gpt":0.2965767467581373,"score_spread":0.27533760422395426,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386544688","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.39903915,0.018053474,0.49561214,0.0036171717,0.0049432055,0.0011717753,0.012741582,0.029196098,0.0356254],"genre_scores_gemma":[0.77328116,0.003762833,0.17100374,0.0007900385,0.0022458758,0.00065367256,0.029694196,0.0026214116,0.01594718],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9899401,0.0047499537,0.0009435359,0.0011942942,0.002729881,0.00044218652],"domain_scores_gemma":[0.9344703,0.05425738,0.0013625321,0.0039451937,0.0049730437,0.0009916008],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012415789,0.0021048158,0.0027933696,0.010628666,0.0013306513,0.0036120948,0.001407776,0.0023855928,0.010467688],"category_scores_gemma":[0.07745351,0.00029888627,0.001345214,0.005937131,0.0008735848,0.00555961,0.0018216417,0.0012603921,0.012508898],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0026168507,0.00069581915,0.018128002,0.0011962035,0.00085490354,0.0002271707,0.00022752065,0.013839562,0.01384609,0.0020961154,0.035144746,0.91112703],"study_design_scores_gemma":[0.0011011042,0.0036657543,0.08214404,0.00043404292,0.0010206588,0.0020031887,0.001964289,0.73414344,0.03680155,0.06398706,0.07220417,0.0005306851],"about_ca_topic_score_codex":0.004852616,"about_ca_topic_score_gemma":0.0052577015,"teacher_disagreement_score":0.012415789,"about_ca_system_score_codex":0.000723468,"about_ca_system_score_gemma":0.0021289557,"threshold_uncertainty_score":0.06566173},"labels":[],"label_agreement":null},{"id":"W4386566357","doi":"10.18653/v1/2023.loresmt-1.6","title":"Improving Neural Machine Translation of Indigenous Languages with Multilingual Transfer Learning","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Machine translation; Natural language processing; Artificial intelligence; Indigenous; Orthography; Agglutinative language; Training set; Constraint (computer-aided design); Set (abstract data type); Transfer of learning; Linguistics; Parsing; Programming language; Reading (process); Mathematics","score_opus":0.01180066918539328,"score_gpt":0.2718675698223221,"score_spread":0.2600669006369288,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386566357","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.45718265,0.0013592192,0.4886009,0.0012904562,0.0003260501,0.00020699702,0.0009535458,0.012482361,0.037597816],"genre_scores_gemma":[0.86913204,0.00041823933,0.11870349,0.0002809978,0.0000703882,0.00010327222,0.0018914236,0.00044215214,0.008957967],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9995396,0.00015176872,0.000025363486,0.00014575357,0.00007441318,0.00006303313],"domain_scores_gemma":[0.9989446,0.0004388746,0.0000800656,0.00019617281,0.00029746053,0.00004291975],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012234371,0.0011016137,0.0005644931,0.0008354747,0.00081866106,0.0012356956,0.0010369443,0.0006955416,0.0041352855],"category_scores_gemma":[0.0047650356,0.00034158473,0.0005666472,0.0010234158,0.00047680511,0.0019393492,0.0015024057,0.0015106982,0.0024547868],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00027827345,0.00041671036,0.009824357,0.00032338855,0.00028492007,0.0004384825,0.00080467894,0.3234138,0.02730475,0.009054749,0.009122634,0.6187333],"study_design_scores_gemma":[0.000029370429,0.00016695591,0.002434743,0.00004409545,0.00007959823,0.00013635683,0.00040408786,0.96179324,0.015481358,0.012667453,0.0067292447,0.000033354452],"about_ca_topic_score_codex":0.013499844,"about_ca_topic_score_gemma":0.026127206,"teacher_disagreement_score":0.013499844,"about_ca_system_score_codex":0.0008512675,"about_ca_system_score_gemma":0.001570511,"threshold_uncertainty_score":0.026842535},"labels":[],"label_agreement":null},{"id":"W4386566370","doi":"10.18653/v1/2023.vardial-1.15","title":"Dialect and Variant Identification as a Multi-Label Classification Task: A Proposal Based on Near-Duplicate Analysis","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Computer science; Classifier (UML); Identifier; Artificial intelligence; Annotation; Multi-label classification; Identification (biology); Test data; Natural language processing; Pattern recognition (psychology); Machine learning","score_opus":0.029675739753499738,"score_gpt":0.3153611143868859,"score_spread":0.28568537463338617,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386566370","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09397895,0.0013032667,0.8934806,0.0016283504,0.0003837975,0.00042979984,0.0008108985,0.004070183,0.003914188],"genre_scores_gemma":[0.39993465,0.00050016947,0.58523136,0.0009872539,0.00048359297,0.0005593257,0.0035889903,0.0009248716,0.0077897855],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98882025,0.0031508692,0.0006203456,0.0041047414,0.0026301767,0.00067355024],"domain_scores_gemma":[0.9849021,0.0046889456,0.0011582718,0.004546843,0.004056623,0.00064731087],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009645657,0.0016078746,0.002379494,0.006984501,0.002665836,0.006830855,0.004444504,0.0035579097,0.001944519],"category_scores_gemma":[0.020830462,0.00065142685,0.0025546837,0.004467602,0.0025848683,0.007479997,0.005609614,0.004552337,0.0025478872],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011428244,0.0008195995,0.04722872,0.00064984965,0.00056647393,0.0009519161,0.0039524143,0.019718485,0.036701884,0.025575886,0.017906846,0.8447851],"study_design_scores_gemma":[0.00014942583,0.00074769586,0.024851933,0.0002586152,0.0006206324,0.003223971,0.004034161,0.7400899,0.050081912,0.12283789,0.052687913,0.0004159463],"about_ca_topic_score_codex":0.006534413,"about_ca_topic_score_gemma":0.0074000326,"teacher_disagreement_score":0.009645657,"about_ca_system_score_codex":0.0017970351,"about_ca_system_score_gemma":0.0028961184,"threshold_uncertainty_score":0.05101174},"labels":[],"label_agreement":null},{"id":"W4386566401","doi":"10.18653/v1/2023.loresmt-1.9","title":"Findings from the Bambara - French Machine Translation Competition (BFMT 2023)","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Mila - Quebec Artificial Intelligence Institute","funders":"","keywords":"Machine translation; Competition (biology); Computer science; Artificial intelligence; Ecology","score_opus":0.018410524614958984,"score_gpt":0.26221658360973193,"score_spread":0.24380605899477295,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386566401","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2565412,0.05053604,0.013140545,0.1575377,0.024488384,0.0010550872,0.3206061,0.013206156,0.16288878],"genre_scores_gemma":[0.31282654,0.007176455,0.018636659,0.020585936,0.004644291,0.0010025777,0.56368417,0.0043550446,0.0670884],"study_design_codex":"not_applicable","study_design_gemma":"observational","domain_scores_codex":[0.9801122,0.008798412,0.0011344787,0.0020233355,0.006120284,0.0018112863],"domain_scores_gemma":[0.9633594,0.015093507,0.0013497025,0.0036480809,0.012514897,0.0040344526],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.027290665,0.0025090303,0.0023693938,0.0037500653,0.0049918005,0.007962579,0.003289865,0.0058449158,0.018400704],"category_scores_gemma":[0.042508896,0.0006074405,0.0013048912,0.0067938017,0.0023341419,0.0062424806,0.0069114957,0.003116894,0.018041953],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014082503,0.0004788073,0.0052220407,0.0008757895,0.0002023224,0.00046504493,0.0006759343,0.0013124833,0.0015351445,0.0019393825,0.93685573,0.04902908],"study_design_scores_gemma":[0.0018853676,0.00079503114,0.083804384,0.0008590952,0.0006514233,0.0017601955,0.0057690158,0.018313158,0.010880637,0.0144217545,0.8604183,0.00044152563],"about_ca_topic_score_codex":0.085384734,"about_ca_topic_score_gemma":0.1382003,"teacher_disagreement_score":0.085384734,"about_ca_system_score_codex":0.0034625367,"about_ca_system_score_gemma":0.005522324,"threshold_uncertainty_score":0.16977549},"labels":[],"label_agreement":null},{"id":"W4386566472","doi":"10.18653/v1/2023.findings-eacl.99","title":"Modelling Language Acquisition through Syntactico-Semantic Pattern Finding","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Vlaamse regering; Atomic Energy of Canada Limited; Fonds Wetenschappelijk Onderzoek; European Commission","keywords":"Computer science; Learnability; Language acquisition; Natural language processing; Psycholinguistics; Artificial intelligence; Abstraction; Second-language acquisition; Computational linguistics; Linguistics; Cognition; Psychology","score_opus":0.029992018433520826,"score_gpt":0.30379048579080625,"score_spread":0.2737984673572854,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386566472","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.31490555,0.0008824224,0.6686122,0.0018432323,0.000028027027,0.00013819602,0.0014770445,0.0011361658,0.010977243],"genre_scores_gemma":[0.7142111,0.0005757164,0.27837837,0.00012301237,0.000020989435,0.00028652517,0.0019533562,0.00018482057,0.004266089],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99952257,0.00019366492,0.000025336605,0.0001487948,0.000059422044,0.000050177583],"domain_scores_gemma":[0.99661833,0.002632327,0.00019003829,0.0003043585,0.00016789767,0.000087015826],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011393663,0.00051835214,0.00039117996,0.0012020321,0.00040815203,0.0015759374,0.0019013851,0.0013124608,0.0030334655],"category_scores_gemma":[0.0068069967,0.0006220366,0.0011490498,0.0009975876,0.0017230599,0.0040083188,0.0015576844,0.0017041982,0.0005111035],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012249874,0.00014603727,0.009824693,0.00015648057,0.000060999962,0.00038485567,0.0006828174,0.8241655,0.0029555776,0.11907593,0.0016128451,0.040811714],"study_design_scores_gemma":[0.000014274073,0.00001844444,0.0006417948,0.000011454522,0.000008413805,0.00004095369,0.00003776751,0.926268,0.00042612097,0.07129133,0.0012330237,0.000008455391],"about_ca_topic_score_codex":0.013286822,"about_ca_topic_score_gemma":0.021928139,"teacher_disagreement_score":0.013286822,"about_ca_system_score_codex":0.0016861124,"about_ca_system_score_gemma":0.0010777485,"threshold_uncertainty_score":0.026418984},"labels":[],"label_agreement":null},{"id":"W4386566545","doi":"10.18653/v1/2023.fieldmatters-1.7","title":"Approaches to Corpus Creation for Low-Resource Language Technology: the Case of Southern Kurdish and Laki","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"National Science Foundation","keywords":"Standardization; Computer science; Resource (disambiguation); Task (project management); Face (sociological concept); World Wide Web; Language technology; Linguistics; Natural language processing; Natural language; Engineering","score_opus":0.030085791296938513,"score_gpt":0.2709553102570255,"score_spread":0.24086951896008701,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386566545","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.22318022,0.006440341,0.66656363,0.012585325,0.0007840885,0.0021633878,0.0056597763,0.0068303538,0.07579295],"genre_scores_gemma":[0.2840302,0.0013172147,0.6956576,0.00039771074,0.00015428038,0.00093990593,0.006479033,0.0010616947,0.009962349],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99605703,0.0020636043,0.00035445104,0.00067207107,0.0006452408,0.00020760755],"domain_scores_gemma":[0.98808926,0.006594244,0.0007442597,0.002375334,0.0017573457,0.0004395215],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0061478475,0.00067349145,0.0006557388,0.00635876,0.0042272196,0.0053128987,0.0021710745,0.0013201168,0.005864757],"category_scores_gemma":[0.015428926,0.00078651647,0.00055522536,0.0058118263,0.0036843682,0.0055711954,0.006199506,0.0019173569,0.0021684223],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003210646,0.00036343726,0.0146574285,0.0032215805,0.00014695527,0.005379211,0.07959749,0.010681709,0.05209921,0.18073662,0.036432784,0.61636245],"study_design_scores_gemma":[0.00012091957,0.00020025404,0.024018092,0.0011195776,0.00012773977,0.0048644436,0.067548476,0.054714568,0.032618474,0.09746404,0.716891,0.0003123341],"about_ca_topic_score_codex":0.007654131,"about_ca_topic_score_gemma":0.017787095,"teacher_disagreement_score":0.007654131,"about_ca_system_score_codex":0.0022129556,"about_ca_system_score_gemma":0.004322235,"threshold_uncertainty_score":0.03251332},"labels":[],"label_agreement":null},{"id":"W4386566546","doi":"10.18653/v1/2023.eacl-srw.10","title":"Template-guided Grammatical Error Feedback Comment Generation","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Japan Society for the Promotion of Science; Atomic Energy of Canada Limited","keywords":"Computer science; Scope (computer science); Annotation; Cluster analysis; Modular design; Task (project management); Scheme (mathematics); Natural language processing; Artificial intelligence; Component (thermodynamics); Advice (programming); Information retrieval; Programming language","score_opus":0.07603297092259799,"score_gpt":0.3369494482451333,"score_spread":0.2609164773225353,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386566546","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08365053,0.00029633773,0.83098376,0.0008485706,0.0010758399,0.00086300034,0.0047572404,0.072120756,0.0054041045],"genre_scores_gemma":[0.35750178,0.00016858547,0.60547006,0.0004169909,0.00022008545,0.0009273104,0.017831378,0.005037575,0.012426201],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9953667,0.0018836076,0.00028811223,0.0013331434,0.00089592976,0.00023254093],"domain_scores_gemma":[0.97727966,0.0101311,0.00083373714,0.005126542,0.006197043,0.00043195486],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004605922,0.0014796068,0.0010670343,0.0013810904,0.0008386448,0.0014256482,0.003098829,0.0015615522,0.00843238],"category_scores_gemma":[0.024418375,0.00047062116,0.0012369391,0.001038679,0.00080400164,0.002134308,0.0028024442,0.0020487322,0.0074758627],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00072841323,0.00040112104,0.007332534,0.0006583572,0.00009785139,0.00071540027,0.002206852,0.03757722,0.038426757,0.006010476,0.08313019,0.82271475],"study_design_scores_gemma":[0.00021006587,0.00026083033,0.0019394718,0.000090275695,0.00008156928,0.00060846814,0.00081402546,0.8496404,0.09506321,0.009843394,0.041352823,0.00009542546],"about_ca_topic_score_codex":0.002379183,"about_ca_topic_score_gemma":0.0034780405,"teacher_disagreement_score":0.00843238,"about_ca_system_score_codex":0.00088999607,"about_ca_system_score_gemma":0.0015200336,"threshold_uncertainty_score":0.02820915},"labels":[],"label_agreement":null},{"id":"W4386566564","doi":"10.18653/v1/2023.fieldmatters-1.4","title":"Speech Database (Speech-DB) – An on-line platform for storing, validating, searching, and recording spoken language data","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Alberta","funders":"Social Sciences and Humanities Research Council of Canada; University of Alberta","keywords":"Computer science; Spoken language; Documentation; Speech corpus; Speech recognition; The Internet; Line (geometry); Speech synthesis; Database; Speech processing; Natural language processing; World Wide Web; Programming language","score_opus":0.12410023579023886,"score_gpt":0.3997001292812089,"score_spread":0.27559989349097,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386566564","genre_codex":"software","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":"software","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015107304,0.0012925005,0.2512473,0.00073611946,0.0006762399,0.0013893457,0.1479723,0.517274,0.06430489],"genre_scores_gemma":[0.11879334,0.0016134632,0.29839808,0.0014407621,0.00050727814,0.002534106,0.41757166,0.07263842,0.08650283],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9977964,0.00036567726,0.00028171053,0.00049595506,0.000906761,0.0001535445],"domain_scores_gemma":[0.993946,0.001846489,0.00034755166,0.0014240898,0.0018245729,0.000611304],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030811515,0.0015450162,0.0011545258,0.0020840208,0.0008388099,0.0030345386,0.002057913,0.0010215212,0.06477644],"category_scores_gemma":[0.008670549,0.0008023121,0.0007423182,0.0013562351,0.00071864587,0.0039590574,0.0041945805,0.0012971523,0.08586894],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021844131,0.00020515859,0.0028301666,0.0018359938,0.0002020991,0.0007898401,0.002362706,0.0011269001,0.03551299,0.011097006,0.6696077,0.27224505],"study_design_scores_gemma":[0.00037408504,0.00025196545,0.007582295,0.00036803415,0.0001123207,0.0015306202,0.0008306958,0.01567109,0.06446662,0.009864809,0.89855975,0.0003878016],"about_ca_topic_score_codex":0.008531438,"about_ca_topic_score_gemma":0.008450896,"teacher_disagreement_score":0.06477644,"about_ca_system_score_codex":0.00078400935,"about_ca_system_score_gemma":0.0024022462,"threshold_uncertainty_score":0.21669883},"labels":[],"label_agreement":null},{"id":"W4386566624","doi":"10.18653/v1/2023.eacl-main.224","title":"Weakly-Supervised Questions for Zero-Shot Relation Extraction","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Relationship extraction; Computer science; Relation (database); Template; Task (project management); Artificial intelligence; Domain (mathematical analysis); Natural language processing; Machine learning; Data mining; Information retrieval; Pattern recognition (psychology); Mathematics; Programming language","score_opus":0.03741334796121323,"score_gpt":0.33324323840984693,"score_spread":0.2958298904486337,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386566624","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022939209,0.0012875601,0.9279722,0.00066889916,0.00019553542,0.0004977157,0.0046444912,0.036226198,0.0055681],"genre_scores_gemma":[0.25421575,0.0004022769,0.6909397,0.0010127599,0.00017575578,0.0006177079,0.036825098,0.0018018814,0.014008938],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99573934,0.0012386922,0.0003390723,0.0018601561,0.0006133018,0.00020937328],"domain_scores_gemma":[0.9915001,0.0040784744,0.0003556362,0.0028802056,0.00097623654,0.00020947398],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003997505,0.0021127185,0.0014143256,0.0026872735,0.0010571631,0.0019878754,0.003711103,0.0030247287,0.012556006],"category_scores_gemma":[0.013552686,0.00073986023,0.0020384456,0.001533136,0.0012380554,0.008718832,0.0048585427,0.0034666343,0.010647687],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006082971,0.0006206832,0.0039927885,0.0011131952,0.00015321454,0.0004981789,0.0010755913,0.018665602,0.03441263,0.025205774,0.06365116,0.8500029],"study_design_scores_gemma":[0.00010299797,0.00031729572,0.0033396815,0.00019614313,0.00010185802,0.0012710397,0.00048691483,0.7404114,0.0641602,0.09606435,0.09344836,0.00009969015],"about_ca_topic_score_codex":0.002299794,"about_ca_topic_score_gemma":0.004986001,"teacher_disagreement_score":0.012556006,"about_ca_system_score_codex":0.0012957786,"about_ca_system_score_gemma":0.001250069,"threshold_uncertainty_score":0.04200405},"labels":[],"label_agreement":null},{"id":"W4386566948","doi":"10.18653/v1/2023.c3nlp-1.4","title":"Strengthening Relationships Between Indigenous Communities, Documentary Linguists, and Computational Linguists in the Era of NLP-Assisted Language Revitalization","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Indigenous; Documentation; Indigenous language; Software deployment; Computer science; Linguistics; Sociology; Artificial intelligence; Programming language; Software engineering","score_opus":0.031831888067232805,"score_gpt":0.3068022091072971,"score_spread":0.2749703210400643,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386566948","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4275752,0.009202572,0.04110674,0.4052691,0.0017441738,0.00036820685,0.000060691036,0.00042695232,0.11424633],"genre_scores_gemma":[0.9369992,0.0030201704,0.02242059,0.018437343,0.00037170594,0.000287595,0.00006123777,0.00014731086,0.018254694],"study_design_codex":"qualitative","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9758573,0.016479505,0.0006564103,0.0016592948,0.0025833808,0.0027640255],"domain_scores_gemma":[0.9577619,0.017639577,0.00244485,0.0030062273,0.008152408,0.010995169],"candidate_categories":["sts"],"consensus_categories":[],"category_scores_codex":[0.029292306,0.00038721118,0.0005651399,0.0030059558,0.022421427,0.012785312,0.002297305,0.0043497896,0.0071106013],"category_scores_gemma":[0.038747143,0.0005139954,0.00034916718,0.0018779052,0.016477894,0.018233284,0.025685143,0.007494447,0.0010746282],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007179728,0.0002492842,0.010627934,0.00032899054,0.000044007487,0.0013468488,0.8008772,0.00020361008,0.0034760425,0.06529316,0.01089572,0.1065854],"study_design_scores_gemma":[0.000022109072,0.00006948803,0.0061359685,0.0006396335,0.000039233753,0.00049319596,0.7118192,0.00085109577,0.0007497911,0.062557034,0.21652092,0.000102361555],"about_ca_topic_score_codex":0.020411095,"about_ca_topic_score_gemma":0.02978729,"teacher_disagreement_score":0.9775786,"about_ca_system_score_codex":0.0046510375,"about_ca_system_score_gemma":0.02211992,"threshold_uncertainty_score":0.15491432},"labels":[],"label_agreement":null},{"id":"W4386576650","doi":"10.18653/v1/2023.rail-1.1","title":"Automatic Spell Checker and Correction for Under-represented Spoken Languages: Case Study on Wolof","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Computer science; Spell; Natural language processing; Annotation; Lexicon; Levenshtein distance; Artificial intelligence; Focus (optics); Linguistics","score_opus":0.034231477378881395,"score_gpt":0.35078985138777996,"score_spread":0.3165583740088986,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386576650","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7841553,0.0010455113,0.19292636,0.001011848,0.00020361133,0.00027484415,0.0037904533,0.010052782,0.0065392586],"genre_scores_gemma":[0.83200085,0.00032165766,0.15858468,0.00017975175,0.00003114699,0.00009322036,0.003375611,0.0016681533,0.0037449568],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9963779,0.0013406469,0.0005060786,0.0007542649,0.00079448347,0.00022664943],"domain_scores_gemma":[0.98079944,0.011722185,0.001670569,0.0027190282,0.0027468111,0.00034201052],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002430774,0.0007060257,0.0006513272,0.0023566969,0.0014811992,0.0012881556,0.0010609941,0.00080206525,0.00306275],"category_scores_gemma":[0.019695131,0.00026616786,0.0002932115,0.0019110915,0.0013903137,0.0024188925,0.0019106742,0.00092634023,0.0012901372],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001096511,0.00020718908,0.039001945,0.0022294838,0.000089544956,0.013946792,0.036734372,0.007329785,0.10737829,0.009598896,0.016898578,0.7654886],"study_design_scores_gemma":[0.00016870243,0.00071905914,0.048737153,0.001260178,0.0002532567,0.034109328,0.041615307,0.11531341,0.42352238,0.01962426,0.31389233,0.0007846952],"about_ca_topic_score_codex":0.006606127,"about_ca_topic_score_gemma":0.009309019,"teacher_disagreement_score":0.006606127,"about_ca_system_score_codex":0.00057744276,"about_ca_system_score_gemma":0.0019171843,"threshold_uncertainty_score":0.013135374},"labels":[],"label_agreement":null},{"id":"W4386576695","doi":"10.18653/v1/2023.mwe-1.1","title":"Token-level Identification of Multiword Expressions using Pre-trained Multilingual Language Models","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of New Brunswick","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Security token; Identification (biology); Natural language processing; Artificial intelligence; Linguistics","score_opus":0.06292493646583523,"score_gpt":0.3594634093157567,"score_spread":0.29653847284992146,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386576695","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.29869297,0.0013155063,0.67239213,0.0005774871,0.0003853749,0.00029500443,0.001971529,0.014295036,0.010075085],"genre_scores_gemma":[0.7966148,0.00033319902,0.18962806,0.00030459324,0.00009169993,0.00017667502,0.005501613,0.00073639024,0.0066128965],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99865603,0.00033138043,0.00008145492,0.0006582115,0.00013455361,0.00013838925],"domain_scores_gemma":[0.9979461,0.00093974045,0.00011304208,0.00045894654,0.00044810344,0.00009404474],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015945709,0.0017700621,0.0008754097,0.00087569334,0.00063315046,0.0019299375,0.0014301266,0.0008564395,0.003870505],"category_scores_gemma":[0.005353811,0.0006330479,0.00093498884,0.00067958154,0.0005109424,0.004111366,0.0017033402,0.0020660225,0.0050326046],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010919994,0.0008185201,0.020509541,0.00040651357,0.00059667096,0.00090877526,0.0006178773,0.16514473,0.08638346,0.004494448,0.007418386,0.7116091],"study_design_scores_gemma":[0.000027664426,0.00021885897,0.0031792982,0.000031505217,0.00014173408,0.00028389142,0.00029881002,0.9440843,0.04262407,0.0044710613,0.0045656627,0.00007321496],"about_ca_topic_score_codex":0.0068736104,"about_ca_topic_score_gemma":0.011789275,"teacher_disagreement_score":0.0068736104,"about_ca_system_score_codex":0.0009875022,"about_ca_system_score_gemma":0.0015025702,"threshold_uncertainty_score":0.013667226},"labels":[],"label_agreement":null},{"id":"W4386576757","doi":"10.18653/v1/2023.fieldmatters-1","title":"Proceedings of the Second Workshop on NLP Applications to Field Linguistics","year":2023,"lang":"en","type":"paratext","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"European Regional Development Fund; Social Sciences and Humanities Research Council of Canada; European Commission; University of Alberta; National Science Foundation","keywords":"Computer science; Natural language processing; Artificial intelligence; Field (mathematics); Computational linguistics; Linguistics; Applied linguistics; Corpus linguistics; Philosophy; Mathematics","score_opus":0.02181696403742358,"score_gpt":0.3163809824572098,"score_spread":0.29456401841978624,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386576757","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014347383,0.017634299,0.7502628,0.03354713,0.023377469,0.0013525705,0.009022112,0.023267558,0.12718874],"genre_scores_gemma":[0.082033455,0.014314061,0.59948355,0.004478787,0.0058496324,0.0019748271,0.031428233,0.008700098,0.25173736],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9963876,0.0019648944,0.0002816481,0.00069691153,0.00048819496,0.00018077517],"domain_scores_gemma":[0.98551846,0.007554559,0.00019626156,0.0029808718,0.0026052743,0.0011444511],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011210666,0.0013642574,0.0015845229,0.0027156337,0.0016205134,0.007612441,0.002647675,0.0023060543,0.065064214],"category_scores_gemma":[0.016023455,0.0008425962,0.0014690364,0.0022658266,0.0023537893,0.0085371835,0.005111207,0.0041386075,0.027671471],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00038341462,0.0003417111,0.00097057066,0.00059293647,0.00010475318,0.0005281434,0.0012955446,0.002570317,0.0060781725,0.015300061,0.4345765,0.5372578],"study_design_scores_gemma":[0.00007785908,0.000078137346,0.0013985508,0.00038396617,0.000040602237,0.00040142433,0.0008830805,0.01539442,0.0033954785,0.032423176,0.9454775,0.00004586476],"about_ca_topic_score_codex":0.008037213,"about_ca_topic_score_gemma":0.010537023,"teacher_disagreement_score":0.065064214,"about_ca_system_score_codex":0.0018612427,"about_ca_system_score_gemma":0.0034512188,"threshold_uncertainty_score":0.2176615},"labels":[],"label_agreement":null},{"id":"W4386576858","doi":"10.18653/v1/2023.findings-eacl.17","title":"Translate First Reorder Later: Leveraging Monotonicity in Semantic Parsing","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"European Regional Development Fund; Universitat Politècnica de Catalunya; Atomic Energy of Canada Limited; European Commission","keywords":"Computer science; Parsing; Generalization; Monotonic function; Natural language processing; Modular design; Artificial intelligence; Principle of compositionality; Exploit; Programming language; Mathematics","score_opus":0.018620762152092494,"score_gpt":0.2807202262412388,"score_spread":0.2620994640891463,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386576858","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03740581,0.0004022875,0.9505951,0.000621458,0.0000674748,0.00012481399,0.0006878945,0.007208788,0.0028864713],"genre_scores_gemma":[0.29905853,0.00046793593,0.6886619,0.00090915366,0.00013392238,0.00030830153,0.0036106668,0.0030344194,0.003815156],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9969625,0.0014549753,0.00021263803,0.00087269134,0.00034717965,0.00015002959],"domain_scores_gemma":[0.9885012,0.006813821,0.00061794894,0.0028636958,0.0009999762,0.00020339632],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0063621565,0.0016049523,0.0011521506,0.0015867212,0.00088637817,0.0017070075,0.0020017743,0.0011595418,0.0041709235],"category_scores_gemma":[0.018015696,0.0009173528,0.0014783011,0.0016614294,0.0021624041,0.007103044,0.003304512,0.003115617,0.0030183755],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007691628,0.0004952559,0.008268104,0.00087853504,0.0001967066,0.0011784452,0.0028386205,0.075107,0.052138843,0.09708915,0.016064662,0.7449755],"study_design_scores_gemma":[0.00007705374,0.00022605462,0.0016166454,0.000120255565,0.0001346584,0.00050929753,0.00055608316,0.62302816,0.04230801,0.31500676,0.016307881,0.00010910234],"about_ca_topic_score_codex":0.0023321628,"about_ca_topic_score_gemma":0.00641104,"teacher_disagreement_score":0.0063621565,"about_ca_system_score_codex":0.00077471003,"about_ca_system_score_gemma":0.002646488,"threshold_uncertainty_score":0.033646703},"labels":[],"label_agreement":null},{"id":"W4386576877","doi":"10.18653/v1/2023.findings-eacl.160","title":"Decipherment as Regression: Solving Historical Substitution Ciphers by Learning Symbol Recurrence Relations","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Natural Sciences and Engineering Research Council of Canada; Ministère de la Défense Nationale","keywords":"Decipherment; Ciphertext; Cipher; Computer science; Artificial intelligence; Theoretical computer science; Algorithm; Natural language processing; Encryption; Linguistics; Computer security","score_opus":0.01653283092511668,"score_gpt":0.28324596139288677,"score_spread":0.26671313046777007,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386576877","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.26088202,0.00091582263,0.7305689,0.0009014199,0.00007501495,0.00012270082,0.0006116277,0.0029054254,0.0030170772],"genre_scores_gemma":[0.8456372,0.0003808687,0.14662915,0.00019800273,0.00006941733,0.00009464807,0.0012527681,0.00019768404,0.005540146],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99961984,0.00010492341,0.000023471715,0.00015617345,0.0000548281,0.00004069469],"domain_scores_gemma":[0.9983127,0.00109151,0.00020475214,0.00021622074,0.00012214713,0.000052824053],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00079473644,0.0007650641,0.000582652,0.0007240746,0.0003358931,0.0007552987,0.0010451534,0.00081379217,0.0020823],"category_scores_gemma":[0.004045748,0.00032026102,0.0008958017,0.0005735311,0.0006810435,0.0022224912,0.0006794572,0.0015523562,0.0007418805],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00033674462,0.00025543774,0.0066417563,0.00029562268,0.0001070581,0.0005987542,0.00038691764,0.632019,0.01615395,0.028307097,0.005564186,0.30933347],"study_design_scores_gemma":[0.000010871981,0.000036176305,0.0002164672,0.000007423366,0.000009795984,0.000054056916,0.000029687659,0.9888285,0.00363242,0.006541203,0.0006273407,0.000005957258],"about_ca_topic_score_codex":0.0032497717,"about_ca_topic_score_gemma":0.0046704663,"teacher_disagreement_score":0.0032497717,"about_ca_system_score_codex":0.0007760818,"about_ca_system_score_gemma":0.0010341785,"threshold_uncertainty_score":0.006965935},"labels":[],"label_agreement":null},{"id":"W4386605939","doi":"10.1080/24751839.2023.2254956","title":"The formalization of interlanguage: the example of object clitic pronouns acquisition in French L2","year":2023,"lang":"en","type":"article","venue":"Journal of Information and Telecommunication","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Trois-Rivières; Concordia University","funders":"Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada","keywords":"Interlanguage; Clitic; Linguistics; Computer science; Grammar; Object (grammar); Context (archaeology); Second-language acquisition; Natural language processing; Artificial intelligence; Psychology; History","score_opus":0.01076782327522001,"score_gpt":0.27004955524906565,"score_spread":0.25928173197384563,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386605939","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6663059,0.0017219522,0.26891065,0.0035079198,0.000067800465,0.00014230257,0.0004906284,0.00093331817,0.05791953],"genre_scores_gemma":[0.9701305,0.00018444704,0.027441675,0.00007674641,0.000017130285,0.000034518693,0.00015565734,0.00004807516,0.0019112383],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"qualitative","domain_scores_codex":[0.99883825,0.0006649373,0.00005514987,0.00018072291,0.00015434629,0.000106621446],"domain_scores_gemma":[0.9972076,0.0016183164,0.00030721832,0.00037777654,0.00041162965,0.00007744071],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014015378,0.000382402,0.00023261363,0.0019314177,0.0012353137,0.0025968319,0.0005211132,0.0010519285,0.0030117389],"category_scores_gemma":[0.0028548613,0.00026084838,0.0003497597,0.0012130857,0.004576064,0.002868454,0.0011200012,0.0010465253,0.0003296643],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010309529,0.000102245205,0.017198613,0.00036365207,0.000034745703,0.004331738,0.066637345,0.00252948,0.041718174,0.79346746,0.0019155773,0.071597815],"study_design_scores_gemma":[0.00008133784,0.00037288826,0.10061532,0.0005886503,0.00010157442,0.011647076,0.06268036,0.08221629,0.057677336,0.42843568,0.25521427,0.00036919056],"about_ca_topic_score_codex":0.010028083,"about_ca_topic_score_gemma":0.009545224,"teacher_disagreement_score":0.010028083,"about_ca_system_score_codex":0.0017791298,"about_ca_system_score_gemma":0.0011467251,"threshold_uncertainty_score":0.019939423},"labels":[],"label_agreement":null},{"id":"W4386694320","doi":"10.48550/arxiv.2309.04919","title":"The Emergence of Chunking Structures with Hierarchical RNN","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"University of Alberta","keywords":"Chunking (psychology); Computer science; Artificial intelligence; Natural language processing; Parsing; Sentence; Recurrent neural network; Conditional random field; Task (project management); Phrase; Artificial neural network","score_opus":0.058031621697777085,"score_gpt":0.21009602886827874,"score_spread":0.15206440717050165,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386694320","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06492671,0.00035547427,0.9306045,0.00020302748,0.0000361806,0.000055007073,0.000108788496,0.0019881406,0.0017220303],"genre_scores_gemma":[0.6567208,0.00023657066,0.33941397,0.00014633944,0.000037561076,0.00013309778,0.00041700833,0.0002817077,0.0026129247],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995572,0.00017464282,0.000024103985,0.00014309748,0.00006486785,0.000036155598],"domain_scores_gemma":[0.9978362,0.0012454342,0.00019178074,0.00041921216,0.00025961298,0.000047834157],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010649064,0.00056431716,0.00038087287,0.00035388532,0.00035307405,0.0004302697,0.001043807,0.0005507258,0.00072959857],"category_scores_gemma":[0.0046624555,0.0005613027,0.0003979245,0.000403768,0.0007189449,0.0017661386,0.000843344,0.0012868048,0.00039985467],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002623562,0.000119863485,0.003574486,0.0002436974,0.0001492155,0.00035402502,0.0010967153,0.49413112,0.07942618,0.031793445,0.0039225384,0.38492632],"study_design_scores_gemma":[0.0000047362964,0.000026204885,0.00034388155,0.000006298134,0.0000096128815,0.000019268318,0.000016993565,0.9814683,0.0055363546,0.012020281,0.00054049544,0.000007474377],"about_ca_topic_score_codex":0.005728427,"about_ca_topic_score_gemma":0.009950676,"teacher_disagreement_score":0.005728427,"about_ca_system_score_codex":0.0007344629,"about_ca_system_score_gemma":0.0006184652,"threshold_uncertainty_score":0.01139015},"labels":[],"label_agreement":null},{"id":"W4386702871","doi":"10.13052/rp-9788770040723.010","title":"An Insight into Neural Machine Translation","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Bioethics Society; Dalhousie University","funders":"","keywords":"Computer science; Machine translation; Translation (biology); Artificial intelligence; Chemistry","score_opus":0.019529600129668333,"score_gpt":0.3015111298683142,"score_spread":0.2819815297386459,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386702871","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008314824,0.01572426,0.842941,0.015216394,0.0008265498,0.00004532889,0.0004460952,0.00036588399,0.11611969],"genre_scores_gemma":[0.4891145,0.028289562,0.3941176,0.004778796,0.0023052364,0.00039221635,0.0007062999,0.0003319083,0.079963855],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9996469,0.0001354679,0.000020917993,0.00008369392,0.00008550798,0.000027453292],"domain_scores_gemma":[0.9995408,0.00027305615,0.000026580534,0.00007427927,0.00006868202,0.00001664278],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007729679,0.0004609261,0.00034953587,0.00086132117,0.0005612216,0.001657717,0.00089476997,0.0013660502,0.008295183],"category_scores_gemma":[0.002492812,0.00030870634,0.00046006747,0.0010249116,0.0018511086,0.003978363,0.0010008677,0.0022296836,0.001801263],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000013686481,0.0000106640855,0.00013464177,0.00008958441,0.000009037537,0.000053537595,0.00010963578,0.006560555,0.0006274278,0.9561803,0.0029091581,0.033301663],"study_design_scores_gemma":[0.000004189398,0.000011591439,0.00014914543,0.000053225158,0.0000039207557,0.00008993793,0.000026822569,0.037473075,0.00039115717,0.9414977,0.020291457,0.000007851363],"about_ca_topic_score_codex":0.0013811233,"about_ca_topic_score_gemma":0.0013296595,"teacher_disagreement_score":0.008295183,"about_ca_system_score_codex":0.0010765248,"about_ca_system_score_gemma":0.0005566415,"threshold_uncertainty_score":0.027750134},"labels":[],"label_agreement":null},{"id":"W4386803989","doi":"10.1007/978-3-031-43418-1_35","title":"Exploring Word-Sememe Graph-Centric Chinese Antonym Detection","year":2023,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Innovation Cluster (Canada)","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Ambiguity; Leverage (statistics); Transitive relation; Inference; Graph; Word (group theory); Information retrieval; Theoretical computer science; Linguistics; Mathematics","score_opus":0.035016672826055646,"score_gpt":0.26338349275181117,"score_spread":0.22836681992575553,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386803989","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7413822,0.0037953367,0.2179618,0.00058903947,0.00054496754,0.0003902716,0.006203178,0.007542067,0.021591231],"genre_scores_gemma":[0.8490093,0.001024864,0.13076249,0.00020927461,0.00017810026,0.00012891168,0.010475419,0.00065697095,0.007554597],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99936885,0.00009685131,0.000047515274,0.00024286847,0.00014162408,0.00010220966],"domain_scores_gemma":[0.9990396,0.00038181472,0.00010237263,0.000111060865,0.00029812386,0.000067011206],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00047304106,0.0010956972,0.00082852633,0.004318594,0.0008914711,0.0012453893,0.0010069947,0.000612813,0.0049420283],"category_scores_gemma":[0.0019390086,0.00028133288,0.00077553914,0.0044883336,0.00041833005,0.0018978574,0.0016704306,0.00064412376,0.002619908],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012569614,0.00029902082,0.020880016,0.0013996045,0.000267415,0.002691616,0.0011057152,0.0034309053,0.18758683,0.0077341204,0.016987344,0.7563605],"study_design_scores_gemma":[0.00019546687,0.00087143463,0.09945017,0.00021284993,0.0008397987,0.0061503155,0.004859283,0.64635736,0.15238681,0.03277716,0.055722468,0.00017684289],"about_ca_topic_score_codex":0.0054479414,"about_ca_topic_score_gemma":0.010066857,"teacher_disagreement_score":0.0054479414,"about_ca_system_score_codex":0.00041064472,"about_ca_system_score_gemma":0.0011716807,"threshold_uncertainty_score":0.01653272},"labels":[],"label_agreement":null},{"id":"W4386952323","doi":"10.1109/iceccme57830.2023.10252703","title":"BenCo: First Step Towards Coreference Resolution in Bengali","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Bengali; Coreference; Computer science; Natural language processing; Artificial intelligence; Resolution (logic); Domain (mathematical analysis); Language understanding","score_opus":0.031141013480909634,"score_gpt":0.29002267784012636,"score_spread":0.2588816643592167,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386952323","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.49086455,0.0068719275,0.26039904,0.0040344386,0.0013622718,0.0013064158,0.07380388,0.08141299,0.07994453],"genre_scores_gemma":[0.5974825,0.00095126266,0.24666744,0.0007950577,0.00013458569,0.00063808047,0.12405774,0.0027093708,0.026563989],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99673456,0.0010739507,0.00020851493,0.001149812,0.00045926363,0.00037390072],"domain_scores_gemma":[0.9972038,0.00074429734,0.0001113377,0.0010030643,0.00079306855,0.00014448883],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015981566,0.0013683406,0.0009844712,0.0029045232,0.0032419763,0.0025957997,0.0021780636,0.0014817994,0.0072998093],"category_scores_gemma":[0.00744065,0.00044307482,0.00089838775,0.0026769852,0.00080655847,0.0024492072,0.003953386,0.0017394447,0.00738231],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0022406715,0.000508018,0.023619458,0.001919569,0.00058871857,0.0021130184,0.008892196,0.013279943,0.094677635,0.011895641,0.1533175,0.6869477],"study_design_scores_gemma":[0.00018146411,0.00031731615,0.04296195,0.00034567225,0.00040308636,0.0017456396,0.011684304,0.16018404,0.21456343,0.01678591,0.5505507,0.0002764086],"about_ca_topic_score_codex":0.073064916,"about_ca_topic_score_gemma":0.13086382,"teacher_disagreement_score":0.073064916,"about_ca_system_score_codex":0.0023729412,"about_ca_system_score_gemma":0.0021330216,"threshold_uncertainty_score":0.14527929},"labels":[],"label_agreement":null},{"id":"W4387227690","doi":"10.1017/9781108178501.006","title":"Bilingual Lexical and Conceptual Memory","year":2023,"lang":"en","type":"book-chapter","venue":"Cambridge University Press eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Wilfrid Laurier University","funders":"","keywords":"Connectionism; Lexicon; Computer science; Set (abstract data type); Mental lexicon; Word (group theory); Natural language processing; Association (psychology); Lexical item; Linguistics; Artificial intelligence; Psychology; Artificial neural network","score_opus":0.02916885817911167,"score_gpt":0.22670786262792073,"score_spread":0.19753900444880906,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4387227690","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022028506,0.05543711,0.020729523,0.009638216,0.00071713957,0.00002805476,0.0004128284,0.00030030482,0.8907083],"genre_scores_gemma":[0.5694724,0.051523637,0.016001798,0.004251566,0.0010189196,0.00009841831,0.0012229339,0.00028081457,0.3561296],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99980694,0.000045729197,0.0000112399875,0.00004986461,0.00006135956,0.000024906294],"domain_scores_gemma":[0.9997316,0.00012855377,0.000014800802,0.00004138493,0.000045356934,0.00003820952],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005553714,0.00031724825,0.00029540402,0.00080921326,0.0005716319,0.002451234,0.00043398506,0.000662759,0.023735646],"category_scores_gemma":[0.0011821784,0.0001563773,0.00020667948,0.0010490738,0.0014788109,0.0039368486,0.0012422629,0.0009041753,0.0033473917],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000056129884,0.000028836643,0.00069624884,0.000340931,0.00001364932,0.00021839913,0.0013931246,0.00032070308,0.0010587741,0.75907105,0.034774847,0.20202719],"study_design_scores_gemma":[0.00002036565,0.000044549786,0.002928087,0.00025303374,0.000021408208,0.0011815291,0.0009781317,0.0012388786,0.0011401945,0.48196185,0.51021236,0.000019591804],"about_ca_topic_score_codex":0.003677282,"about_ca_topic_score_gemma":0.006840648,"teacher_disagreement_score":0.023735646,"about_ca_system_score_codex":0.0016451965,"about_ca_system_score_gemma":0.0015287459,"threshold_uncertainty_score":0.07940364},"labels":[],"label_agreement":null},{"id":"W4387331844","doi":"10.7202/1106330ar","title":"The leverage of “text type” on translation choices: An empirical study with a logical focus","year":2023,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Typology; Linguistics; Focus (optics); Computer science; Leverage (statistics); Meaning (existential); Relation (database); Source text; Natural language processing; Systemic functional linguistics; Translation studies; Psychology; Artificial intelligence; Sociology; Philosophy","score_opus":0.09123256728144732,"score_gpt":0.3468116958673568,"score_spread":0.2555791285859095,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4387331844","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9931553,0.0001090986,0.0022495806,0.00020063815,0.00001110934,0.00007751056,0.000043830314,0.0000096637605,0.004143351],"genre_scores_gemma":[0.99743134,0.00007391025,0.0017560432,0.000076466946,0.000012606601,0.0000812299,0.00004604197,0.000024967116,0.0004973297],"study_design_codex":"qualitative","study_design_gemma":"observational","domain_scores_codex":[0.9748926,0.020090973,0.0013134304,0.0013587653,0.0018670093,0.00047715523],"domain_scores_gemma":[0.7417031,0.23195721,0.014316756,0.004726341,0.0061084004,0.0011881798],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.018699808,0.00038651857,0.0003997011,0.0012427298,0.0019004559,0.0033681008,0.00072689337,0.001026218,0.0044484283],"category_scores_gemma":[0.12507217,0.0003631896,0.00033794768,0.002059457,0.0039434135,0.0054680635,0.0021858208,0.001669766,0.00062957435],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0028498417,0.0019744071,0.25411886,0.001318287,0.00015486171,0.00294988,0.618369,0.0008364244,0.012358648,0.011927441,0.001408926,0.09173349],"study_design_scores_gemma":[0.0007214191,0.0034606312,0.3083945,0.0010241169,0.0006065985,0.0044405963,0.60211426,0.0138578005,0.016704427,0.017252494,0.03103343,0.00038982314],"about_ca_topic_score_codex":0.0012725773,"about_ca_topic_score_gemma":0.0015254115,"teacher_disagreement_score":0.018699808,"about_ca_system_score_codex":0.0012658848,"about_ca_system_score_gemma":0.0011659646,"threshold_uncertainty_score":0.09889525},"labels":[],"label_agreement":null},{"id":"W4387331995","doi":"10.7202/1106331ar","title":"La néologie de forme en traductologie : une étude outillée de la revue Meta 1966-2019","year":2023,"lang":"fr","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Humanities; Philosophy; Art","score_opus":0.07210853275883002,"score_gpt":0.33517744611449923,"score_spread":0.26306891335566923,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4387331995","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.77219224,0.051244482,0.009694524,0.008615473,0.0007391929,0.00027222667,0.0066359728,0.0001452801,0.15046059],"genre_scores_gemma":[0.9360568,0.015357903,0.0050357557,0.0010789677,0.00029872966,0.00032254079,0.002630618,0.00033201886,0.038886726],"study_design_codex":"qualitative","study_design_gemma":"observational","domain_scores_codex":[0.99578154,0.0014383394,0.00045319996,0.0006774869,0.0013739378,0.000275547],"domain_scores_gemma":[0.97530156,0.016118608,0.0025100526,0.0018107599,0.0036639331,0.00059502904],"candidate_categories":["bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.008318945,0.0004026512,0.0004655691,0.010761183,0.0034666853,0.0059123384,0.00068525167,0.0009470865,0.0083068255],"category_scores_gemma":[0.017728342,0.00043019545,0.00034148656,0.015432331,0.006162201,0.0070204465,0.0028700458,0.0015646697,0.0013216829],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005171141,0.000074519805,0.052227944,0.003551261,0.00007704158,0.0012742988,0.54931855,0.00037906948,0.009697809,0.10978225,0.013382209,0.25971797],"study_design_scores_gemma":[0.00001708609,0.000088630186,0.100174755,0.002359795,0.000051573927,0.0007304881,0.14300689,0.00019851189,0.0033622007,0.0050127613,0.74492323,0.00007403168],"about_ca_topic_score_codex":0.039073,"about_ca_topic_score_gemma":0.07423342,"teacher_disagreement_score":0.9892388,"about_ca_system_score_codex":0.010593864,"about_ca_system_score_gemma":0.0066349157,"threshold_uncertainty_score":0.07769114},"labels":[],"label_agreement":null},{"id":"W4387332007","doi":"10.7202/1106325ar","title":"Analyse outillée des décalages informationnels dans l’évaluation de la qualité de la traduction de l’ensemble des lois codifiées du Québec","year":2023,"lang":"fr","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":true,"ca_institutions":"Université du Québec à Montréal; Université du Québec à Trois-Rivières","funders":"","keywords":"Humanities; Philosophy; Lexicography; Linguistics","score_opus":0.050637200465041304,"score_gpt":0.32658858392150447,"score_spread":0.2759513834564632,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4387332007","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.94009817,0.003475642,0.0151265785,0.0012550979,0.00008974396,0.00012694512,0.012916934,0.00043843855,0.026472308],"genre_scores_gemma":[0.95871407,0.00086932967,0.012142046,0.00015331662,0.000027611932,0.00013053177,0.010094826,0.00017865447,0.017689705],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99664503,0.00071830733,0.00027994998,0.00061964674,0.0014786203,0.00025854056],"domain_scores_gemma":[0.9741746,0.008479302,0.0012398332,0.0010542825,0.014684437,0.0003674715],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035224224,0.00050045893,0.0005081715,0.0055503515,0.0019646774,0.0036958929,0.00062235387,0.00063852227,0.004300727],"category_scores_gemma":[0.017492577,0.00035542372,0.00033747553,0.008022692,0.0011955024,0.0014430191,0.0013034218,0.00072619406,0.0008505114],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009510325,0.0000777969,0.42888576,0.0020722798,0.00070617266,0.0011588417,0.054148015,0.0076831547,0.041677844,0.0071878782,0.025532648,0.42991865],"study_design_scores_gemma":[0.000022563654,0.00009640384,0.8348105,0.0005937536,0.0003248164,0.00047124317,0.025545975,0.017721198,0.02090814,0.0011267172,0.09820218,0.00017663263],"about_ca_topic_score_codex":0.7755398,"about_ca_topic_score_gemma":0.8818475,"teacher_disagreement_score":0.22446018,"about_ca_system_score_codex":0.009176479,"about_ca_system_score_gemma":0.007279695,"threshold_uncertainty_score":0.45156413},"labels":[],"label_agreement":null},{"id":"W4387412204","doi":"10.2196/50814","title":"The Use of Machine Translation for Outreach and Health Communication in Epidemiology and Public Health: Scoping Review","year":2023,"lang":"en","type":"article","venue":"JMIR Public Health and Surveillance","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"Leibniz-Gemeinschaft","keywords":"Outreach; Generalizability theory; Scopus; Systematic review; Public health; Population; Population health; Knowledge translation; MEDLINE; Computer science; Medical education; Medicine; Knowledge management; Psychology; Environmental health; Political science; Nursing","score_opus":0.26484807324605464,"score_gpt":0.4451489610399774,"score_spread":0.18030088779392278,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4387412204","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00066602323,0.9924434,0.0017409442,0.0012013317,0.00054817984,0.0015537221,0.00042306117,0.000027929094,0.0013954676],"genre_scores_gemma":[0.0070538763,0.9854395,0.0034941006,0.0006532307,0.00023308236,0.0026079488,0.00031350483,0.000019339088,0.00018550914],"study_design_codex":"systematic_review","study_design_gemma":"systematic_review","domain_scores_codex":[0.93538266,0.02637564,0.0258795,0.0023642168,0.009223085,0.00077488605],"domain_scores_gemma":[0.7209584,0.23209508,0.022223191,0.0054065944,0.018601023,0.00071561075],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0558478,0.0022935169,0.007280839,0.040455244,0.0023098302,0.0074801557,0.0030697566,0.0047289883,0.005741284],"category_scores_gemma":[0.22461593,0.0017326849,0.008021595,0.037033964,0.00338255,0.008186183,0.0049920776,0.0024346518,0.0010208805],"study_design_candidate":"systematic_review","study_design_consensus":"systematic_review","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007958729,0.000024733768,0.000612075,0.84359294,0.0022398587,0.00016130318,0.0010601695,0.0002223722,0.0001834346,0.0016654853,0.0033372222,0.1468209],"study_design_scores_gemma":[0.000020771875,0.00004410514,0.00067433,0.9652349,0.00552157,0.00016509242,0.0004334928,0.00009679669,0.00014484237,0.00072367856,0.026919011,0.000021564425],"about_ca_topic_score_codex":0.009070852,"about_ca_topic_score_gemma":0.0153311035,"teacher_disagreement_score":0.0558478,"about_ca_system_score_codex":0.0067951526,"about_ca_system_score_gemma":0.038245883,"threshold_uncertainty_score":0.29535496},"labels":[],"label_agreement":null},{"id":"W4387674059","doi":"10.1145/3622869","title":"Saggitarius: A DSL for Specifying Grammatical Domains","year":2023,"lang":"en","type":"article","venue":"Proceedings of the ACM on Programming Languages","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Advanced Research Projects Agency; Defense Advanced Research Projects Agency; U.S. Department of Defense","keywords":"Computer science; Parsing; Rule-based machine translation; Suite; Natural language processing; Programming language; Grammar; Set (abstract data type); SQL; Artificial intelligence; Domain (mathematical analysis); Digital subscriber line; Context (archaeology); Linguistics","score_opus":0.025292021713337907,"score_gpt":0.30766713401946005,"score_spread":0.28237511230612217,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4387674059","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.002824766,0.00014737109,0.9040072,0.0003587888,0.000090752525,0.00022991165,0.011258008,0.0768127,0.004270581],"genre_scores_gemma":[0.05173205,0.0005725779,0.8813773,0.0009959618,0.000076324424,0.0010224577,0.028195383,0.02859245,0.0074354997],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99871707,0.0002958062,0.0002623646,0.00027526272,0.00035692612,0.000092597016],"domain_scores_gemma":[0.9965624,0.0019939204,0.00025014655,0.0006981908,0.00037247245,0.00012288414],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002337748,0.0016109239,0.0009874932,0.002128734,0.0008908046,0.0031770938,0.0027445403,0.0016629131,0.01416006],"category_scores_gemma":[0.0064878208,0.002348946,0.002737839,0.0013848599,0.0017260342,0.0045774216,0.0031610574,0.0035497588,0.0064219595],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005481912,0.00020091096,0.007263327,0.0024466258,0.00023608525,0.0010352456,0.0028410563,0.06251957,0.023067772,0.4392267,0.21430059,0.24631388],"study_design_scores_gemma":[0.00016513988,0.000092665825,0.00076511677,0.00030271735,0.00007150854,0.0010201144,0.00035123967,0.16361798,0.025181493,0.14954178,0.6587319,0.00015831958],"about_ca_topic_score_codex":0.0055210902,"about_ca_topic_score_gemma":0.011002524,"teacher_disagreement_score":0.01416006,"about_ca_system_score_codex":0.0016885189,"about_ca_system_score_gemma":0.0029149724,"threshold_uncertainty_score":0.047370136},"labels":[],"label_agreement":null},{"id":"W4387859680","doi":"10.1002/pra2.889","title":"Translating Research into Practice: Plain Language and Writing for Machine Translation Guidelines","year":2023,"lang":"en","type":"article","venue":"Proceedings of the Association for Information Science and Technology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Narodowa Agencja Wymiany Akademickiej","keywords":"Plain language; Machine translation; Computer science; Plain English; Natural language processing; Data science; Artificial intelligence; World Wide Web; Linguistics","score_opus":0.04354848267276326,"score_gpt":0.3977433497088849,"score_spread":0.3541948670361217,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4387859680","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009441473,0.0017037899,0.8840163,0.041719396,0.0052707735,0.004084713,0.0019237133,0.008046787,0.043792963],"genre_scores_gemma":[0.04948178,0.0016057151,0.9300522,0.0030199294,0.0009261156,0.0029457489,0.0014938076,0.0018553232,0.008619472],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.87514234,0.08581152,0.021521322,0.0024605358,0.0141862035,0.0008779777],"domain_scores_gemma":[0.58510476,0.24317984,0.021762062,0.03323352,0.11335131,0.0033685688],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.07554867,0.0013975245,0.0011236827,0.0057201847,0.0029792278,0.009919353,0.0025091544,0.003944412,0.01073777],"category_scores_gemma":[0.2775521,0.0012759118,0.0006151185,0.0036941706,0.00410775,0.0073431632,0.004403948,0.004790816,0.009071037],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002324503,0.00027683747,0.0013855728,0.005683452,0.000055107364,0.0011848557,0.03110749,0.004241798,0.011085267,0.21644098,0.3617027,0.36660343],"study_design_scores_gemma":[0.00008621537,0.00028774957,0.0011800629,0.005187986,0.000060172937,0.00075091963,0.007881681,0.011350742,0.010134294,0.116247505,0.84662324,0.00020950746],"about_ca_topic_score_codex":0.0025291857,"about_ca_topic_score_gemma":0.0051056016,"teacher_disagreement_score":0.07554867,"about_ca_system_score_codex":0.0031560715,"about_ca_system_score_gemma":0.009430626,"threshold_uncertainty_score":0.39954436},"labels":[],"label_agreement":null},{"id":"W4387896177","doi":"10.54254/2755-2721/13/20230734","title":"The development and advance of machine translation","year":2023,"lang":"en","type":"article","venue":"Applied and Computational Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Machine translation; Computer science; Translation (biology); Example-based machine translation; Artificial intelligence; Evaluation of machine translation; Natural language processing; Key (lock); Machine translation software usability; Machine learning; Transfer-based machine translation; Encoder","score_opus":0.006754346068365035,"score_gpt":0.21969552781179624,"score_spread":0.2129411817434312,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4387896177","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01974192,0.22381434,0.6390841,0.018644169,0.0051847575,0.00033909915,0.0009066892,0.0022060007,0.09007895],"genre_scores_gemma":[0.22472635,0.16929631,0.5574814,0.0047409455,0.007745252,0.0005008559,0.002370638,0.0007549396,0.03238332],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9966613,0.0013269503,0.0002580632,0.00060094666,0.0010152631,0.00013751129],"domain_scores_gemma":[0.9944159,0.0025864209,0.00027291264,0.0009900222,0.0015675277,0.00016729145],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0043648183,0.0008796075,0.00089882227,0.002256695,0.0007772716,0.0029901993,0.0014052197,0.0021495433,0.006454417],"category_scores_gemma":[0.011936817,0.00055874937,0.0008511348,0.003302631,0.0019917968,0.0073409597,0.0023374578,0.0027352064,0.005037215],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001422367,0.0000715544,0.0009264085,0.002111295,0.00005892757,0.00024138491,0.0005491889,0.0053630914,0.009198815,0.22043811,0.01856886,0.7423302],"study_design_scores_gemma":[0.000037038255,0.0003256314,0.0023741801,0.0010243878,0.000064825304,0.0012652305,0.00045256029,0.053881314,0.012769684,0.22332428,0.70436513,0.00011568558],"about_ca_topic_score_codex":0.0013240599,"about_ca_topic_score_gemma":0.0006236212,"teacher_disagreement_score":0.006454417,"about_ca_system_score_codex":0.001456054,"about_ca_system_score_gemma":0.00247749,"threshold_uncertainty_score":0.023083627},"labels":[],"label_agreement":null},{"id":"W4387934025","doi":"10.2139/ssrn.4612990","title":"Open-Vocabulary Object Detection Via Debiased Curriculum Self-Training","year":2023,"lang":"en","type":"preprint","venue":"SSRN Electronic Journal","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Vocabulary; Curriculum; Training (meteorology); Artificial intelligence; Object (grammar); Natural language processing; Psychology; Machine learning; Pedagogy; Linguistics","score_opus":0.01889662155245932,"score_gpt":0.2883796717198865,"score_spread":0.2694830501674272,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4387934025","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.36841476,0.00049451727,0.60832214,0.00041638917,0.00025600716,0.00020815358,0.000502599,0.010174448,0.011210918],"genre_scores_gemma":[0.8227377,0.000115796436,0.16435093,0.000213601,0.000048059243,0.000117451826,0.0010253494,0.00025585058,0.011135243],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993919,0.00007943324,0.000026378631,0.00024789353,0.00012355996,0.00013085417],"domain_scores_gemma":[0.99894315,0.00031056753,0.00007613035,0.0002506882,0.00032439517,0.000095116215],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00071470544,0.0005216497,0.00062261167,0.0007985248,0.00040342117,0.0006652707,0.001153748,0.0008940524,0.0053731897],"category_scores_gemma":[0.002718219,0.00022187752,0.00034553919,0.00054375984,0.00030730446,0.0012713955,0.0020583724,0.0008220514,0.0027714926],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003391241,0.00028596338,0.0038508906,0.000068916386,0.000026587277,0.00006759051,0.00013000128,0.0048073926,0.055145603,0.0010147647,0.0030185601,0.93124455],"study_design_scores_gemma":[0.00009681984,0.00070887245,0.016484035,0.00004846246,0.00008903382,0.00046257564,0.00042569274,0.82302016,0.14263724,0.0066530644,0.009323286,0.000050870727],"about_ca_topic_score_codex":0.0026916629,"about_ca_topic_score_gemma":0.0038773115,"teacher_disagreement_score":0.0053731897,"about_ca_system_score_codex":0.00032952506,"about_ca_system_score_gemma":0.00081779447,"threshold_uncertainty_score":0.017975092},"labels":[],"label_agreement":null},{"id":"W4388044549","doi":"10.5539/elt.v16n11p83","title":"Actual Usage of Machine Translation by Japanese University Students and Verification of Test Results","year":2023,"lang":"en","type":"article","venue":"English Language Teaching","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Active listening; Psychology; Task (project management); Mathematics education; Machine translation; Test (biology); Reading (process); Sample (material); Statistical analysis; Process (computing); Computer science; Natural language processing; Linguistics; Statistics; Communication","score_opus":0.011597136863039242,"score_gpt":0.2719583308277018,"score_spread":0.26036119396466256,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4388044549","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99872583,0.0000465863,0.00034400955,0.000032183336,0.000006164091,0.000011278692,0.000037446498,0.00002438259,0.0007721341],"genre_scores_gemma":[0.9981578,0.00006064273,0.0007732719,0.000029226883,0.000007759019,0.000024138833,0.00009054356,0.00001331396,0.0008432966],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99545443,0.0020397315,0.00061847334,0.00052129076,0.0009269761,0.00043900058],"domain_scores_gemma":[0.978912,0.007705082,0.003279219,0.0017290508,0.0068960823,0.0014785209],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004110031,0.00038603946,0.000407516,0.0011560627,0.00081812026,0.0012111173,0.0004661117,0.0006409947,0.0018854673],"category_scores_gemma":[0.02486555,0.00021123221,0.00030843922,0.0010829149,0.0006602203,0.00073041045,0.00081214326,0.00046372364,0.00126674],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006399239,0.0010692759,0.84253114,0.00019441212,0.000047379035,0.0011373191,0.032224506,0.000385776,0.019281635,0.00015573451,0.0009951836,0.10133789],"study_design_scores_gemma":[0.000027130352,0.0022543757,0.9318545,0.00006534273,0.00009137138,0.0019568077,0.031431608,0.003537598,0.022341805,0.00012963872,0.006189515,0.00012041072],"about_ca_topic_score_codex":0.0024753155,"about_ca_topic_score_gemma":0.0034656774,"teacher_disagreement_score":0.004110031,"about_ca_system_score_codex":0.00044614606,"about_ca_system_score_gemma":0.0005350477,"threshold_uncertainty_score":0.021736145},"labels":[],"label_agreement":null},{"id":"W4388048690","doi":"10.1016/j.cancergen.2023.08.027","title":"19. Developing a generalized model for variants in CIViC","year":2023,"lang":"en","type":"article","venue":"Cancer Genetics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"BC Cancer Agency","funders":"","keywords":"Computational biology; Biology; Gene; Genetics","score_opus":0.0632904558404019,"score_gpt":0.35233520057734863,"score_spread":0.28904474473694675,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4388048690","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022545718,0.0001276773,0.9666918,0.001345544,0.00010404413,0.00009601498,0.0015469602,0.0009900319,0.006552263],"genre_scores_gemma":[0.5271132,0.00029005765,0.4256965,0.0009673344,0.00016385275,0.0005371812,0.004143123,0.0005967909,0.040491823],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99937904,0.00020095574,0.000029358855,0.00020885174,0.000070876435,0.0001109326],"domain_scores_gemma":[0.9988625,0.0005700828,0.00006423807,0.00020125031,0.00022031722,0.000081701524],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017971711,0.00054505037,0.00072676974,0.0008944422,0.0008070905,0.0015302196,0.0020181993,0.0015986741,0.009985248],"category_scores_gemma":[0.004386887,0.0005650769,0.001535817,0.00077962864,0.00073450845,0.0019519058,0.0011960925,0.0020577798,0.0026991218],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022924395,0.00013734706,0.011572663,0.00009658505,0.00016567709,0.0005502256,0.00036168846,0.5140285,0.0022469426,0.35017568,0.018121447,0.10231409],"study_design_scores_gemma":[0.000024567073,0.000018358733,0.0007676534,0.00001508103,0.000035642388,0.00006981682,0.000033175056,0.8951461,0.00048440532,0.09935316,0.004039158,0.0000130050375],"about_ca_topic_score_codex":0.02735945,"about_ca_topic_score_gemma":0.036378443,"teacher_disagreement_score":0.02735945,"about_ca_system_score_codex":0.0015346725,"about_ca_system_score_gemma":0.0020258548,"threshold_uncertainty_score":0.054400384},"labels":[],"label_agreement":null},{"id":"W4388132093","doi":"10.1007/978-3-031-47240-4_26","title":"Assessing the Generalization Capabilities of Neural Machine Translation Models for SPARQL Query Generation","year":2023,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Artificial intelligence; SPARQL; Machine translation; Machine learning; Generalization; Training set; Natural language processing; Information retrieval; RDF; Semantic Web","score_opus":0.0573266660516145,"score_gpt":0.30834279112851026,"score_spread":0.25101612507689575,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4388132093","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.93836564,0.0028951983,0.048029356,0.0013314844,0.00018875086,0.00012343547,0.0013481249,0.0024130403,0.005304911],"genre_scores_gemma":[0.98122233,0.00045629696,0.014961559,0.0001383365,0.000066157874,0.000056137447,0.0020100565,0.00010926695,0.0009797089],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99763155,0.0012888358,0.00020941754,0.0004359796,0.00030942322,0.00012472925],"domain_scores_gemma":[0.9663684,0.029139316,0.0007182751,0.002044989,0.0014608329,0.00026819992],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0075466656,0.000976923,0.0006809592,0.0010739968,0.0004399182,0.0012647297,0.0011360954,0.0018825729,0.0022863355],"category_scores_gemma":[0.032099787,0.00040477116,0.00073970546,0.001317018,0.000601189,0.0029691362,0.0010687103,0.0016284356,0.0007642585],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0025371248,0.0007161058,0.015121796,0.00045995708,0.0005695599,0.00017242915,0.0003399924,0.68378204,0.0064523504,0.002967747,0.006454572,0.28042632],"study_design_scores_gemma":[0.000047329577,0.00024410986,0.0019809713,0.000017554432,0.00006436933,0.00003515485,0.00006791485,0.99253017,0.00247271,0.0022829073,0.00024120446,0.000015598665],"about_ca_topic_score_codex":0.014137924,"about_ca_topic_score_gemma":0.0106770685,"teacher_disagreement_score":0.014137924,"about_ca_system_score_codex":0.0013268668,"about_ca_system_score_gemma":0.0008729031,"threshold_uncertainty_score":0.03991103},"labels":[],"label_agreement":null},{"id":"W4388351791","doi":"10.1093/oso/9780199266623.003.0021","title":"Native Languages of Alaska","year":2007,"lang":"en","type":"book-chapter","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":54,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Homeland; Geography; Arctic; The arctic; Archaeology; Ethnology; Genealogy; History; Oceanography; Geology; Political science","score_opus":0.020282521242903015,"score_gpt":0.29802187406438263,"score_spread":0.2777393528214796,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4388351791","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010370505,0.017009249,0.0008051494,0.0005346379,0.00036605223,0.000010387291,0.0007039683,0.00017289893,0.9700271],"genre_scores_gemma":[0.20776837,0.041234773,0.0033277501,0.0003795781,0.0002646254,0.00006520462,0.0017525847,0.00021857952,0.7449885],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9998536,0.00002417687,0.000013501523,0.000032779943,0.00005319264,0.000022830704],"domain_scores_gemma":[0.9998242,0.000041093783,0.000013255467,0.000016308857,0.000074379095,0.000030783867],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00019438934,0.00033764786,0.00015214701,0.0014298653,0.0013739131,0.0021975122,0.00030259552,0.0002692498,0.033271633],"category_scores_gemma":[0.0004717325,0.00013127479,0.00013759008,0.0015644461,0.0004939625,0.0012567461,0.00088173087,0.0004197701,0.007335958],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005584807,0.00003689143,0.004378249,0.000906882,0.0000118676635,0.0010798245,0.024968134,0.0005514383,0.00235889,0.2598729,0.18812867,0.5176504],"study_design_scores_gemma":[0.0000014993692,0.000008421733,0.0027526151,0.00025586016,0.0000051427182,0.0006428699,0.0022903841,0.00007010721,0.00024085009,0.006853755,0.9868702,0.000008212439],"about_ca_topic_score_codex":0.02412225,"about_ca_topic_score_gemma":0.04603433,"teacher_disagreement_score":0.033271633,"about_ca_system_score_codex":0.0008490122,"about_ca_system_score_gemma":0.0013455196,"threshold_uncertainty_score":0.1113047},"labels":[],"label_agreement":null},{"id":"W4388391873","doi":"10.1016/j.jslw.2023.101069","title":"Another contradiction in AI-assisted second language writing","year":2023,"lang":"en","type":"article","venue":"Journal of Second Language Writing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":32,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Contradiction; Computer science; Linguistics; Natural language processing; Philosophy","score_opus":0.012511001411722015,"score_gpt":0.2835201668505848,"score_spread":0.2710091654388628,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4388391873","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.058559887,0.0046054227,0.49247012,0.13819577,0.002251265,0.00020297349,0.0004059694,0.0031266573,0.3001819],"genre_scores_gemma":[0.7493421,0.0018382653,0.18288453,0.012482907,0.00083127245,0.00034086214,0.00035202384,0.0011229025,0.050805047],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9770065,0.012797861,0.0010959,0.0020394928,0.0065325103,0.00052780734],"domain_scores_gemma":[0.893489,0.078282125,0.0019858195,0.010616567,0.014591629,0.001034911],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012164784,0.00046881495,0.00075452134,0.0019299303,0.0028353136,0.008870123,0.002515167,0.0042840955,0.013271596],"category_scores_gemma":[0.0696769,0.000692156,0.0005131711,0.0013875036,0.0067375633,0.012272054,0.0035816347,0.0056074886,0.0059320955],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00038939025,0.0002936714,0.0021605927,0.0008027763,0.00006309268,0.00031620497,0.015579221,0.0012813418,0.007613371,0.71433413,0.037835903,0.21933033],"study_design_scores_gemma":[0.00019916843,0.00030344023,0.0047129365,0.0008660926,0.0000725863,0.0027229055,0.013868525,0.035525184,0.029380696,0.62822115,0.28388858,0.00023882395],"about_ca_topic_score_codex":0.002025203,"about_ca_topic_score_gemma":0.0013150665,"teacher_disagreement_score":0.013271596,"about_ca_system_score_codex":0.001546739,"about_ca_system_score_gemma":0.0027823986,"threshold_uncertainty_score":0.06433433},"labels":[],"label_agreement":null},{"id":"W4388409533","doi":"10.1145/3615887.3627762","title":"From Uncertainty to Action","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"The Scarborough Hospital; University of Toronto","funders":"","keywords":"Focus (optics); Computer science; Digital library; Colonialism; Action (physics); Toponymy; Data science; History; Archaeology; Literature; Art","score_opus":0.031160288519250096,"score_gpt":0.32890036622835483,"score_spread":0.2977400777091047,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4388409533","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023039693,0.014415672,0.64907634,0.08307593,0.0013996047,0.00014284682,0.0013394236,0.00042379813,0.22708665],"genre_scores_gemma":[0.8712659,0.008607268,0.0992932,0.0040854765,0.0009861375,0.00029224032,0.00060196046,0.00018158268,0.014686088],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9949463,0.002814764,0.00026508272,0.0009432968,0.000774877,0.00025573187],"domain_scores_gemma":[0.98814964,0.008849803,0.0007629476,0.0011512291,0.0007331621,0.00035311378],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005841794,0.0008685755,0.00070117,0.002801982,0.0029758292,0.009743168,0.0015366721,0.0023678727,0.010813658],"category_scores_gemma":[0.020205265,0.00053608476,0.0009206873,0.0019174251,0.022025095,0.01636957,0.006433812,0.003309573,0.00090298033],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000025759166,0.000006997377,0.0005065083,0.0001367984,0.00002686675,0.00016766677,0.002402529,0.0030921747,0.00011057214,0.97051615,0.0031219108,0.019886155],"study_design_scores_gemma":[0.00000360932,0.0000064651,0.0001595417,0.00011668758,0.000007244761,0.00005359861,0.0011313865,0.0021136266,0.00011332638,0.9716626,0.02461818,0.00001367095],"about_ca_topic_score_codex":0.007609912,"about_ca_topic_score_gemma":0.005525142,"teacher_disagreement_score":0.010813658,"about_ca_system_score_codex":0.0047840546,"about_ca_system_score_gemma":0.0029193105,"threshold_uncertainty_score":0.03617525},"labels":[],"label_agreement":null},{"id":"W4388496578","doi":"10.31234/osf.io/b6978","title":"The roles of neural networks in language acquisition","year":2023,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; Mila - Quebec Artificial Intelligence Institute","funders":"","keywords":"LEAPS; Computer science; Cognitive science; Language acquisition; Philosophy of language; Natural language; Cognition; Field (mathematics); Artificial intelligence; Psychology; Epistemology","score_opus":0.013285645550076693,"score_gpt":0.28591579843282555,"score_spread":0.27263015288274883,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4388496578","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.29611918,0.011481817,0.553677,0.030615192,0.0001909017,0.00007147558,0.00036926844,0.0005738064,0.10690135],"genre_scores_gemma":[0.94033015,0.0038330478,0.04903012,0.0003719006,0.000094188625,0.00008564219,0.00012599576,0.00008035295,0.006048613],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9987207,0.0006781895,0.000044318334,0.00025684867,0.00021262435,0.00008730347],"domain_scores_gemma":[0.993411,0.0050376607,0.00041193445,0.0004773427,0.00044656638,0.0002155824],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029763922,0.00044844404,0.00039102972,0.0009825493,0.00066214945,0.0039199498,0.0012863373,0.0014370134,0.0030966175],"category_scores_gemma":[0.015807712,0.00061745866,0.00045111557,0.0008641013,0.0044280374,0.013052296,0.0018027743,0.00267318,0.0006336706],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006650825,0.000047749247,0.0067499164,0.00019075362,0.0000538852,0.00014204945,0.0016222626,0.038594082,0.002263191,0.88635117,0.0011170758,0.06280143],"study_design_scores_gemma":[0.000006570937,0.000024023777,0.0021265242,0.000058967616,0.00001238556,0.00007774764,0.00026670896,0.09163343,0.0009473643,0.90150166,0.0033206225,0.000023959119],"about_ca_topic_score_codex":0.0034855604,"about_ca_topic_score_gemma":0.0032095264,"teacher_disagreement_score":0.0039199498,"about_ca_system_score_codex":0.0017331386,"about_ca_system_score_gemma":0.0012100056,"threshold_uncertainty_score":0.015740812},"labels":[],"label_agreement":null},{"id":"W4388496996","doi":"10.18280/ria.370503","title":"Lexical Based Reordering Models for English to Telugu Machine Translation","year":2023,"lang":"en","type":"article","venue":"Revue d intelligence artificielle","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"State of New Jersey Department of Education","keywords":"Telugu; Natural language processing; Computer science; Machine translation; Translation (biology); Artificial intelligence; Linguistics; Biology; Philosophy","score_opus":0.06599171833415778,"score_gpt":0.313901693747489,"score_spread":0.2479099754133312,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4388496996","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13722028,0.0020330823,0.8478221,0.00085217785,0.00019765538,0.00020836764,0.0014645146,0.004644171,0.005557679],"genre_scores_gemma":[0.8437655,0.0009223328,0.14174372,0.0001706084,0.00008212826,0.00030976383,0.0032609445,0.0005219671,0.009222902],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99946386,0.00026524387,0.00005709226,0.000086970074,0.00009042224,0.00003647568],"domain_scores_gemma":[0.9992112,0.00043457816,0.000082088045,0.0000654169,0.0001911763,0.000015521067],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008195985,0.00064348476,0.00048665114,0.0007922869,0.00037969573,0.0006744983,0.00074747077,0.0004558666,0.0028033054],"category_scores_gemma":[0.0019861404,0.0003076976,0.0008832247,0.0008332212,0.00023847462,0.0010942254,0.00033460572,0.0008813147,0.0021528127],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00049801916,0.00024119316,0.005063283,0.00035891574,0.00017791457,0.00062354526,0.00046932438,0.6329358,0.024870692,0.013183706,0.006124262,0.31545332],"study_design_scores_gemma":[0.00000851393,0.00006694771,0.0007757732,0.000011288691,0.000019115447,0.000059620612,0.00004450591,0.9914625,0.0024519009,0.0036742406,0.001413321,0.000012233675],"about_ca_topic_score_codex":0.0088898055,"about_ca_topic_score_gemma":0.014895779,"teacher_disagreement_score":0.0088898055,"about_ca_system_score_codex":0.0006174101,"about_ca_system_score_gemma":0.00092654023,"threshold_uncertainty_score":0.017676115},"labels":[],"label_agreement":null},{"id":"W4388676769","doi":"10.26615/978-954-452-092-2_039","title":"Mapping Explicit and Implicit Discourse Relations between the RST-DT and the PDTB 3.0","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Natural language processing; Artificial intelligence; Linguistics; Philosophy","score_opus":0.01957050585783704,"score_gpt":0.29213022566202473,"score_spread":0.2725597198041877,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4388676769","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.29708198,0.004956138,0.5708402,0.0017198459,0.00083384564,0.0012570141,0.052687887,0.01032841,0.06029476],"genre_scores_gemma":[0.47512794,0.0011474111,0.43625686,0.0002847042,0.00014433655,0.0013436021,0.07566669,0.0027324138,0.0072960774],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99442977,0.0023083913,0.00053053343,0.0012089985,0.0013250629,0.00019717384],"domain_scores_gemma":[0.9785984,0.012037411,0.0016406772,0.0029202579,0.0043587387,0.00044454107],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0044434136,0.00074378366,0.00048309605,0.005725907,0.0014022881,0.002858783,0.0009505235,0.0008460847,0.009191557],"category_scores_gemma":[0.0239619,0.00059243606,0.00063429406,0.0048293574,0.0011027597,0.0025865273,0.0034485792,0.0018328286,0.0046711843],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017738218,0.00045669216,0.031149637,0.0048745475,0.00030935157,0.0011841446,0.011180911,0.005706464,0.19317465,0.049392857,0.049105417,0.6516916],"study_design_scores_gemma":[0.00016277243,0.00047654688,0.12570152,0.001064477,0.00027052898,0.0027621863,0.008517103,0.069135234,0.15672976,0.04202202,0.5927921,0.00036573154],"about_ca_topic_score_codex":0.005030316,"about_ca_topic_score_gemma":0.0057287286,"teacher_disagreement_score":0.009191557,"about_ca_system_score_codex":0.001374074,"about_ca_system_score_gemma":0.0020935335,"threshold_uncertainty_score":0.030748785},"labels":[],"label_agreement":null},{"id":"W4388731527","doi":"10.1017/s1351324923000529","title":"Korean named entity recognition based on language-specific features – CORRIGENDUM","year":2023,"lang":"en","type":"erratum","venue":"Natural Language Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"National Research Foundation of Korea; National Research Foundation","keywords":"Computer science; Content (measure theory); Natural language processing; Information retrieval; Artificial intelligence; Database; World Wide Web","score_opus":0.012516482264891337,"score_gpt":0.24193848754718925,"score_spread":0.22942200528229792,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4388731527","genre_codex":"editorial","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00818963,0.00768495,0.066276655,0.041893438,0.77666897,0.0003066548,0.015267618,0.0085116085,0.075200476],"genre_scores_gemma":[0.049323853,0.0108989095,0.07350224,0.017696485,0.028717458,0.00024474287,0.040096596,0.005645645,0.77387404],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99922514,0.00010362883,0.00014809234,0.00014157427,0.00033309375,0.00004859775],"domain_scores_gemma":[0.9948537,0.0005056783,0.00014649011,0.00063057843,0.0037378373,0.0001258131],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008945065,0.0010885131,0.0008209739,0.0015389748,0.00094739604,0.002120643,0.0012801582,0.0010731411,0.07033081],"category_scores_gemma":[0.007221515,0.00042499238,0.0006416319,0.0017452494,0.000541109,0.0018039905,0.00092916755,0.0014088188,0.071740635],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000048113256,0.000020030937,0.00016565746,0.00014720493,0.000016505339,0.0006597371,0.000025367883,0.00018243308,0.0008989096,0.0012556391,0.9561004,0.040479936],"study_design_scores_gemma":[0.000018909435,0.000037583457,0.0021162229,0.00014194295,0.000055041583,0.0011082622,0.00012619454,0.0023601633,0.005663846,0.0025396468,0.9857772,0.000055014927],"about_ca_topic_score_codex":0.012396283,"about_ca_topic_score_gemma":0.019838339,"teacher_disagreement_score":0.07033081,"about_ca_system_score_codex":0.0010301945,"about_ca_system_score_gemma":0.0010784047,"threshold_uncertainty_score":0.23528004},"labels":[],"label_agreement":null},{"id":"W4388787825","doi":"10.48550/arxiv.2311.09696","title":"Fumbling in Babel: An Investigation into ChatGPT's Language Identification Ability","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Computer science; Variety (cybernetics); Benchmark (surveying); Set (abstract data type); Identification (biology); Resource (disambiguation); Natural language processing; Language identification; Gamut; Artificial intelligence; Range (aeronautics); Compiler; Linguistics; Programming language; Natural language","score_opus":0.07339839838717734,"score_gpt":0.24188215235871405,"score_spread":0.1684837539715367,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4388787825","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7951217,0.0026596687,0.094039865,0.0011609608,0.00075313885,0.00023930628,0.007733861,0.0782686,0.020022893],"genre_scores_gemma":[0.87733513,0.00048454112,0.08209149,0.0007449098,0.000120630684,0.00023716048,0.02560729,0.0062589985,0.00711992],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99471116,0.0022940838,0.0002791178,0.0013452132,0.0008833148,0.00048717577],"domain_scores_gemma":[0.9706461,0.02264142,0.0005411438,0.0032196068,0.0022275667,0.0007241332],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0067292876,0.0014237766,0.0011768927,0.0021981858,0.0013013013,0.002727551,0.0025142448,0.0015075462,0.005128218],"category_scores_gemma":[0.028710349,0.00080508075,0.0010023348,0.0020317682,0.0019197848,0.005345218,0.004526863,0.0027839853,0.0028551517],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0058180653,0.0011237696,0.096271515,0.004464683,0.0011452554,0.0036002556,0.01085335,0.18505861,0.035726357,0.02772568,0.12679413,0.5014183],"study_design_scores_gemma":[0.00025420386,0.0012487022,0.017246295,0.00034205336,0.00026439704,0.0013456615,0.003611583,0.8586271,0.047047675,0.019604448,0.050154626,0.00025328988],"about_ca_topic_score_codex":0.009775208,"about_ca_topic_score_gemma":0.010984026,"teacher_disagreement_score":0.009775208,"about_ca_system_score_codex":0.0011650204,"about_ca_system_score_gemma":0.0012851518,"threshold_uncertainty_score":0.035588324},"labels":[],"label_agreement":null},{"id":"W4388830286","doi":"10.5430/wjel.v14n1p187","title":"Do Self-translating Poets Have Equally Distributed Equivalent Words in the Target and Original Texts? A Corpus Examination of Yu Guangzhong’s Poems","year":2023,"lang":"en","type":"article","venue":"World Journal of English Language","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Equivalence (formal languages); Poetry; Computer science; Linguistics; Literal translation; Literal (mathematical logic); Reflexive pronoun; Word (group theory); Modal; Natural language processing; Artificial intelligence; Source text; Philosophy","score_opus":0.016298714194596824,"score_gpt":0.28687683163048044,"score_spread":0.2705781174358836,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4388830286","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9940896,0.00019556485,0.0009735785,0.00008795074,0.000013592327,0.00003614324,0.00021361363,0.000011722203,0.004378311],"genre_scores_gemma":[0.99722207,0.00017478003,0.001363532,0.000028985018,0.0000077072955,0.000051710034,0.00050736836,0.000030172034,0.00061368773],"study_design_codex":"qualitative","study_design_gemma":"not_applicable","domain_scores_codex":[0.99859315,0.00065086636,0.00014842908,0.00023652377,0.00028403732,0.000087015214],"domain_scores_gemma":[0.9907204,0.005921237,0.0009569292,0.0011263693,0.0011026069,0.00017246002],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022623397,0.00016906059,0.00034554413,0.0031478878,0.0010872817,0.0012274302,0.00029143813,0.00020962312,0.0018082295],"category_scores_gemma":[0.013123988,0.00019050649,0.00012379658,0.0041996106,0.0021874697,0.001867134,0.0017251266,0.0004277095,0.00018280621],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000849655,0.00019825119,0.20782574,0.0017093457,0.00015700184,0.0028581964,0.5390191,0.0006676101,0.024905365,0.02859818,0.002839155,0.19037245],"study_design_scores_gemma":[0.00006367318,0.00018113304,0.7387389,0.0002826123,0.00009955303,0.0016345236,0.19450775,0.002587144,0.0070709772,0.005618793,0.049156133,0.000058745703],"about_ca_topic_score_codex":0.0031011484,"about_ca_topic_score_gemma":0.005524062,"teacher_disagreement_score":0.0031478878,"about_ca_system_score_codex":0.0007850664,"about_ca_system_score_gemma":0.0007525796,"threshold_uncertainty_score":0.01196456},"labels":[],"label_agreement":null},{"id":"W4388878632","doi":"10.1007/978-3-031-48312-7_6","title":"Improvements in Language Modeling, Voice Activity Detection, and Lexicon in OpenASR21 Low Resource Languages","year":2023,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Computer Research Institute of Montréal","funders":"","keywords":"Computer science; Lexicon; Word error rate; Artificial intelligence; Natural language processing; Vocabulary; Speech recognition; Phone; Resource (disambiguation); Linguistics","score_opus":0.013351292275710512,"score_gpt":0.27445992424551624,"score_spread":0.2611086319698057,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4388878632","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029463766,0.0004847472,0.8693701,0.0005991597,0.00034256728,0.00015618066,0.0032419967,0.08295804,0.0133834565],"genre_scores_gemma":[0.24580348,0.00038273845,0.7045876,0.00058738806,0.00029278724,0.00029898837,0.015374265,0.013448707,0.019224105],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9965842,0.0008848679,0.000426389,0.00085349515,0.0010112652,0.00023975233],"domain_scores_gemma":[0.9950275,0.0017453291,0.00019917662,0.0016082053,0.0013186703,0.000101152786],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021791977,0.0014389922,0.0013045968,0.0014068118,0.0007869978,0.0034666706,0.003309246,0.00082728383,0.01829352],"category_scores_gemma":[0.0065988363,0.0010735804,0.0020630695,0.0011627593,0.00074191234,0.0070891315,0.002547617,0.002290032,0.015516695],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008010167,0.00038021174,0.0031502692,0.00061197503,0.00015541751,0.0003700116,0.0006863204,0.029257085,0.08236356,0.053000305,0.030904662,0.7983191],"study_design_scores_gemma":[0.0001537323,0.00036787044,0.0019370017,0.0001158369,0.0002918988,0.0007968863,0.00035019487,0.6795689,0.1279314,0.05611392,0.13211928,0.00025304945],"about_ca_topic_score_codex":0.0066830404,"about_ca_topic_score_gemma":0.011746079,"teacher_disagreement_score":0.01829352,"about_ca_system_score_codex":0.0010494316,"about_ca_system_score_gemma":0.001784936,"threshold_uncertainty_score":0.061197937},"labels":[],"label_agreement":null},{"id":"W4389009377","doi":"10.18653/v1/2023.inlg-main.1","title":"Guided Beam Search to Improve Generalization in Low-Resource Data-to-Text Generation","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"Fonds de recherche du Québec – Nature et technologies; Natural Sciences and Engineering Research Council of Canada","keywords":"Interpretability; Computer science; Generalization; Decoding methods; Set (abstract data type); Machine learning; Artificial intelligence; Process (computing); Resource (disambiguation); Beam search; Data mining; Algorithm; Search algorithm; Mathematics","score_opus":0.06224814578580187,"score_gpt":0.34486372976616586,"score_spread":0.282615583980364,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389009377","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03444529,0.0004889357,0.95845675,0.00046951417,0.00007046433,0.00016737329,0.00020360501,0.004192601,0.0015054593],"genre_scores_gemma":[0.40637597,0.00020828402,0.5844525,0.0014117928,0.00016370954,0.00076055416,0.0019366915,0.0012929906,0.0033975085],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9966859,0.0018266684,0.00019645334,0.00055901567,0.00054180797,0.00019025354],"domain_scores_gemma":[0.97971946,0.015875842,0.00049520825,0.0020069068,0.0016245292,0.0002780427],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007831682,0.0018194745,0.0018893867,0.0019753284,0.0009577528,0.0014908094,0.0031366663,0.0026210707,0.005231722],"category_scores_gemma":[0.032503594,0.0009659423,0.0011284578,0.0014193044,0.0014196613,0.0027630844,0.0025973718,0.0032991813,0.0020714772],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00063701725,0.00041515715,0.0055434797,0.00021264263,0.00026996978,0.00032790322,0.0006884666,0.5488022,0.0076791476,0.012370867,0.012286307,0.4107669],"study_design_scores_gemma":[0.00006650577,0.0000579979,0.00021098836,0.000014965175,0.000016688553,0.000044524844,0.000032434553,0.98645544,0.0019553984,0.010412955,0.0007225029,0.000009616462],"about_ca_topic_score_codex":0.0043212194,"about_ca_topic_score_gemma":0.0067908196,"teacher_disagreement_score":0.007831682,"about_ca_system_score_codex":0.0010784145,"about_ca_system_score_gemma":0.0015884776,"threshold_uncertainty_score":0.041418374},"labels":[],"label_agreement":null},{"id":"W4389009557","doi":"10.18653/v1/2023.inlg-main.36","title":"Mod-D2T: A Multi-layer Dataset for Modular Data-to-Text Generation","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"European Commission","keywords":"Computer science; Modular design; Fluency; Pipeline (software); Text generation; Leverage (statistics); Artificial intelligence; Documentation; Natural language processing; Code (set theory); Data mining; Information retrieval; Programming language","score_opus":0.19763519033315677,"score_gpt":0.39224375399369577,"score_spread":0.194608563660539,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389009557","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.041924573,0.0010426929,0.048263475,0.0011734547,0.00063090894,0.0011998459,0.84471065,0.0470856,0.013968814],"genre_scores_gemma":[0.028266393,0.00014492907,0.058048986,0.000313643,0.00003746267,0.0011774126,0.90777045,0.0010404397,0.003200211],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.998818,0.00028490406,0.00014277444,0.00039542362,0.00028030496,0.00007865182],"domain_scores_gemma":[0.9968989,0.0012574986,0.00020802153,0.0010675458,0.00038060747,0.00018745048],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001499832,0.002242264,0.00057711784,0.0020856704,0.0010848474,0.001662128,0.003883174,0.0024755665,0.014192299],"category_scores_gemma":[0.0080512,0.0006159939,0.0019465606,0.0020491583,0.00092708855,0.0021388116,0.0025317855,0.0024403005,0.01327585],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007363496,0.00066082255,0.0074277194,0.002894007,0.0003138073,0.0007371365,0.00040684812,0.01739393,0.0107467845,0.007151093,0.8397008,0.11183076],"study_design_scores_gemma":[0.0015169592,0.00071644835,0.016553756,0.0004982924,0.00022531193,0.0020386858,0.0008440512,0.1683988,0.046891928,0.03274644,0.7292569,0.00031244673],"about_ca_topic_score_codex":0.007698634,"about_ca_topic_score_gemma":0.020004937,"teacher_disagreement_score":0.014192299,"about_ca_system_score_codex":0.0012068569,"about_ca_system_score_gemma":0.0014971087,"threshold_uncertainty_score":0.04747802},"labels":[],"label_agreement":null},{"id":"W4389010521","doi":"10.18653/v1/2023.sigdial-1.14","title":"The Road to Quality is Paved with Good Revisions: A Detailed Evaluation Methodology for Revision Policies in Incremental Sequence Labelling","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Atomic Energy of Canada Limited","keywords":"Computer science; Prefix; Sequence (biology); Encoder; Labelling; Transformer; Quality (philosophy); Engineering","score_opus":0.23837580453810525,"score_gpt":0.47627028374973013,"score_spread":0.23789447921162488,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389010521","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06844691,0.00055525865,0.9224662,0.00053153554,0.000082286395,0.0005517618,0.0004312621,0.004484517,0.0024502957],"genre_scores_gemma":[0.5422049,0.00021665658,0.4547195,0.00016591387,0.00004233978,0.00033030324,0.0006544428,0.0007233115,0.00094259274],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.95915866,0.020800628,0.0043158573,0.0038746228,0.010582986,0.0012671263],"domain_scores_gemma":[0.82083607,0.10573093,0.014940579,0.028891925,0.025576048,0.004024393],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03360151,0.0014035659,0.0012733648,0.0033042184,0.0011041026,0.0045571434,0.002782372,0.0024638074,0.0017790698],"category_scores_gemma":[0.16626574,0.0008978825,0.00095819274,0.0021060018,0.0029396694,0.007310363,0.0033444117,0.0029963525,0.00059471256],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001986356,0.00084801746,0.032573532,0.001352515,0.0004612929,0.00025081483,0.0035683673,0.2718223,0.029257733,0.057712037,0.0050076116,0.5951595],"study_design_scores_gemma":[0.000105198014,0.0014298521,0.005613454,0.00017613808,0.0001410028,0.0002859535,0.00039710992,0.9169579,0.032652386,0.03719946,0.0048751445,0.0001663696],"about_ca_topic_score_codex":0.0074856733,"about_ca_topic_score_gemma":0.008112233,"teacher_disagreement_score":0.03360151,"about_ca_system_score_codex":0.0037459705,"about_ca_system_score_gemma":0.0042538154,"threshold_uncertainty_score":0.17770386},"labels":[],"label_agreement":null},{"id":"W4389133516","doi":"","title":"4th Workshop on Computational Approaches to Discourse: Proceedings of the Workshop held in conjunction with ACL 2023 - 9-14 July, 2023, Toronto (Canada)","year":2023,"lang":"en","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science","score_opus":0.03467081037405783,"score_gpt":0.24958586249437192,"score_spread":0.2149150521203141,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389133516","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021137705,0.04511592,0.547945,0.073968865,0.02925807,0.0016967898,0.027020546,0.013538675,0.24031836],"genre_scores_gemma":[0.112548426,0.0170013,0.37380102,0.0049840324,0.004815823,0.0021976063,0.065901145,0.006105644,0.41264507],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99595636,0.002183538,0.00023704734,0.000777444,0.00050738885,0.0003381115],"domain_scores_gemma":[0.99097073,0.0041288016,0.00015053093,0.0012882367,0.0022223631,0.0012393849],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0072880667,0.0019493474,0.0026476723,0.0028859288,0.004421301,0.015653823,0.0040746303,0.0030507762,0.09126369],"category_scores_gemma":[0.0097677205,0.001159624,0.0029182478,0.0031481618,0.0039109783,0.010775689,0.0067163436,0.004528477,0.021568112],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006316526,0.00030188938,0.0010387052,0.0015492168,0.00014528798,0.0003428968,0.0058723153,0.0016145357,0.004183516,0.092801414,0.6684405,0.22307803],"study_design_scores_gemma":[0.00007581022,0.00005735316,0.0014604317,0.0007169501,0.00008667174,0.00019749673,0.0029657145,0.0063533895,0.0026119666,0.04828131,0.93712825,0.000064739055],"about_ca_topic_score_codex":0.0575423,"about_ca_topic_score_gemma":0.09117858,"teacher_disagreement_score":0.9424577,"about_ca_system_score_codex":0.008034719,"about_ca_system_score_gemma":0.009908943,"threshold_uncertainty_score":0.30530745},"labels":[],"label_agreement":null},{"id":"W4389430421","doi":"10.7202/1107567ar","title":"L’outil Ultrad de La Presse Canadienne : la traduction automatique dans un contexte journalistique","year":2023,"lang":"fr","type":"article","venue":"TTR traduction terminologie rédaction","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":true,"ca_institutions":"Université du Québec à Montréal; Université du Québec à Trois-Rivières","funders":"","keywords":"Humanities; Philosophy; Art","score_opus":0.035837575371077124,"score_gpt":0.3098170628529139,"score_spread":0.27397948748183676,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389430421","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.32830402,0.068178676,0.23295185,0.04070822,0.00927138,0.002712684,0.034023967,0.044717833,0.2391314],"genre_scores_gemma":[0.5013516,0.016715677,0.29760754,0.0042626904,0.0018191035,0.0011350675,0.02512765,0.00786213,0.14411856],"study_design_codex":"design_other","study_design_gemma":"qualitative","domain_scores_codex":[0.9862169,0.0043523656,0.0013842115,0.0024029363,0.0051201787,0.00052347063],"domain_scores_gemma":[0.9512709,0.020702738,0.0027114446,0.0059771338,0.017494295,0.0018435579],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010678257,0.0009941085,0.000981608,0.00832769,0.003720487,0.012306592,0.0012794129,0.0015573826,0.013363684],"category_scores_gemma":[0.03633339,0.00095461064,0.0008701912,0.008873108,0.0033224549,0.007122623,0.0032298723,0.0022359681,0.007191033],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008814789,0.00016513132,0.026264738,0.0057430663,0.00024850454,0.0014910202,0.04028516,0.0014179861,0.037394144,0.024446853,0.10403891,0.757623],"study_design_scores_gemma":[0.000067552486,0.00014266398,0.02292971,0.0009386685,0.0001360598,0.000597927,0.006665971,0.0025260823,0.01785238,0.0027673824,0.9451699,0.00020566156],"about_ca_topic_score_codex":0.08585905,"about_ca_topic_score_gemma":0.07980839,"teacher_disagreement_score":0.91414094,"about_ca_system_score_codex":0.005852229,"about_ca_system_score_gemma":0.013109705,"threshold_uncertainty_score":0.17071861},"labels":[],"label_agreement":null},{"id":"W4389459778","doi":"10.29173/irie499","title":"Multilingualism in cyberspace: a practical reality?","year":2022,"lang":"en","type":"article","venue":"The International Review of Information Ethics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Cyberspace; Multilingualism; Linguistics; Computer science; Sociology; World Wide Web; The Internet; Philosophy","score_opus":0.05456000649711222,"score_gpt":0.41645242930188314,"score_spread":0.3618924228047709,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389459778","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023207592,0.075170174,0.05369316,0.7203701,0.0031522298,0.000037392838,0.00008132751,0.00015253111,0.12413543],"genre_scores_gemma":[0.83827835,0.044324864,0.016565884,0.07728796,0.006496306,0.00018766237,0.00009238858,0.00023957377,0.016526988],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.98140746,0.012954243,0.0006990386,0.0011222982,0.0028894902,0.00092742266],"domain_scores_gemma":[0.96707946,0.022200445,0.0028370973,0.0029645152,0.0030319176,0.0018866054],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01722693,0.000391829,0.00063014403,0.0016817492,0.0046009426,0.01235989,0.0012936298,0.0042941044,0.003802411],"category_scores_gemma":[0.023835244,0.00028105217,0.00037549032,0.0016879297,0.033781815,0.024163527,0.009514038,0.008481405,0.0010226459],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000015002401,0.000014312429,0.0007125603,0.0003046249,0.000015700369,0.0002288181,0.0060367924,0.00021764166,0.00019365686,0.9391556,0.0134580415,0.039647207],"study_design_scores_gemma":[0.000007139056,0.000024569628,0.00057569507,0.0007632465,0.000009135835,0.0005750843,0.0098556345,0.00040061565,0.00033248175,0.8151486,0.17228624,0.000021536795],"about_ca_topic_score_codex":0.0017471621,"about_ca_topic_score_gemma":0.0019018385,"teacher_disagreement_score":0.01722693,"about_ca_system_score_codex":0.0032113304,"about_ca_system_score_gemma":0.0048133787,"threshold_uncertainty_score":0.09110582},"labels":[],"label_agreement":null},{"id":"W4389473428","doi":"10.16995/dscn.9669","title":"Alignement sémantique et manque de données : l’apport des modèles de langue. Le cas du latin et du grec","year":2023,"lang":"fr","type":"article","venue":"Digital Studies / Le champ numérique","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Philosophy","score_opus":0.05121940641463512,"score_gpt":0.30317674317464594,"score_spread":0.2519573367600108,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389473428","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009575886,0.0015801591,0.97929513,0.0022741503,0.00015882797,0.00014700525,0.0014133318,0.002253273,0.0033021418],"genre_scores_gemma":[0.195559,0.0017473375,0.79189,0.00046290163,0.00015333627,0.0004178416,0.0035454405,0.00083889236,0.005385367],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9915991,0.0038197092,0.00080812746,0.0020567628,0.0015544995,0.00016177322],"domain_scores_gemma":[0.9879858,0.007513611,0.00077165477,0.0019829294,0.001566765,0.00017921477],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010595049,0.00096574344,0.00093857554,0.004647959,0.0010975647,0.009619041,0.0017380029,0.0016240019,0.004276781],"category_scores_gemma":[0.03401042,0.0010848349,0.0024857197,0.005393984,0.0027589577,0.011864134,0.0028281626,0.0027800472,0.0018314177],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00049286726,0.00014131318,0.011708729,0.0011409346,0.00055417436,0.0004236633,0.0071501983,0.04768804,0.007186423,0.38919088,0.012457727,0.52186507],"study_design_scores_gemma":[0.000046522513,0.00014453218,0.0056318534,0.0005467238,0.00027253336,0.0006704054,0.0023838435,0.5424685,0.005625994,0.29595825,0.14610083,0.00014994954],"about_ca_topic_score_codex":0.02400355,"about_ca_topic_score_gemma":0.01918412,"teacher_disagreement_score":0.02400355,"about_ca_system_score_codex":0.002905598,"about_ca_system_score_gemma":0.0035792834,"threshold_uncertainty_score":0.056032658},"labels":[],"label_agreement":null},{"id":"W4389483684","doi":"10.1016/j.procs.2023.10.021","title":"FreCDo: A Large Corpus for French Cross-Domain Dialect Identification","year":2023,"lang":"en","type":"article","venue":"Procedia Computer Science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Artificial intelligence; Support vector machine; Discriminative model; Classifier (UML); Identification (biology); Natural language processing; Domain (mathematical analysis); Word (group theory); Task (project management); Linguistics","score_opus":0.015374244764587817,"score_gpt":0.3110712238581493,"score_spread":0.29569697909356146,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389483684","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.343623,0.011527774,0.04630568,0.0017493153,0.0015878269,0.0012695307,0.53949505,0.013385383,0.041056424],"genre_scores_gemma":[0.21400388,0.0013678193,0.04697345,0.00064457895,0.0003647603,0.001422697,0.7226179,0.0016439067,0.010960936],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99805176,0.0006411548,0.00018685896,0.00053780724,0.00039829005,0.00018400322],"domain_scores_gemma":[0.99564254,0.0019356109,0.00021704477,0.0006772989,0.0012926946,0.00023485869],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016182924,0.0012850381,0.00072637707,0.007300265,0.0018286627,0.0016960099,0.0010643582,0.0015405839,0.01207978],"category_scores_gemma":[0.0069652307,0.00029383946,0.0006507486,0.0040097237,0.00071191083,0.0015148744,0.001787832,0.0009831191,0.007820887],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009119414,0.00057031144,0.024762057,0.004294705,0.0002609617,0.00383251,0.005764667,0.0031975757,0.0313264,0.007928882,0.57133275,0.34581724],"study_design_scores_gemma":[0.00024730622,0.00022680097,0.09143102,0.0005912043,0.00012789885,0.0049725724,0.0033597401,0.008393181,0.012552559,0.0025231296,0.8753427,0.00023190874],"about_ca_topic_score_codex":0.027190791,"about_ca_topic_score_gemma":0.031381864,"teacher_disagreement_score":0.027190791,"about_ca_system_score_codex":0.0011784654,"about_ca_system_score_gemma":0.001432832,"threshold_uncertainty_score":0.05406505},"labels":[],"label_agreement":null},{"id":"W4389518279","doi":"10.18653/v1/2023.arabicnlp-1.20","title":"Octopus: A Multitask Model and Toolkit for Arabic Natural Language Generation","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Alliance de recherche numérique du Canada; Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Computer science; Python (programming language); Natural language processing; Arabic; Artificial intelligence; Transformer; Language model; Task (project management); Machine translation; Machine learning; Programming language; Linguistics","score_opus":0.024932707679304545,"score_gpt":0.30170132300915786,"score_spread":0.2767686153298533,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389518279","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015830075,0.00046733973,0.683236,0.0007659965,0.0005825094,0.0006446672,0.016457245,0.27274823,0.0092678955],"genre_scores_gemma":[0.23532297,0.0005616949,0.6703553,0.0010058412,0.00014943995,0.0025571708,0.047565997,0.01967764,0.022803936],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9996536,0.00010713757,0.000029418665,0.00009452016,0.00007436618,0.000041001047],"domain_scores_gemma":[0.9991436,0.00041200971,0.000043301978,0.00015760175,0.00016936651,0.00007400148],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010109643,0.0012897305,0.0004777145,0.00074500457,0.0004902504,0.0010000298,0.002429454,0.0008293375,0.019775154],"category_scores_gemma":[0.004559571,0.00063692016,0.0011224326,0.00052735646,0.0003803431,0.0017663267,0.0018871273,0.0024960414,0.011468127],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011822201,0.0003897557,0.0037192793,0.0011479241,0.00031007724,0.0005515539,0.0007488247,0.15812442,0.014449732,0.01655819,0.3641638,0.4386543],"study_design_scores_gemma":[0.00009994232,0.00008998185,0.00061425025,0.000049612907,0.0000297633,0.00017755988,0.00006959527,0.9251657,0.009121178,0.01348177,0.051042788,0.000057843943],"about_ca_topic_score_codex":0.00845233,"about_ca_topic_score_gemma":0.0146698635,"teacher_disagreement_score":0.019775154,"about_ca_system_score_codex":0.00078921404,"about_ca_system_score_gemma":0.0018086564,"threshold_uncertainty_score":0.06615448},"labels":[],"label_agreement":null},{"id":"W4389518309","doi":"10.18653/v1/2023.arabicnlp-1.6","title":"TARJAMAT: Evaluation of Bard and ChatGPT on Machine Translation of Ten Arabic Varieties","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Alliance de recherche numérique du Canada; Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Arabic; Constraint (computer-aided design); Computer science; Linguistics; Modern Standard Arabic; Natural language processing; Machine translation; Artificial intelligence; Engineering","score_opus":0.032619065177320314,"score_gpt":0.30797262700948475,"score_spread":0.2753535618321644,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389518309","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7895118,0.008234525,0.07424331,0.0018468047,0.0019897397,0.001204685,0.019068303,0.07474978,0.029151093],"genre_scores_gemma":[0.8003255,0.0013532696,0.11488709,0.0008575559,0.00016344505,0.000758705,0.06828087,0.0023304925,0.011043058],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99707913,0.0013396457,0.00030156274,0.0007847313,0.00035404292,0.0001408697],"domain_scores_gemma":[0.9943903,0.0031823607,0.00015898069,0.00082124706,0.0010604652,0.00038663842],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004124805,0.0018585083,0.00095369306,0.0012608729,0.0010223258,0.002237093,0.0019545802,0.0019114463,0.0066914437],"category_scores_gemma":[0.0131477555,0.00039976722,0.00084449997,0.0012145275,0.00075949286,0.0028889752,0.0022326398,0.0021790275,0.0061520855],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0056987535,0.0019663193,0.014738221,0.0038789734,0.0010030025,0.0009959965,0.0022848074,0.07538374,0.024778211,0.003345747,0.09411094,0.77181536],"study_design_scores_gemma":[0.0016368513,0.0042304024,0.022880955,0.00065902265,0.00058068405,0.001123076,0.00386665,0.8230035,0.05380603,0.0063909558,0.08147902,0.00034284874],"about_ca_topic_score_codex":0.0101057775,"about_ca_topic_score_gemma":0.0121887745,"teacher_disagreement_score":0.0101057775,"about_ca_system_score_codex":0.0012866105,"about_ca_system_score_gemma":0.0016069371,"threshold_uncertainty_score":0.02238512},"labels":[],"label_agreement":null},{"id":"W4389518365","doi":"10.18653/v1/2023.arabicnlp-1.67","title":"Frank at NADI 2023 Shared Task: Trio-Based Ensemble Approach for Arabic Dialect Identification","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Computer science; Task (project management); Arabic; Hyperparameter; Identification (biology); Ensemble learning; Artificial intelligence; Ensemble forecasting; Natural language processing; Machine learning; Speech recognition; Linguistics; Engineering","score_opus":0.027033761765870336,"score_gpt":0.2806555018163519,"score_spread":0.25362174005048155,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389518365","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.22117276,0.0023573579,0.61284804,0.0012035384,0.0015252642,0.0010629103,0.016181778,0.11317416,0.030474145],"genre_scores_gemma":[0.5367527,0.00037471676,0.39413604,0.00060451566,0.00029296492,0.00061464653,0.03238707,0.002747391,0.032090005],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9983523,0.00040338322,0.000086811655,0.0006327007,0.00030392158,0.00022097085],"domain_scores_gemma":[0.9985448,0.00021865909,0.000054790136,0.0004690462,0.0005179598,0.00019477063],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024246906,0.001753468,0.0013748772,0.0016879564,0.0017443824,0.0016982834,0.001954104,0.0017900596,0.007910914],"category_scores_gemma":[0.0033809994,0.00044519038,0.0011339184,0.0010244543,0.00029692255,0.002517344,0.0033828137,0.001999789,0.010186568],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010668788,0.00055972364,0.006480498,0.00025894135,0.00040541412,0.00053352286,0.0009922592,0.0152298575,0.05914584,0.0016144078,0.10284752,0.8108651],"study_design_scores_gemma":[0.00017038712,0.0007756474,0.012282729,0.00008535739,0.0004000811,0.0012333925,0.0012673821,0.76791745,0.10941108,0.008974298,0.09714835,0.0003338672],"about_ca_topic_score_codex":0.012969435,"about_ca_topic_score_gemma":0.020941757,"teacher_disagreement_score":0.012969435,"about_ca_system_score_codex":0.00060390675,"about_ca_system_score_gemma":0.0012801053,"threshold_uncertainty_score":0.026464641},"labels":[],"label_agreement":null},{"id":"W4389518418","doi":"10.18653/v1/2023.arabicnlp-1.15","title":"CamelParser2.0: A State-of-the-Art Dependency Parser for Arabic","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"York University; New York University Abu Dhabi","keywords":"Computer science; Treebank; Lexical analysis; Natural language processing; Parsing; Artificial intelligence; Dependency (UML); Modern Standard Arabic; Dependency grammar; Pipeline (software); Arabic; Python (programming language); Programming language; Linguistics","score_opus":0.01820604790219253,"score_gpt":0.27636426207507625,"score_spread":0.2581582141728837,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389518418","genre_codex":"software","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0094795665,0.00074436,0.29785836,0.00068159215,0.00055952254,0.00048054586,0.048677657,0.6261162,0.01540222],"genre_scores_gemma":[0.108223245,0.0008542711,0.58246046,0.0016620618,0.0003400338,0.001067559,0.19983248,0.078516506,0.02704334],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986744,0.00022590085,0.00014066354,0.0004667177,0.00036158858,0.00013074026],"domain_scores_gemma":[0.9972625,0.00078477786,0.000168413,0.0005900709,0.0010213426,0.00017293211],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018392175,0.0026674632,0.0011274591,0.0024627305,0.0011433107,0.002239065,0.004532447,0.001307485,0.050446726],"category_scores_gemma":[0.007238053,0.0018108926,0.0015828868,0.001431723,0.000758614,0.0047469456,0.0038310383,0.0031623498,0.048559435],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00067179074,0.00027507386,0.0032914167,0.0017541471,0.00028949653,0.0008558607,0.0008762121,0.006045514,0.031398863,0.0103412345,0.6543013,0.28989917],"study_design_scores_gemma":[0.0003504179,0.00023765514,0.0051046098,0.00039506805,0.00026319805,0.0022263434,0.00035564773,0.19488323,0.1298147,0.032329053,0.6334345,0.00060565287],"about_ca_topic_score_codex":0.006213591,"about_ca_topic_score_gemma":0.008698271,"teacher_disagreement_score":0.050446726,"about_ca_system_score_codex":0.0010561016,"about_ca_system_score_gemma":0.0025198236,"threshold_uncertainty_score":0.16876107},"labels":[],"label_agreement":null},{"id":"W4389518424","doi":"10.18653/v1/2023.arabicnlp-1.9","title":"Beyond English: Evaluating LLMs for Arabic Grammatical Error Correction","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Alliance de recherche numérique du Canada; Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Computer science; Arabic; Context (archaeology); Natural language processing; Artificial intelligence; Task (project management); Linguistics","score_opus":0.04045139428244435,"score_gpt":0.35261773387421874,"score_spread":0.3121663395917744,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389518424","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.76421845,0.0126642445,0.12286206,0.0028819784,0.0019119419,0.0007628574,0.0060610916,0.07512667,0.013510659],"genre_scores_gemma":[0.87159693,0.0012087064,0.09996064,0.001116924,0.00025278472,0.00033568713,0.017321998,0.0011486436,0.0070577264],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99856657,0.00053647475,0.00011388271,0.00050589535,0.00017248317,0.00010471145],"domain_scores_gemma":[0.99567837,0.0024447313,0.00014881273,0.00067776465,0.0008027604,0.00024753457],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031371883,0.0024387955,0.0011136105,0.0011796902,0.00072421576,0.0012821741,0.0023000936,0.0021430545,0.0031230813],"category_scores_gemma":[0.012297201,0.0004056492,0.0009680059,0.00085935777,0.00078789337,0.0026738655,0.0014776805,0.0029263883,0.0027240214],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018017231,0.0011162215,0.009265843,0.0009901467,0.0005346846,0.00047912728,0.00049829145,0.3359324,0.014606269,0.001356034,0.0321388,0.6012805],"study_design_scores_gemma":[0.0001616595,0.00058064284,0.0019248801,0.00007711714,0.00014831401,0.00013488968,0.00025248964,0.97180206,0.016408732,0.0020704714,0.0063850833,0.000053618834],"about_ca_topic_score_codex":0.0181478,"about_ca_topic_score_gemma":0.023596749,"teacher_disagreement_score":0.0181478,"about_ca_system_score_codex":0.0014522763,"about_ca_system_score_gemma":0.002311357,"threshold_uncertainty_score":0.036084354},"labels":[],"label_agreement":null},{"id":"W4389518436","doi":"10.18653/v1/2023.arabicnlp-1.62","title":"NADI 2023: The Fourth Nuanced Arabic Dialect Identification Shared Task","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":28,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Alliance de recherche numérique du Canada; Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Arabic; Task (project management); Identification (biology); Machine translation; Computer science; Natural language processing; Focus (optics); Artificial intelligence; Linguistics; Engineering","score_opus":0.017878874909435044,"score_gpt":0.276753934861565,"score_spread":0.25887505995213,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389518436","genre_codex":"empirical","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.51390225,0.0047433903,0.19325648,0.007467527,0.008900676,0.013531747,0.114114344,0.031717896,0.11236568],"genre_scores_gemma":[0.4119796,0.0004088025,0.33205676,0.0024413355,0.0010452168,0.0119426595,0.18833028,0.0032494846,0.048545923],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9861778,0.00567506,0.00085645996,0.0029268577,0.0027369813,0.0016268347],"domain_scores_gemma":[0.9795853,0.004562498,0.00067524624,0.005099503,0.005259377,0.0048182257],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013064125,0.002498763,0.0027053452,0.0027122588,0.0039287847,0.0040727076,0.0034347284,0.0032735455,0.010241595],"category_scores_gemma":[0.021480016,0.00065408385,0.001861947,0.0018684524,0.0015732956,0.003924225,0.014991593,0.004681718,0.012288742],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0058168396,0.0035936052,0.021779424,0.0031196496,0.0006959907,0.0018164628,0.012960404,0.009238255,0.08639425,0.007864744,0.42800677,0.41871366],"study_design_scores_gemma":[0.0022891802,0.00428318,0.054484464,0.0005603867,0.00038508678,0.003131991,0.014475585,0.061424635,0.0772004,0.025285905,0.75565475,0.00082444446],"about_ca_topic_score_codex":0.0101918,"about_ca_topic_score_gemma":0.017108362,"teacher_disagreement_score":0.013064125,"about_ca_system_score_codex":0.0025346496,"about_ca_system_score_gemma":0.007441364,"threshold_uncertainty_score":0.069090545},"labels":[],"label_agreement":null},{"id":"W4389518627","doi":"10.18653/v1/2023.findings-emnlp.997","title":"Cross-lingual Open-Retrieval Question Answering for African Languages","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia; McGill University; Mila - Quebec Artificial Intelligence Institute; University of Waterloo","funders":"","keywords":"Art; Philosophy; Theology; Art history; Humanities","score_opus":0.02895654518289965,"score_gpt":0.38721177407982016,"score_spread":0.3582552288969205,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389518627","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7012711,0.011193622,0.17269897,0.015488851,0.00078620424,0.00090368825,0.014314806,0.013577527,0.06976524],"genre_scores_gemma":[0.8926787,0.0015868243,0.07315033,0.0006715517,0.00017614757,0.0003095866,0.020641701,0.00067178323,0.010113395],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9968393,0.0016571679,0.00032133466,0.0005374603,0.00037068548,0.0002740661],"domain_scores_gemma":[0.9898281,0.006837578,0.00037882314,0.0011084643,0.0014259715,0.00042116197],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005220341,0.0004888517,0.0006527487,0.0030565187,0.0015182978,0.0029804304,0.00086783787,0.0014156819,0.0076413383],"category_scores_gemma":[0.018411754,0.0004671243,0.00063922553,0.0018955574,0.0008586278,0.010983513,0.0047693043,0.0012563511,0.0041748453],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014343938,0.00087851274,0.040967684,0.002050146,0.00023514457,0.0022841953,0.021561151,0.0032746652,0.035595313,0.038679693,0.116855554,0.7361835],"study_design_scores_gemma":[0.0003096316,0.00058155233,0.106213704,0.0015178934,0.00054105127,0.005198515,0.07796099,0.17368768,0.062738426,0.11372891,0.45704857,0.00047307042],"about_ca_topic_score_codex":0.008155095,"about_ca_topic_score_gemma":0.008429242,"teacher_disagreement_score":0.008155095,"about_ca_system_score_codex":0.00096701353,"about_ca_system_score_gemma":0.0015901347,"threshold_uncertainty_score":0.027608097},"labels":[],"label_agreement":null},{"id":"W4389519071","doi":"10.18653/v1/2023.emnlp-main.484","title":"Training Simultaneous Speech Translation with Robust and Random Wait-k-Tokens Strategy","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Security token; Computer science; Speech recognition; Latency (audio); Machine translation; Inference; Speech translation; Leverage (statistics); Encoder; Task (project management); Artificial intelligence; Natural language processing; Computer network","score_opus":0.04215042546696872,"score_gpt":0.27552553496520016,"score_spread":0.23337510949823143,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389519071","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06937122,0.0010373658,0.90742975,0.0004817406,0.0003245573,0.00016744502,0.0006764242,0.017948901,0.0025626468],"genre_scores_gemma":[0.60141194,0.00026977735,0.38349614,0.0007318487,0.00022041738,0.00044973247,0.0040373765,0.0013921311,0.007990556],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99851626,0.0004042078,0.00010234828,0.0006691446,0.00015468431,0.00015328088],"domain_scores_gemma":[0.99791914,0.0012420067,0.00009687375,0.00035563137,0.00027854485,0.000107884305],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017172784,0.0020774305,0.0015757018,0.00077107677,0.0008894988,0.0011541486,0.0020422088,0.0021074275,0.0052704834],"category_scores_gemma":[0.0050123287,0.00090539997,0.0014175241,0.00086763623,0.0010139071,0.0021563813,0.0015022713,0.0025192448,0.0047459854],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017954877,0.00042750893,0.0021434217,0.0004405326,0.000264297,0.00067610753,0.0003804755,0.32207397,0.039277278,0.007964287,0.018030265,0.60652643],"study_design_scores_gemma":[0.00009665271,0.00013069753,0.00023362612,0.00001488024,0.000039090533,0.00010082191,0.00006434542,0.9790364,0.013868449,0.004880295,0.0015106531,0.000024098532],"about_ca_topic_score_codex":0.0058246464,"about_ca_topic_score_gemma":0.011164279,"teacher_disagreement_score":0.0058246464,"about_ca_system_score_codex":0.0007728092,"about_ca_system_score_gemma":0.0023729175,"threshold_uncertainty_score":0.01763153},"labels":[],"label_agreement":null},{"id":"W4389519082","doi":"10.18653/v1/2023.emnlp-main.675","title":"Systematic word meta-sense extension","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Extension (predicate logic); Computer science; Literal and figurative language; Meaning (existential); Natural language processing; Metonymy; Word (group theory); Artificial intelligence; Analogy; Literal (mathematical logic); Semantics (computer science); Linguistics; Metaphor; Psychology; Algorithm; Programming language","score_opus":0.04220253460520556,"score_gpt":0.2952909589710333,"score_spread":0.25308842436582774,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389519082","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.26539642,0.00051126827,0.723681,0.0005762239,0.0001369657,0.0002853388,0.0010322443,0.004256025,0.004124624],"genre_scores_gemma":[0.778832,0.00014090938,0.21793021,0.00022977953,0.000034600092,0.00019154008,0.0012563142,0.00028024032,0.0011044036],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99801135,0.0006619197,0.00018250282,0.0008177056,0.00023636747,0.00009011199],"domain_scores_gemma":[0.9919761,0.004491499,0.0006628658,0.0021130661,0.0005812941,0.0001751734],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029532895,0.0013347764,0.00071521645,0.0012309197,0.0005931505,0.0014475397,0.001674212,0.0010760741,0.0027976339],"category_scores_gemma":[0.013834668,0.00060420984,0.002341433,0.00083875953,0.0013936596,0.0062029967,0.0031546943,0.0025356202,0.00064122875],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00081756664,0.0005960132,0.05081474,0.0012843626,0.00060423533,0.0009436547,0.0044776,0.123317346,0.061779715,0.07698768,0.007919403,0.6704578],"study_design_scores_gemma":[0.000064777894,0.00030570474,0.006415958,0.0001239879,0.00012240394,0.00054020545,0.00064709183,0.77841806,0.018643858,0.18588573,0.008736836,0.000095505544],"about_ca_topic_score_codex":0.00076969335,"about_ca_topic_score_gemma":0.0019472963,"teacher_disagreement_score":0.0029532895,"about_ca_system_score_codex":0.00052165467,"about_ca_system_score_gemma":0.001130676,"threshold_uncertainty_score":0.015618682},"labels":[],"label_agreement":null},{"id":"W4389519444","doi":"10.18653/v1/2023.emnlp-main.19","title":"Understanding Compositional Data Augmentation in Typologically Diverse Morphological Inflection","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Inflection; Complement (music); Phonotactics; Set (abstract data type); Artificial intelligence; Natural language processing; Linguistics; Phonology","score_opus":0.3123113950205255,"score_gpt":0.378841462431484,"score_spread":0.06653006741095846,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389519444","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5591352,0.0014927833,0.4258286,0.001003731,0.00015327396,0.00018919399,0.001986911,0.0055207373,0.00468951],"genre_scores_gemma":[0.72594297,0.00032058978,0.26603034,0.00024064386,0.000055425946,0.00023319176,0.0055973874,0.00034308777,0.0012363752],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9983394,0.0008238487,0.000109203225,0.00040591025,0.0002444356,0.00007712626],"domain_scores_gemma":[0.99097985,0.0059441035,0.00043962192,0.0018227113,0.00066963217,0.00014414296],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035098318,0.000839252,0.00056069985,0.0011742931,0.0006098081,0.0017987199,0.001434221,0.0011980878,0.0017738568],"category_scores_gemma":[0.014634554,0.00044868627,0.0008390749,0.001126348,0.0012607379,0.0028653052,0.002404646,0.0016407415,0.0011705813],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013840769,0.0005698319,0.038402345,0.0009981076,0.0002469937,0.0008678125,0.0023229602,0.12445366,0.08648256,0.013447376,0.008293433,0.7225309],"study_design_scores_gemma":[0.00009761815,0.00043563996,0.011248796,0.00014777535,0.00008266403,0.000870374,0.0013529029,0.8698169,0.06138025,0.03926067,0.015238741,0.00006764746],"about_ca_topic_score_codex":0.0009046459,"about_ca_topic_score_gemma":0.0027684947,"teacher_disagreement_score":0.0035098318,"about_ca_system_score_codex":0.0003681786,"about_ca_system_score_gemma":0.0007703342,"threshold_uncertainty_score":0.018562019},"labels":[],"label_agreement":null},{"id":"W4389519504","doi":"10.18653/v1/2023.wmt-1.65","title":"Metric Score Landscape Challenge (MSLC23): Understanding Metrics’ Performance on a Wider Landscape of Translation Quality","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Metric (unit); Quality (philosophy); Computer science; Task (project management); Machine translation; Set (abstract data type); Translation (biology); Artificial intelligence; Range (aeronautics); Machine learning; Natural language processing; Systems engineering; Engineering; Operations management","score_opus":0.13104806180208917,"score_gpt":0.3312294656292608,"score_spread":0.20018140382717164,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389519504","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.44315577,0.012857882,0.06735412,0.0056055477,0.002977439,0.0011770447,0.37852284,0.04305789,0.04529144],"genre_scores_gemma":[0.34267876,0.0007493656,0.053193923,0.00086171663,0.00042258663,0.0008450681,0.588129,0.0037756048,0.009344007],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99059373,0.003964731,0.0009768121,0.0015442573,0.0024479777,0.00047252918],"domain_scores_gemma":[0.9844988,0.0052736388,0.000996526,0.0040865312,0.0039704684,0.0011740311],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0068690265,0.0029469333,0.001963637,0.004666635,0.0015579833,0.0031171427,0.001846459,0.0027512112,0.008133251],"category_scores_gemma":[0.032628607,0.00033939237,0.0015956705,0.005422994,0.0014105325,0.003831084,0.004692997,0.0023683063,0.0074010976],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016397844,0.0006096955,0.022786895,0.002877302,0.00083415885,0.0005509459,0.0013855072,0.021495607,0.013974094,0.0059627527,0.6333719,0.29451126],"study_design_scores_gemma":[0.0017887456,0.0029597594,0.11664926,0.0008017706,0.000596188,0.0035886564,0.0027868822,0.20882517,0.035962556,0.03801897,0.58724093,0.0007811597],"about_ca_topic_score_codex":0.013388428,"about_ca_topic_score_gemma":0.020899007,"teacher_disagreement_score":0.013388428,"about_ca_system_score_codex":0.0019003458,"about_ca_system_score_gemma":0.0018412785,"threshold_uncertainty_score":0.036327362},"labels":[],"label_agreement":null},{"id":"W4389519548","doi":"10.18653/v1/2023.wmt-1.51","title":"Results of WMT23 Metrics Shared Task: Metrics Might Be Guilty but References Are Not Innocent","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"Bundesministerium für Bildung und Forschung; Deutsche Forschungsgemeinschaft","keywords":"George (robot); Task (project management); Computer science; Machine translation; Artificial intelligence; Engineering; Systems engineering","score_opus":0.07787230308748376,"score_gpt":0.3156563963070204,"score_spread":0.23778409321953664,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389519548","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2223986,0.019002903,0.037487265,0.023638334,0.018969823,0.0021002465,0.47784668,0.0915569,0.106999286],"genre_scores_gemma":[0.28206694,0.00095054525,0.040428296,0.0044721947,0.0014674673,0.0014162421,0.61954033,0.013548932,0.036109023],"study_design_codex":"not_applicable","study_design_gemma":"observational","domain_scores_codex":[0.9676339,0.014172668,0.0029117265,0.0042899605,0.00926548,0.0017263719],"domain_scores_gemma":[0.93191254,0.02469779,0.0027564308,0.017056564,0.01822443,0.005352273],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.020379936,0.005040739,0.0029063565,0.009366162,0.0045202943,0.0060578007,0.0033622044,0.0056557124,0.0204068],"category_scores_gemma":[0.09116952,0.0007894732,0.0027415613,0.005209699,0.0017774166,0.007811032,0.008688097,0.0033366662,0.024899961],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012175519,0.0004885866,0.005627603,0.0011389529,0.0004757126,0.00040318858,0.000523224,0.0024202599,0.0017903118,0.0018794588,0.9132499,0.07078524],"study_design_scores_gemma":[0.0028108559,0.0019687444,0.04599922,0.0012303997,0.00091373385,0.00204161,0.003951402,0.061188865,0.019982284,0.036135696,0.8229981,0.0007790632],"about_ca_topic_score_codex":0.034197785,"about_ca_topic_score_gemma":0.06814105,"teacher_disagreement_score":0.034197785,"about_ca_system_score_codex":0.0029803356,"about_ca_system_score_gemma":0.0043851254,"threshold_uncertainty_score":0.107780635},"labels":[],"label_agreement":null},{"id":"W4389519857","doi":"10.18653/v1/2023.findings-emnlp.494","title":"Towards Formality-Aware Neural Machine Translation by Leveraging Context Information","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"Institute for Information and Communications Technology Promotion; Samsung; Ministry of Science and ICT, South Korea; Korea Advanced Institute of Science and Technology; National Research Foundation of Korea; National Research Foundation","keywords":"Formality; Computer science; Machine translation; Classifier (UML); Artificial intelligence; Natural language processing; Machine learning; Linguistics","score_opus":0.02126738435969645,"score_gpt":0.27327086748564317,"score_spread":0.2520034831259467,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389519857","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029833108,0.0006699962,0.9648191,0.00034559236,0.000085983644,0.000043881904,0.00010658916,0.0022851662,0.0018104991],"genre_scores_gemma":[0.5509967,0.0005797324,0.44390777,0.0004351293,0.00015451801,0.00013977574,0.00086266187,0.00040874822,0.002514955],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9991055,0.00040602867,0.00006372249,0.00022831852,0.00013975704,0.00005669439],"domain_scores_gemma":[0.99854493,0.00073623157,0.00017679679,0.00026191492,0.00023581662,0.000044378026],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011577437,0.0010118411,0.00073031784,0.00085340545,0.00051137875,0.00103046,0.00071864005,0.00095807365,0.0015536048],"category_scores_gemma":[0.0046676625,0.0003583753,0.0006836271,0.00083961757,0.0007900323,0.0024885777,0.0015856654,0.0018718214,0.00087200897],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00038264657,0.00017771298,0.0017476692,0.00037132268,0.00012505947,0.00027354708,0.00039861337,0.20445205,0.056085438,0.031001942,0.004811135,0.7001727],"study_design_scores_gemma":[0.000021057207,0.000051868523,0.0002475264,0.00002348259,0.00002826742,0.000080215585,0.00003796462,0.955693,0.007876166,0.03371131,0.0022118469,0.000017170878],"about_ca_topic_score_codex":0.0021351266,"about_ca_topic_score_gemma":0.0045518423,"teacher_disagreement_score":0.0021351266,"about_ca_system_score_codex":0.0006990673,"about_ca_system_score_gemma":0.001009864,"threshold_uncertainty_score":0.0061228275},"labels":[],"label_agreement":null},{"id":"W4389519881","doi":"10.18653/v1/2023.findings-emnlp.542","title":"Knowledge-Selective Pretraining for Attribute Value Extraction","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Zhàng; Value (mathematics); Natural language processing; Computer science; Artificial intelligence; Philosophy; Linguistics; Psychology; Cognitive science; History; China; Machine learning; Archaeology","score_opus":0.041155446691850166,"score_gpt":0.35806146455532134,"score_spread":0.31690601786347117,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389519881","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06470673,0.001920196,0.89367753,0.00073080923,0.00045049837,0.00031030554,0.0026070012,0.023110118,0.012486771],"genre_scores_gemma":[0.4073205,0.0013088018,0.5580822,0.00073633826,0.00022100937,0.00048406553,0.015454499,0.0007618267,0.015630683],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9993363,0.00008627255,0.000058227917,0.0002857332,0.00011193976,0.00012140244],"domain_scores_gemma":[0.9987967,0.000537569,0.000058320766,0.00028383522,0.00027042863,0.00005307989],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00078314875,0.001171371,0.0009718214,0.0016628861,0.0008460472,0.0010824332,0.0019642278,0.0011595545,0.010775577],"category_scores_gemma":[0.0027053352,0.00068034424,0.0011057862,0.0020472773,0.0005342657,0.0027856869,0.0015601959,0.0024112724,0.0065316646],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00027448442,0.00026201853,0.0018615308,0.00016857985,0.00008874339,0.00014173661,0.00013292617,0.012143881,0.021036845,0.0033807044,0.027894016,0.93261456],"study_design_scores_gemma":[0.00007532625,0.00019449298,0.0041405014,0.000109813314,0.00018956984,0.00025755865,0.00029032506,0.88089854,0.068282515,0.019958444,0.02554541,0.00005747551],"about_ca_topic_score_codex":0.010377633,"about_ca_topic_score_gemma":0.022213899,"teacher_disagreement_score":0.010775577,"about_ca_system_score_codex":0.0007706478,"about_ca_system_score_gemma":0.0019778823,"threshold_uncertainty_score":0.036047876},"labels":[],"label_agreement":null},{"id":"W4389519920","doi":"10.18653/v1/2023.newsum-1","title":"Proceedings of the 4th New Frontiers in Summarization Workshop","year":2023,"lang":"en","type":"paratext","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Institute for Information and Communications Technology Promotion; Higher Education Commision, Pakistan; Iran Telecommunication Research Center; Klaus Tschira Stiftung; Agency for Science, Technology and Research; National Research Foundation of Korea; Deutscher Akademischer Austauschdienst; Natural Sciences and Engineering Research Council of Canada; Ministry of Science and ICT, South Korea; University of Tokyo; Japan Society for the Promotion of Science; National Research Foundation","keywords":"Automatic summarization; Computer science; Data science; Information retrieval","score_opus":0.014279124425913782,"score_gpt":0.26351173190843685,"score_spread":0.24923260748252307,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389519920","genre_codex":"commentary","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021126024,0.041447703,0.23885015,0.30145305,0.08752585,0.0016650913,0.016372614,0.01352024,0.27803925],"genre_scores_gemma":[0.066395,0.016874691,0.11407545,0.0151460515,0.017435974,0.001826506,0.039831452,0.00513636,0.7232786],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99710876,0.0011678796,0.00021042099,0.0003880087,0.00087587413,0.00024905102],"domain_scores_gemma":[0.9924529,0.0024565142,0.00016672812,0.0007564839,0.002956171,0.0012112303],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008338384,0.00095745304,0.0009980809,0.0015639904,0.0020188813,0.008129256,0.0024018097,0.0028608714,0.06446222],"category_scores_gemma":[0.009043557,0.00044381735,0.0010897638,0.0015041019,0.0011191943,0.0079568215,0.0029096138,0.003644863,0.029092439],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023171272,0.00012710609,0.00016693398,0.00028344046,0.000019948853,0.00015535124,0.00047708306,0.00057686196,0.0016153209,0.0060218237,0.89378655,0.09653778],"study_design_scores_gemma":[0.000032337073,0.00004715658,0.00046769175,0.0001390846,0.000015421087,0.00008804071,0.00045107267,0.0017400985,0.0012108121,0.006079831,0.98970616,0.00002234449],"about_ca_topic_score_codex":0.003866905,"about_ca_topic_score_gemma":0.0076568406,"teacher_disagreement_score":0.06446222,"about_ca_system_score_codex":0.0022264777,"about_ca_system_score_gemma":0.0031385573,"threshold_uncertainty_score":0.21564758},"labels":[],"label_agreement":null},{"id":"W4389520036","doi":"10.18653/v1/2023.findings-emnlp.82","title":"The Past, Present, and Future of Typological Databases in NLP","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; Mila - Quebec Artificial Intelligence Institute","funders":"McGill University","keywords":"Categorical variable; Computer science; Typology; Natural language processing; Variation (astronomy); Rule-based machine translation; Resource (disambiguation); Artificial intelligence; Argument (complex analysis); Computational linguistics; Coding (social sciences); Linguistic typology; Linguistics; Database; Geography; Sociology; Machine learning","score_opus":0.027888613215679647,"score_gpt":0.31348798577571435,"score_spread":0.2855993725600347,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389520036","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06983934,0.07190867,0.41049725,0.37651703,0.0019327982,0.00025368712,0.0018684953,0.00070876157,0.066474065],"genre_scores_gemma":[0.64715517,0.037640244,0.29335138,0.012499468,0.002541014,0.00096236425,0.0017085724,0.00050840003,0.0036333075],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9130268,0.06049082,0.0070453654,0.007060597,0.010241331,0.0021351231],"domain_scores_gemma":[0.6326394,0.27734363,0.012357033,0.04897505,0.023505537,0.005179352],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1462796,0.0007559431,0.0020264091,0.011825847,0.008491259,0.039936673,0.007561402,0.009541632,0.0061950153],"category_scores_gemma":[0.20226821,0.0017239026,0.0011900647,0.025816709,0.03690281,0.104552545,0.01388621,0.011506555,0.0012802278],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000907362,0.000045290828,0.0047760345,0.00046805336,0.000028206588,0.0000981052,0.005827958,0.0022342657,0.00017017369,0.90555286,0.004582355,0.07612596],"study_design_scores_gemma":[0.000016903692,0.000020918671,0.0013597703,0.0015184169,0.000019137411,0.00020293807,0.011361674,0.00724873,0.00034152635,0.91253895,0.065282874,0.00008826209],"about_ca_topic_score_codex":0.0074177175,"about_ca_topic_score_gemma":0.0049428702,"teacher_disagreement_score":0.1462796,"about_ca_system_score_codex":0.011530345,"about_ca_system_score_gemma":0.0103279725,"threshold_uncertainty_score":0.77360976},"labels":[],"label_agreement":null},{"id":"W4389520667","doi":"10.18653/v1/2023.emnlp-main.396","title":"Advancements in Arabic Grammatical Error Detection and Correction: An Empirical Investigation","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"York University; New York University Abu Dhabi","keywords":"Computer science; Preprocessor; Natural language processing; Artificial intelligence; Arabic; Benchmark (surveying); Transformer; Task (project management); Language model; Class (philosophy); Linguistics","score_opus":0.03459840212741797,"score_gpt":0.3380574985760335,"score_spread":0.3034590964486155,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389520667","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.94372964,0.0048348145,0.019637637,0.0022732995,0.000431856,0.00023263892,0.010921572,0.005073813,0.012864805],"genre_scores_gemma":[0.9383354,0.0011556788,0.02301728,0.0006655202,0.00015834552,0.0001679147,0.0322878,0.00082956045,0.0033824136],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.991227,0.004116808,0.0006542671,0.0022884454,0.0014212073,0.00029239157],"domain_scores_gemma":[0.93567353,0.042975955,0.002853632,0.01101111,0.006389653,0.0010961426],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011423481,0.0019695284,0.0009352468,0.0031640772,0.001429838,0.0022747826,0.002476346,0.0020914455,0.0033528113],"category_scores_gemma":[0.06072438,0.0005443959,0.0010487206,0.0033245808,0.0015704264,0.0059365937,0.002981224,0.0036059492,0.0040105996],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016203776,0.001619358,0.29674703,0.0017511655,0.00062277704,0.0013773846,0.0040129293,0.041360363,0.0071314704,0.004899175,0.0889272,0.5499308],"study_design_scores_gemma":[0.00030686296,0.001532826,0.25801927,0.0009591535,0.0007283566,0.0056065214,0.008428855,0.57666653,0.02706228,0.01991328,0.10037431,0.0004017985],"about_ca_topic_score_codex":0.010228384,"about_ca_topic_score_gemma":0.011787609,"teacher_disagreement_score":0.011423481,"about_ca_system_score_codex":0.00146365,"about_ca_system_score_gemma":0.001230772,"threshold_uncertainty_score":0.060413897},"labels":[],"label_agreement":null},{"id":"W4389521068","doi":"10.18653/v1/2023.crac-main.6","title":"The pragmatics of characters’ mental perspectives in pronominal reference resolution","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Pragmatics; Computer science; Pronoun; Linguistics; Resolution (logic); Interpretation (philosophy); Cognition; Cognitive science; Cognitive linguistics; Computational linguistics; Natural language processing; Artificial intelligence; Psychology; Philosophy","score_opus":0.028589603729367023,"score_gpt":0.30237418898344076,"score_spread":0.27378458525407373,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389521068","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.40544498,0.002284071,0.2735914,0.0138936,0.00041197563,0.00013635814,0.00025625466,0.00044641257,0.30353498],"genre_scores_gemma":[0.9831693,0.00023441354,0.014486872,0.0001377502,0.000055210567,0.00002932888,0.000052120246,0.00009698207,0.001737987],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98817784,0.008825584,0.00030401925,0.0009442575,0.0012654249,0.0004828364],"domain_scores_gemma":[0.98734784,0.008094151,0.0011347695,0.0013514601,0.0014865466,0.00058516685],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008197906,0.0004068765,0.0003634759,0.0019302865,0.0028924292,0.009638899,0.0010343496,0.0020510587,0.0046411543],"category_scores_gemma":[0.03617568,0.00095082057,0.000681947,0.0010948391,0.010662607,0.013730667,0.0038942178,0.002475085,0.00068300916],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002263918,0.00004360798,0.0025228283,0.00019566456,0.000039255807,0.00040537398,0.05582724,0.0012592392,0.0061883284,0.9106975,0.0014655207,0.021128882],"study_design_scores_gemma":[0.00013130048,0.00013918828,0.010835396,0.00022530329,0.00009856082,0.00081108755,0.027857516,0.014658885,0.006123222,0.8919307,0.047026236,0.00016274033],"about_ca_topic_score_codex":0.0030249073,"about_ca_topic_score_gemma":0.0029163891,"teacher_disagreement_score":0.009638899,"about_ca_system_score_codex":0.0019864512,"about_ca_system_score_gemma":0.0012018307,"threshold_uncertainty_score":0.043355167},"labels":[],"label_agreement":null},{"id":"W4389523795","doi":"10.18653/v1/2023.emnlp-main.11","title":"Better Quality Pre-training Data and T5 Models for African Languages","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada; Compute Canada","keywords":"Computer science; Natural language processing; Quality (philosophy); Training set; Training (meteorology); Artificial intelligence; Philosophy; Geography; Epistemology","score_opus":0.12010097908356883,"score_gpt":0.39519336307749964,"score_spread":0.2750923839939308,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389523795","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5024521,0.0119606955,0.393058,0.0091318935,0.0035254143,0.0006491548,0.02699649,0.032132145,0.020094113],"genre_scores_gemma":[0.69140136,0.0013474697,0.21883184,0.0012560949,0.00036945325,0.00044685559,0.071512364,0.00278167,0.012053015],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99840444,0.00075937895,0.00012961976,0.00036528066,0.00012791065,0.00021336901],"domain_scores_gemma":[0.99465114,0.0028442491,0.00013878992,0.000936891,0.0012007201,0.00022817955],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004301969,0.002217225,0.0011189923,0.0016084411,0.0013493617,0.0018002061,0.0024317177,0.002740543,0.014345179],"category_scores_gemma":[0.01405524,0.000837882,0.0018158574,0.0013835792,0.00056584383,0.005729519,0.0023088842,0.0041452004,0.008157163],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0047976784,0.0011060799,0.018170582,0.0009123211,0.00048212375,0.0006562431,0.0005600964,0.14587787,0.014084331,0.0059800902,0.12667257,0.6807],"study_design_scores_gemma":[0.00051854196,0.000410283,0.0063588615,0.0003354104,0.00024514843,0.00021453862,0.0010329945,0.94182867,0.017095793,0.010870407,0.020977741,0.00011151888],"about_ca_topic_score_codex":0.025331896,"about_ca_topic_score_gemma":0.032667965,"teacher_disagreement_score":0.025331896,"about_ca_system_score_codex":0.0010623119,"about_ca_system_score_gemma":0.0018271366,"threshold_uncertainty_score":0.050368845},"labels":[],"label_agreement":null},{"id":"W4389524428","doi":"10.18653/v1/2023.nlposs-1.25","title":"The Vault: A Comprehensive Multilingual Dataset for Advancing Code Understanding and Generation","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Mila - Quebec Artificial Intelligence Institute; McGill University","funders":"","keywords":"Open source software; Computer science; Code (set theory); Open source; Vault (architecture); Ho chi minh; Natural language processing; Artificial intelligence; Software; Programming language; History; Cartography; Geography; Archaeology","score_opus":0.09684993925383269,"score_gpt":0.36150218654341576,"score_spread":0.2646522472895831,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389524428","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019063786,0.0021565007,0.02338144,0.0009705604,0.00045491333,0.0003693143,0.9076862,0.03740224,0.008515033],"genre_scores_gemma":[0.0088661825,0.00023619477,0.018430151,0.00014423736,0.000033326196,0.0003174604,0.969349,0.001340408,0.0012829606],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9960001,0.00080140785,0.0005536619,0.0010429625,0.0012758723,0.00032594605],"domain_scores_gemma":[0.9936799,0.0014464513,0.0005701106,0.0020271384,0.0016200289,0.0006563298],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002484832,0.0023848973,0.0010893403,0.0075218924,0.0021037154,0.002670268,0.0035533938,0.0034160987,0.0068172235],"category_scores_gemma":[0.010487211,0.0008836283,0.0017287139,0.0050201043,0.0010486785,0.004317602,0.0058961026,0.0027680183,0.015105864],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005904198,0.00028184374,0.009377114,0.0021124177,0.00021873991,0.00049998105,0.0006603181,0.004089077,0.0069941385,0.005770979,0.90440786,0.064997055],"study_design_scores_gemma":[0.0005736642,0.00020128716,0.01637899,0.00059942453,0.00015399669,0.00059305684,0.00082953274,0.027698947,0.013286936,0.011888011,0.9275185,0.0002776637],"about_ca_topic_score_codex":0.022493148,"about_ca_topic_score_gemma":0.048597503,"teacher_disagreement_score":0.022493148,"about_ca_system_score_codex":0.0015382189,"about_ca_system_score_gemma":0.00383472,"threshold_uncertainty_score":0.044724464},"labels":[],"label_agreement":null},{"id":"W4389524555","doi":"10.18653/v1/2023.findings-emnlp.936","title":"RWKV: Reinventing RNNs for the Transformer Era","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":303,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Nexen (Canada)","funders":"","keywords":"STELLA (programming language); Zhàng; Art history; Art; Humanities; Philosophy; History; Archaeology; China","score_opus":0.02679878535284542,"score_gpt":0.30042658354931817,"score_spread":0.27362779819647276,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389524555","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0080895955,0.001755177,0.7423385,0.00077424507,0.001635239,0.00014694165,0.0026786516,0.2345413,0.008040421],"genre_scores_gemma":[0.18761922,0.0018657014,0.71837485,0.0017200469,0.0006159995,0.0004378697,0.017087486,0.037807588,0.034471285],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99858344,0.00038628915,0.000107535234,0.00046410543,0.00033045284,0.00012821659],"domain_scores_gemma":[0.9976357,0.0011283833,0.0000882548,0.00063326454,0.00041929664,0.000095040654],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027197422,0.002035334,0.0009737381,0.0015341362,0.00075489184,0.0023015924,0.0037058122,0.001956325,0.0210024],"category_scores_gemma":[0.009341056,0.0013380877,0.0012841503,0.00084990176,0.0007677916,0.008092727,0.003730047,0.0036435851,0.018289436],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010661946,0.0001990066,0.0011707227,0.0006762093,0.00032457366,0.00042719548,0.00032948834,0.045108724,0.014747644,0.01746338,0.15462229,0.7638646],"study_design_scores_gemma":[0.00024085309,0.00019709724,0.00031116407,0.00015365283,0.00012333042,0.00023841445,0.000113891234,0.8455394,0.022310218,0.038940735,0.09174083,0.00009045861],"about_ca_topic_score_codex":0.009399713,"about_ca_topic_score_gemma":0.020411586,"teacher_disagreement_score":0.0210024,"about_ca_system_score_codex":0.00082353246,"about_ca_system_score_gemma":0.0014613068,"threshold_uncertainty_score":0.07026005},"labels":[],"label_agreement":null},{"id":"W4389559334","doi":"10.1007/s00224-023-10154-8","title":"A Closer Look at the Expressive Power of Logics Based on Word Equations","year":2023,"lang":"en","type":"article","venue":"Theory of Computing Systems","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Deutsche Forschungsgemeinschaft","keywords":"Recursively enumerable language; Undecidable problem; Algorithm; Computer science; Word (group theory); Recursively enumerable set; Set (abstract data type); Mathematics; Artificial intelligence; Discrete mathematics; Decidability; Programming language; Geometry","score_opus":0.02294902408483427,"score_gpt":0.27999414746681406,"score_spread":0.25704512338197977,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389559334","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03454218,0.007792087,0.854416,0.01988398,0.0008606819,0.00011763814,0.0008262732,0.0020176787,0.07954351],"genre_scores_gemma":[0.56918716,0.0070965965,0.38477436,0.009295818,0.0017143139,0.00024814298,0.0014948796,0.001295105,0.02489352],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9875754,0.0062176744,0.001013355,0.0017508328,0.0027171248,0.00072550244],"domain_scores_gemma":[0.9727003,0.021283321,0.00086491834,0.0030401098,0.0015924374,0.0005189295],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013203257,0.0011002666,0.001175121,0.002730093,0.002054336,0.010485643,0.0033599841,0.0020071445,0.010547323],"category_scores_gemma":[0.022415861,0.0015107939,0.0046477164,0.0029711244,0.008626266,0.026934585,0.006992228,0.009984249,0.0015786919],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007690156,0.00002952334,0.00019510806,0.00013349725,0.00004261363,0.00012138103,0.00058503874,0.0040427623,0.0012235708,0.9781728,0.0030587958,0.012318104],"study_design_scores_gemma":[0.000035635614,0.000025906136,0.00008945961,0.00008464754,0.00003706468,0.00009050291,0.0001674288,0.021275632,0.0013801948,0.95588046,0.020895666,0.00003745331],"about_ca_topic_score_codex":0.004664819,"about_ca_topic_score_gemma":0.0043436023,"teacher_disagreement_score":0.013203257,"about_ca_system_score_codex":0.0053939577,"about_ca_system_score_gemma":0.0018558159,"threshold_uncertainty_score":0.069826365},"labels":[],"label_agreement":null},{"id":"W4389616800","doi":"10.4995/jclr.2023.20408","title":"Autosupervisión de Alucinaciones en Grandes Modelos del Lenguaje: LLteaM","year":2023,"lang":"es","type":"article","venue":"Journal of Computer-Assisted Linguistic Research","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"WSP (Canada)","funders":"","keywords":"Humanities; Philosophy; Art","score_opus":0.07493056431312235,"score_gpt":0.4078939249829552,"score_spread":0.33296336066983284,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389616800","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.21087646,0.0014369746,0.7378401,0.002223045,0.0002227729,0.00036057472,0.0005039904,0.0052266405,0.04130944],"genre_scores_gemma":[0.8319693,0.0007551806,0.14712922,0.00035667053,0.00006156675,0.00034453493,0.0007261718,0.0005442854,0.018113086],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99733114,0.00089428265,0.00015211888,0.00076701737,0.0006423027,0.00021325085],"domain_scores_gemma":[0.9932679,0.0030123242,0.0006151098,0.0014713246,0.0013086954,0.00032455815],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032054773,0.0009885352,0.0007299766,0.0013618915,0.0010186537,0.00504533,0.0016464439,0.0016271453,0.0068715313],"category_scores_gemma":[0.012242831,0.0006561161,0.0012931641,0.0009047445,0.002032383,0.005649358,0.0028300998,0.0020741166,0.0015760395],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011471298,0.0006103989,0.02954149,0.0011286943,0.0003124598,0.0012412512,0.009227505,0.19948839,0.025157632,0.2696634,0.009371306,0.45311043],"study_design_scores_gemma":[0.00007525766,0.0003469484,0.005691834,0.0002918804,0.00017091843,0.0004096516,0.001916212,0.78333706,0.011983534,0.1582491,0.037419487,0.00010803299],"about_ca_topic_score_codex":0.009612366,"about_ca_topic_score_gemma":0.007991914,"teacher_disagreement_score":0.009612366,"about_ca_system_score_codex":0.0024318427,"about_ca_system_score_gemma":0.0028258841,"threshold_uncertainty_score":0.022987545},"labels":[],"label_agreement":null},{"id":"W4389810683","doi":"","title":"CATMuS-Medieval: Consistent Approaches to Transcribing ManuScripts: A generalized set of guidelines and models for Latin scripts from Middle Ages (8th--16th century)","year":2024,"lang":"en","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canada Research Chairs; University of Toronto; Université de Montréal","funders":"","keywords":"Computer science; Art; Natural language processing; History; Classics","score_opus":0.17051253424152404,"score_gpt":0.27818196954543895,"score_spread":0.1076694353039149,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389810683","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03265939,0.0005372679,0.94926,0.00070288585,0.0000564422,0.00040075328,0.004100267,0.0036365222,0.0086464025],"genre_scores_gemma":[0.27813858,0.00036274083,0.705053,0.00022405657,0.000042194966,0.0007005145,0.009659298,0.0012533274,0.004566351],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9957957,0.0021222061,0.00043183705,0.0009634555,0.00053094735,0.00015575488],"domain_scores_gemma":[0.9929489,0.002954305,0.00046388732,0.0016154982,0.0018375642,0.00017982892],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004462326,0.0008714952,0.0006397509,0.0024236827,0.0014937075,0.004720775,0.0026218735,0.0012508811,0.006093347],"category_scores_gemma":[0.01378204,0.0011216624,0.0014608998,0.0018606891,0.0019858803,0.00404712,0.0025620512,0.0018157338,0.002483349],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006894728,0.0002689233,0.019083673,0.00095478253,0.00023793404,0.0005508321,0.01450545,0.09017633,0.010009538,0.2514635,0.025338415,0.5867211],"study_design_scores_gemma":[0.000097712786,0.00016169698,0.007701426,0.0005658427,0.00016245901,0.0004725075,0.005116228,0.62608755,0.0139720915,0.2512456,0.09428166,0.00013529565],"about_ca_topic_score_codex":0.018091762,"about_ca_topic_score_gemma":0.04164919,"teacher_disagreement_score":0.018091762,"about_ca_system_score_codex":0.002193041,"about_ca_system_score_gemma":0.0043517333,"threshold_uncertainty_score":0.035972953},"labels":[],"label_agreement":null},{"id":"W4389921680","doi":"10.1093/oxfordhb/9780198852889.013.52","title":"Word Classes in Lexical Functional Grammar","year":2023,"lang":"en","type":"book-chapter","venue":"Oxford University Press eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Lexical functional grammar; Linguistics; Part of speech; Computer science; Word (group theory); Grammar; Natural language processing; Lexical item; Lexical grammar; Artificial intelligence; Word formation; Emergent grammar; Generative grammar; Mildly context-sensitive grammar formalism; Philosophy","score_opus":0.035498709802130476,"score_gpt":0.22535999942878246,"score_spread":0.189861289626652,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389921680","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03282029,0.01800795,0.2679758,0.008383221,0.00073519186,0.00007839683,0.0006204998,0.00066401693,0.6707147],"genre_scores_gemma":[0.76631534,0.008727291,0.088861,0.0017677433,0.00064425555,0.00019367436,0.0013467611,0.0007620638,0.13138199],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99956006,0.00017520017,0.000023675288,0.00008079242,0.00012495511,0.000035179422],"domain_scores_gemma":[0.9995097,0.0003287815,0.00001664223,0.000053236956,0.000070708404,0.00002095955],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006110368,0.00040705304,0.00031364275,0.0013280811,0.0007386376,0.0027761185,0.00046338764,0.0006596857,0.009172296],"category_scores_gemma":[0.0014574281,0.0002679558,0.0003458048,0.0013828501,0.0032821079,0.004636073,0.0009232989,0.0011968726,0.0015053987],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000029974851,0.0000026751463,0.00007394749,0.000032298612,0.0000016234809,0.000021044909,0.0002684449,0.00043266368,0.00017739742,0.9749015,0.0044990866,0.019586386],"study_design_scores_gemma":[0.0000013831394,0.0000021265464,0.00012361711,0.000040239567,0.0000017876262,0.00003857929,0.000080077996,0.0014378448,0.00011416677,0.94401586,0.05414093,0.0000033773338],"about_ca_topic_score_codex":0.0024443076,"about_ca_topic_score_gemma":0.0023616664,"teacher_disagreement_score":0.009172296,"about_ca_system_score_codex":0.0021612542,"about_ca_system_score_gemma":0.00067085034,"threshold_uncertainty_score":0.030684352},"labels":[],"label_agreement":null},{"id":"W4390025639","doi":"10.1353/dic.2023.a915066","title":"Modern Wendat Lexicography: Using XML to Reflect the Grammar and Lexicon of an Iroquoian Language","year":2023,"lang":"en","type":"article","venue":"Dictionaries","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Victoria","funders":"","keywords":"Computer science; Linguistics; Grammar; Lexicon; Schema (genetic algorithms); XML; Natural language processing; Verb; Programming language; Artificial intelligence; World Wide Web; Information retrieval; Philosophy","score_opus":0.02369973581418356,"score_gpt":0.31908837601863943,"score_spread":0.2953886402044559,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4390025639","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11299762,0.0007089441,0.79795635,0.0033954703,0.00049816316,0.000539257,0.007014729,0.0050959196,0.0717935],"genre_scores_gemma":[0.3433123,0.0006694864,0.62752604,0.00054843014,0.000070232134,0.000294073,0.00585478,0.0012529321,0.020471713],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9992181,0.0003091593,0.00016880812,0.00013683164,0.00013163463,0.000035452435],"domain_scores_gemma":[0.99735045,0.0007555442,0.00021914489,0.00083145226,0.00078245933,0.000060996696],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021694119,0.00023271056,0.000175823,0.0020447515,0.0012325065,0.003944421,0.0006928612,0.00044661755,0.004048454],"category_scores_gemma":[0.0035223255,0.00034316862,0.00023372586,0.0022983633,0.0017933737,0.0036785512,0.0012550611,0.0010990488,0.0007918937],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017502843,0.00010436505,0.021536905,0.00056491623,0.000044595763,0.0011238719,0.03541764,0.005584554,0.019513525,0.5989936,0.029216358,0.28772464],"study_design_scores_gemma":[0.000027761185,0.000054173874,0.0062270053,0.00048646878,0.000028932023,0.0010879475,0.011438211,0.019885115,0.022040315,0.035439707,0.90320516,0.000079187186],"about_ca_topic_score_codex":0.024434153,"about_ca_topic_score_gemma":0.039824404,"teacher_disagreement_score":0.024434153,"about_ca_system_score_codex":0.002689608,"about_ca_system_score_gemma":0.0027091263,"threshold_uncertainty_score":0.048583865},"labels":[],"label_agreement":null},{"id":"W4390090535","doi":"10.7557/12.7084","title":"Samtaler i korpusformat","year":2023,"lang":"da","type":"article","venue":"Nordlyd","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Philosophy; Humanities","score_opus":0.022423623689751405,"score_gpt":0.2910751104560893,"score_spread":0.2686514867663379,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4390090535","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016845113,0.0024545423,0.11328391,0.008115834,0.008189247,0.0020954448,0.24315663,0.081155784,0.5247035],"genre_scores_gemma":[0.060681716,0.0029545424,0.11025426,0.0036687036,0.0018112537,0.0036136622,0.24051154,0.054661386,0.5218429],"study_design_codex":"not_applicable","study_design_gemma":"observational","domain_scores_codex":[0.99543506,0.00083291595,0.0005537646,0.0012457286,0.0016707075,0.00026175688],"domain_scores_gemma":[0.99230134,0.0022237643,0.0003529117,0.001990537,0.002626745,0.000504744],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0041881264,0.0018811569,0.0010377117,0.0032543144,0.0026145135,0.0051607667,0.0018523319,0.001653362,0.41107264],"category_scores_gemma":[0.016115261,0.0011355841,0.0013499445,0.0037809233,0.0012368708,0.006130745,0.006660954,0.0019490274,0.31985813],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00045479447,0.00008565063,0.0025279678,0.0009662208,0.000033382985,0.00058897026,0.002469246,0.00039595203,0.0029295322,0.011646214,0.76575655,0.21214552],"study_design_scores_gemma":[0.00003409904,0.000034132165,0.0013775638,0.00028489254,0.000016201533,0.00033325327,0.0007684661,0.0004553903,0.0023002625,0.0038495357,0.9905014,0.000044783377],"about_ca_topic_score_codex":0.0071707615,"about_ca_topic_score_gemma":0.0070045013,"teacher_disagreement_score":0.41107264,"about_ca_system_score_codex":0.0017271957,"about_ca_system_score_gemma":0.003412437,"threshold_uncertainty_score":0.8400334},"labels":[],"label_agreement":null},{"id":"W4390343500","doi":"10.5430/wjel.v14n2p74","title":"Exploring the Role of Machine Translation in Translating English Collocations into Arabic: Insights from Student Translators","year":2023,"lang":"en","type":"article","venue":"World Journal of English Language","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"Prince Sattam bin Abdulaziz University","keywords":"Computer science; Machine translation; Arabic; Context (archaeology); Vocabulary; Facilitator; Natural language processing; Linguistics; Reading (process); Artificial intelligence; Psychology","score_opus":0.02010840904054541,"score_gpt":0.27122598290726363,"score_spread":0.25111757386671824,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4390343500","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99124444,0.00032124613,0.0018320899,0.0014561706,0.000020799216,0.00003940679,0.00001775128,0.00001350857,0.005054577],"genre_scores_gemma":[0.9956573,0.00054525863,0.001061697,0.00042271928,0.00001420885,0.00004611529,0.000021362663,0.00001948598,0.0022118508],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9932702,0.004931385,0.00024858132,0.0003444506,0.0007213349,0.00048410494],"domain_scores_gemma":[0.9835343,0.011534115,0.0013118922,0.00040851705,0.0021062111,0.0011050127],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009081698,0.00052293565,0.00055137207,0.0016781547,0.0046649342,0.0063222772,0.0013338748,0.0020684258,0.001594842],"category_scores_gemma":[0.019552777,0.0005384953,0.0004018986,0.0013926041,0.004772397,0.0044277706,0.0036824117,0.0026831946,0.00069294404],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000020163154,0.000092925904,0.0076926723,0.00006671165,0.000004110948,0.0006105099,0.9828384,0.000047709505,0.0009642186,0.0005846991,0.0002232853,0.0068546524],"study_design_scores_gemma":[0.000006999614,0.00017850981,0.0057540922,0.00012703372,0.000009726653,0.00069214846,0.9820463,0.00039471313,0.0010132274,0.00076586753,0.008990722,0.000020615826],"about_ca_topic_score_codex":0.0020005524,"about_ca_topic_score_gemma":0.0038349747,"teacher_disagreement_score":0.009081698,"about_ca_system_score_codex":0.001621746,"about_ca_system_score_gemma":0.002587525,"threshold_uncertainty_score":0.048029125},"labels":[],"label_agreement":null},{"id":"W4390515814","doi":"10.1007/978-981-99-8602-6","title":"Machine Translation and Foreign Language Learning","year":2023,"lang":"en","type":"book","venue":"New frontiers in translation studies","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Shanghai International Studies University; University of Ottawa","keywords":"Machine translation; Foreign language; Computer science; Translation (biology); Natural language processing; Linguistics; Artificial intelligence; Chemistry; Philosophy","score_opus":0.034878136764572014,"score_gpt":0.2993941935735957,"score_spread":0.2645160568090237,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4390515814","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.002091907,0.113345034,0.19975744,0.0076685897,0.0075914655,0.00008235398,0.0007429529,0.0019409304,0.6667793],"genre_scores_gemma":[0.027258154,0.05522207,0.0819376,0.0017476346,0.0046400656,0.00015699922,0.0017512768,0.0015079351,0.82577825],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99954045,0.0001200295,0.000023911854,0.00006892872,0.00021694196,0.000029699046],"domain_scores_gemma":[0.99926454,0.00047516683,0.000026148686,0.000105132975,0.00010317073,0.000025762985],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00049612793,0.0010416945,0.0011902347,0.0023671607,0.0009797688,0.004362991,0.0010863893,0.0012893351,0.058555856],"category_scores_gemma":[0.0020775185,0.0005342074,0.0005608519,0.004488758,0.0016567941,0.0062246835,0.0011602493,0.0021057003,0.027608918],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003419101,0.000036304446,0.0000645272,0.0004975806,0.0000164843,0.00012945964,0.00020220366,0.000848105,0.0013415171,0.40026483,0.22392951,0.37263533],"study_design_scores_gemma":[0.000008935608,0.0000208001,0.00016383325,0.00015580523,0.000011999158,0.00041098683,0.00009013755,0.0029034282,0.0015683597,0.2579837,0.73666716,0.000014791371],"about_ca_topic_score_codex":0.00089455344,"about_ca_topic_score_gemma":0.001476402,"teacher_disagreement_score":0.058555856,"about_ca_system_score_codex":0.0009531925,"about_ca_system_score_gemma":0.00081919174,"threshold_uncertainty_score":0.19588888},"labels":[],"label_agreement":null},{"id":"W4390623405","doi":"10.26615/978-954-452-092-2_052","title":"Discourse Analysis of Argumentative Essays of English Learners based on CEFR Level","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Argumentative; Linguistics; Computer science; Rhetorical question; Contingency; College English; Natural language processing; Artificial intelligence","score_opus":0.02877514458166023,"score_gpt":0.3286376986099444,"score_spread":0.29986255402828416,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4390623405","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9892596,0.0003855216,0.0037613802,0.00008981208,0.000019840094,0.000060052393,0.00040022447,0.000060910366,0.0059626647],"genre_scores_gemma":[0.99089456,0.00018385735,0.005603382,0.000020406766,0.000019242807,0.00007562456,0.00075567496,0.00003814544,0.002409115],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9965064,0.0017316532,0.00032064517,0.00046408735,0.0008459961,0.00013122233],"domain_scores_gemma":[0.9466101,0.039328773,0.0049572163,0.0018498026,0.0063731615,0.00088091363],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034827073,0.00033898817,0.00037400343,0.004920297,0.0010236542,0.0022987025,0.0003610458,0.00045393797,0.0025918274],"category_scores_gemma":[0.040751576,0.0001279385,0.00017242551,0.0025875915,0.0007391893,0.001693359,0.0014653549,0.0006599545,0.0005400502],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000945679,0.0006301404,0.26840895,0.0015482947,0.00018180732,0.0013438977,0.3552644,0.0011537737,0.09499543,0.009664639,0.0018764336,0.2639866],"study_design_scores_gemma":[0.00004746975,0.00052294857,0.7461095,0.00052074087,0.00012854465,0.001884915,0.14401357,0.009213207,0.034030505,0.006370428,0.05700055,0.00015764494],"about_ca_topic_score_codex":0.0009787132,"about_ca_topic_score_gemma":0.0018058881,"teacher_disagreement_score":0.004920297,"about_ca_system_score_codex":0.00075883995,"about_ca_system_score_gemma":0.00045560588,"threshold_uncertainty_score":0.01841855},"labels":[],"label_agreement":null},{"id":"W4390682031","doi":"10.26615/978-954-452-092-2_121","title":"Hindi to Dravidian Language Neural Machine Translation Systems","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"BC Research (Canada)","funders":"","keywords":"Agglutinative language; Hindi; Computer science; Tamil; Malayalam; Machine translation; Artificial intelligence; Natural language processing; Speech recognition; Parsing; Linguistics","score_opus":0.018027516111166052,"score_gpt":0.29181174822471534,"score_spread":0.2737842321135493,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4390682031","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.46005994,0.00663093,0.25973716,0.00280584,0.0015391357,0.00051759207,0.014520808,0.017797083,0.23639148],"genre_scores_gemma":[0.7785588,0.0013006201,0.14601336,0.0008047085,0.00011206092,0.0001286867,0.012277392,0.00037640252,0.060428016],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99978393,0.00004376546,0.000024187246,0.00006806245,0.000054313045,0.000025691072],"domain_scores_gemma":[0.999689,0.00007126543,0.000021036307,0.000070218666,0.00013369114,0.000014757663],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00030129898,0.0004940736,0.00023277313,0.0003997594,0.00034016478,0.0007243925,0.00037937515,0.00027070334,0.0127289025],"category_scores_gemma":[0.0008943118,0.000108960696,0.0002398867,0.0006076165,0.00016493312,0.0005021383,0.00056924234,0.00036073892,0.004322915],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005710028,0.00018743855,0.0047762156,0.0008089909,0.0001059039,0.0009331589,0.00039465007,0.021026324,0.10294266,0.02246711,0.0449794,0.8008072],"study_design_scores_gemma":[0.00019950457,0.0007037354,0.030264609,0.00020873154,0.00019825637,0.002458865,0.00061963894,0.4861778,0.19975746,0.027615767,0.25168934,0.00010626624],"about_ca_topic_score_codex":0.0034675437,"about_ca_topic_score_gemma":0.008126289,"teacher_disagreement_score":0.0127289025,"about_ca_system_score_codex":0.00053246674,"about_ca_system_score_gemma":0.0005779529,"threshold_uncertainty_score":0.042582393},"labels":[],"label_agreement":null},{"id":"W4390854775","doi":"","title":"HTR-United : un écosystème pour une approche mutualisée de la transcription automatique des écritures manuscrites","year":2022,"lang":"en","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Political science; Geography","score_opus":0.015734283435584195,"score_gpt":0.2547602189678753,"score_spread":0.23902593553229112,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4390854775","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0024813015,0.00047080583,0.957807,0.0003889683,0.00021155766,0.00017136105,0.0003041466,0.03467546,0.003489362],"genre_scores_gemma":[0.036235515,0.0009398093,0.9337831,0.00043513134,0.00022755645,0.00039885752,0.0045779045,0.009086512,0.014315705],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9922253,0.002574386,0.0007414019,0.0014421317,0.00264828,0.000368566],"domain_scores_gemma":[0.9935648,0.0018177287,0.00026268393,0.0027833525,0.0012215337,0.00034994457],"candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.009105597,0.001678205,0.0018213175,0.003374434,0.0013409518,0.005714465,0.004035842,0.0034995554,0.018139178],"category_scores_gemma":[0.016226215,0.0016318036,0.0024466026,0.0024569563,0.0012623204,0.009689351,0.007642098,0.003335404,0.012187374],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014467465,0.00054382265,0.0019427984,0.0010929498,0.0005095383,0.0015553953,0.0016543374,0.01441841,0.035812706,0.057630148,0.04696903,0.8364241],"study_design_scores_gemma":[0.0003837559,0.0006170175,0.0018978453,0.0006701051,0.000370181,0.0025794553,0.0007464396,0.35618517,0.04019017,0.059927452,0.5361179,0.00031454925],"about_ca_topic_score_codex":0.0030141196,"about_ca_topic_score_gemma":0.003311184,"teacher_disagreement_score":0.9942855,"about_ca_system_score_codex":0.00083946675,"about_ca_system_score_gemma":0.002637351,"threshold_uncertainty_score":0.06068164},"labels":[],"label_agreement":null},{"id":"W4390863515","doi":"10.21203/rs.3.rs-3853113/v1","title":"On Block g-Circulant Matrices with Discrete Cosine and Sine Transforms for Transformer-Based Translation Machine","year":2024,"lang":"en","type":"preprint","venue":"Research Square","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"Department of Science and Technology, Ministry of Science and Technology, India","keywords":"Circulant matrix; Transformer; Discrete cosine transform; Computation; Sine; Computer science; Algorithm; Leverage (statistics); Wavelet; Arithmetic; Artificial intelligence; Mathematics; Engineering; Voltage; Electrical engineering","score_opus":0.03704471863958705,"score_gpt":0.3732077080062965,"score_spread":0.33616298936670946,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4390863515","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05229522,0.0008223419,0.9384633,0.00030970213,0.00015248253,0.000098305914,0.00023154818,0.004355606,0.0032714575],"genre_scores_gemma":[0.6145041,0.00078717736,0.3749645,0.0003040826,0.0001494006,0.00016023821,0.0014338958,0.00037931875,0.0073173605],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9996674,0.00010145153,0.000021570448,0.0000899809,0.00007978323,0.000039866307],"domain_scores_gemma":[0.9995441,0.00020293928,0.000042682685,0.000077031764,0.000107123175,0.00002609909],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00055718824,0.0009672389,0.00060956896,0.000725241,0.00031733417,0.0006737608,0.00082784647,0.00065697083,0.0038103452],"category_scores_gemma":[0.0021638174,0.000283724,0.0006058314,0.0010325226,0.00041315463,0.0011730761,0.0006087553,0.001008738,0.0021974538],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003401774,0.00012447544,0.00083221105,0.00013793242,0.00006977308,0.00019493798,0.00007563755,0.23491363,0.020134397,0.015276291,0.0077040605,0.7201965],"study_design_scores_gemma":[0.000011133606,0.000041489075,0.00010627192,0.00000595977,0.000008498011,0.000038117967,0.000010763208,0.9905785,0.0041172314,0.0039713015,0.001105664,0.0000049948653],"about_ca_topic_score_codex":0.0075571304,"about_ca_topic_score_gemma":0.008622911,"teacher_disagreement_score":0.0075571304,"about_ca_system_score_codex":0.00055495626,"about_ca_system_score_gemma":0.00091243547,"threshold_uncertainty_score":0.015026271},"labels":[],"label_agreement":null},{"id":"W4390898156","doi":"","title":"For a common European framework for evaluating AI- based translation technologies","year":2023,"lang":"en","type":"book-chapter","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Translation (biology); Computer science; Artificial intelligence; Biology","score_opus":0.044239737332295595,"score_gpt":0.30087596907273106,"score_spread":0.2566362317404355,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4390898156","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025083564,0.010938749,0.58480275,0.016682833,0.0031275018,0.005891238,0.012392921,0.00524921,0.33583125],"genre_scores_gemma":[0.17053564,0.0033316447,0.7556155,0.0035654875,0.00051070016,0.0059115915,0.023445707,0.0031442728,0.033939485],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.85844475,0.08386044,0.017412944,0.0063639786,0.030084815,0.0038330043],"domain_scores_gemma":[0.8657932,0.039621964,0.0075355023,0.027152443,0.055442378,0.004454511],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.12370891,0.0027991056,0.002960738,0.017895214,0.004490853,0.025346909,0.006461356,0.008289677,0.032699548],"category_scores_gemma":[0.13336426,0.00080824766,0.003828537,0.016703669,0.004717243,0.023243863,0.012664835,0.005169288,0.012062544],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001296923,0.001090778,0.004434717,0.0024381503,0.0006797273,0.00029733786,0.0018852896,0.011835006,0.0075910715,0.51056874,0.07108238,0.3867999],"study_design_scores_gemma":[0.00076288497,0.0026545252,0.01646647,0.010689766,0.0012736558,0.0007847895,0.0064981068,0.03887071,0.022859586,0.34017912,0.55839473,0.00056559866],"about_ca_topic_score_codex":0.011988517,"about_ca_topic_score_gemma":0.009529941,"teacher_disagreement_score":0.12370891,"about_ca_system_score_codex":0.010031393,"about_ca_system_score_gemma":0.013940142,"threshold_uncertainty_score":0.6542431},"labels":[],"label_agreement":null},{"id":"W4391017630","doi":"10.1162/coli_a_00510","title":"Context-aware Transliteration of Romanized South Asian Languages","year":2023,"lang":"en","type":"article","venue":"Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Institute for Catastrophic Loss Reduction","keywords":"Transliteration; Computer science; Romanization; Natural language processing; Context (archaeology); Artificial intelligence; Sentence; Word (group theory); Linguistics; History","score_opus":0.01638905617639773,"score_gpt":0.29555217385449595,"score_spread":0.2791631176780982,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391017630","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7515755,0.004817926,0.1586324,0.0013041827,0.0015821299,0.0005128802,0.01262435,0.052764576,0.016186018],"genre_scores_gemma":[0.85275126,0.0008536799,0.108624965,0.00037336024,0.00020311454,0.00022335847,0.027539866,0.0015277937,0.007902643],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9985709,0.00049016613,0.00014057387,0.00057207246,0.00011351388,0.0001127341],"domain_scores_gemma":[0.99709725,0.0011616655,0.00019828016,0.000694044,0.0007234067,0.00012532703],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014278732,0.001698411,0.0011612235,0.0010408731,0.0009955268,0.0012615751,0.0011229011,0.00071162835,0.0056648157],"category_scores_gemma":[0.0056190076,0.00038991374,0.0011655283,0.00088627136,0.0004469261,0.0019470396,0.0021127567,0.0016467762,0.006483205],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017993454,0.00036606137,0.01171306,0.0022092555,0.00059496757,0.001829821,0.0021872912,0.054054983,0.06950298,0.0019801315,0.044878058,0.808884],"study_design_scores_gemma":[0.00032474077,0.00076223054,0.015092881,0.00037225106,0.0007037601,0.0016888913,0.0033201973,0.6984207,0.19458543,0.007834885,0.07663178,0.00026234981],"about_ca_topic_score_codex":0.007757365,"about_ca_topic_score_gemma":0.013982844,"teacher_disagreement_score":0.007757365,"about_ca_system_score_codex":0.00080582325,"about_ca_system_score_gemma":0.001420765,"threshold_uncertainty_score":0.0189507},"labels":[],"label_agreement":null},{"id":"W4391128740","doi":"10.1109/ictc58733.2023.10393804","title":"Preliminary study for Conversational Korean-Vietnam Neural Machine Translation","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"Electronics and Telecommunications Research Institute","keywords":"Computer science; Subtitle; Machine translation; Vietnamese; Natural language processing; Artificial intelligence; Translation (biology); Speech recognition; Artificial neural network; Linguistics","score_opus":0.03120244419632835,"score_gpt":0.3072259806282179,"score_spread":0.27602353643188954,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391128740","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.77362436,0.0029509752,0.1711231,0.0022944747,0.00040550434,0.0010387007,0.0011844614,0.0021419504,0.045236528],"genre_scores_gemma":[0.8870668,0.0007558086,0.10138775,0.00035117773,0.00010095753,0.00033619342,0.0018430876,0.00020451029,0.007953829],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9973441,0.0016361555,0.00016535966,0.00038526388,0.00032935853,0.000139817],"domain_scores_gemma":[0.9933555,0.003197106,0.0001928347,0.00070010655,0.0022503312,0.00030427094],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0042143133,0.00042483478,0.00043720752,0.00039733222,0.0009363314,0.0010613677,0.0006653516,0.00047207976,0.00761412],"category_scores_gemma":[0.010990985,0.0002465234,0.00035963807,0.0006561786,0.0003553109,0.0027584487,0.0007113375,0.00086312904,0.0018322521],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0028553328,0.0032190098,0.030847719,0.0028489244,0.00035572774,0.0027885013,0.011730959,0.022235561,0.18310668,0.02595392,0.019488342,0.6945693],"study_design_scores_gemma":[0.0004211888,0.004769155,0.04702296,0.0004504777,0.00043432965,0.0041807513,0.011760916,0.5329663,0.23279059,0.0119225895,0.15304254,0.00023820827],"about_ca_topic_score_codex":0.007636896,"about_ca_topic_score_gemma":0.0077508744,"teacher_disagreement_score":0.007636896,"about_ca_system_score_codex":0.0007779225,"about_ca_system_score_gemma":0.00091385795,"threshold_uncertainty_score":0.025471747},"labels":[],"label_agreement":null},{"id":"W4391133721","doi":"10.14305/jn.29960819.2024.1.1.04","title":"LLMs and Linguistic Competency: An exploration of GPT-4 and a non-hegemonic English variety","year":2024,"lang":"en","type":"article","venue":"Newhouse Impact Journal","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University; University of Toronto","funders":"","keywords":"Variety (cybernetics); Hegemony; Linguistics; Sociology; Political science; Computer science; Artificial intelligence; Philosophy; Politics","score_opus":0.018569055074959172,"score_gpt":0.3161384297253882,"score_spread":0.29756937465042904,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391133721","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4778262,0.00044088202,0.023427095,0.0039765034,0.000024587935,0.00002904803,0.000090218986,0.000055667184,0.49412984],"genre_scores_gemma":[0.9955472,0.000039728126,0.0011187111,0.00004108543,0.0000061417127,0.0000072812604,0.000012357584,0.000011549136,0.0032158708],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.9992131,0.00043778305,0.000020150852,0.00008317615,0.00010434557,0.00014150048],"domain_scores_gemma":[0.99859816,0.00090264995,0.000103014856,0.000116613286,0.00013501022,0.00014462548],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00073926535,0.00021712872,0.00022655251,0.0011655709,0.0014275509,0.00442998,0.00064119266,0.00054855744,0.008793029],"category_scores_gemma":[0.0032110251,0.000105145045,0.00024211046,0.0012095735,0.007854719,0.004981575,0.002834292,0.0015997057,0.00033295262],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000015755026,0.000014858915,0.0012788984,0.000012267359,0.0000022122083,0.0001121421,0.0064428938,0.0002890794,0.00015354971,0.9848932,0.00019707753,0.0065880823],"study_design_scores_gemma":[0.000019475116,0.00004376122,0.0046079075,0.000046247867,0.000010719709,0.00027638482,0.026745357,0.004505443,0.00053674023,0.9426526,0.020540943,0.000014410757],"about_ca_topic_score_codex":0.0069855032,"about_ca_topic_score_gemma":0.0063744946,"teacher_disagreement_score":0.008793029,"about_ca_system_score_codex":0.0028305803,"about_ca_system_score_gemma":0.0015729774,"threshold_uncertainty_score":0.029415607},"labels":[],"label_agreement":null},{"id":"W4391229484","doi":"10.3389/fpsyg.2024.1270433","title":"Simon Fraser University Speech Error Database (SFUSED) Cantonese: Methods, design, and usage","year":2024,"lang":"en","type":"article","venue":"Frontiers in Psychology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Workflow; Database; Set (abstract data type); Perspective (graphical); Component (thermodynamics); Quality (philosophy); Natural language processing; Speech recognition; Artificial intelligence; Programming language","score_opus":0.028958291717479264,"score_gpt":0.3557433628741066,"score_spread":0.3267850711566273,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391229484","genre_codex":"dataset","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2880681,0.0014450176,0.21894576,0.002017516,0.00079695426,0.042978093,0.37096885,0.024115663,0.05066414],"genre_scores_gemma":[0.29002973,0.0011671209,0.26001987,0.0008296867,0.0003823242,0.09104866,0.32491454,0.0048502255,0.026757872],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9918324,0.0026342669,0.0013521147,0.0017182007,0.0020928828,0.00037005535],"domain_scores_gemma":[0.9789408,0.004779113,0.001212569,0.005298084,0.0079379985,0.0018314508],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009983302,0.0013565827,0.0008154172,0.003233222,0.0013880314,0.0020868399,0.0022758846,0.0008242245,0.03170744],"category_scores_gemma":[0.023359286,0.0005671995,0.00031644292,0.0025130517,0.0010453202,0.0014306791,0.0033238507,0.0009631105,0.013232108],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0042245598,0.0012484732,0.033168804,0.0032528855,0.00016229875,0.0017099688,0.011014969,0.0040051835,0.04136509,0.0074111945,0.28063297,0.61180365],"study_design_scores_gemma":[0.002058323,0.0021776443,0.17686534,0.0014405565,0.00025789603,0.0017348726,0.0083683245,0.016638327,0.041459233,0.007694538,0.7405404,0.00076453807],"about_ca_topic_score_codex":0.020670168,"about_ca_topic_score_gemma":0.023506885,"teacher_disagreement_score":0.03170744,"about_ca_system_score_codex":0.00129805,"about_ca_system_score_gemma":0.0040442995,"threshold_uncertainty_score":0.10607195},"labels":[],"label_agreement":null},{"id":"W4391312685","doi":"10.33137/rr.v41i4.32462","title":"Stringer, Gary A., gen. ed. DigitalDonne: The Online Variorum, vol. 6","year":2019,"lang":"en","type":"article","venue":"Renaissance and Reformation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Stringer; History; Engineering; Structural engineering","score_opus":0.0072935438185245645,"score_gpt":0.232679718798406,"score_spread":0.22538617497988142,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391312685","genre_codex":"review","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00046206,0.84346086,0.0033228318,0.036915552,0.014673044,0.000018908197,0.00054629025,0.00026411834,0.10033634],"genre_scores_gemma":[0.009096569,0.5431241,0.004440848,0.009485574,0.010975644,0.00006452002,0.0007674495,0.00044771764,0.42159754],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9991222,0.00023066084,0.00009101032,0.00012155667,0.00038961455,0.000044979784],"domain_scores_gemma":[0.9985,0.0007107051,0.000107703796,0.00013793692,0.0004236784,0.0001200169],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016073503,0.0011078154,0.0012715487,0.004814002,0.0017822116,0.0045295237,0.0013225406,0.0024614565,0.09782373],"category_scores_gemma":[0.004805915,0.0007023136,0.0005891599,0.005778186,0.0024307868,0.009283868,0.0015296552,0.0025802634,0.06675643],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000040124145,0.000011919832,0.00014964021,0.00032845238,0.0000068983672,0.000041134736,0.00017740429,0.0000789127,0.00013439514,0.013807672,0.8072416,0.17798184],"study_design_scores_gemma":[0.0000024305562,0.0000062906897,0.00020086723,0.00029480777,0.00000443658,0.00010769433,0.00011073103,0.00004271847,0.00010722588,0.0048490376,0.9942694,0.0000042853267],"about_ca_topic_score_codex":0.005161768,"about_ca_topic_score_gemma":0.017322589,"teacher_disagreement_score":0.09782373,"about_ca_system_score_codex":0.0015295459,"about_ca_system_score_gemma":0.0019735217,"threshold_uncertainty_score":0.32725298},"labels":[],"label_agreement":null},{"id":"W4391421638","doi":"10.4108/eai.18-12-2023.2348140","title":"Neural Machine Translation for Mooré, a Low-Resource Language","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"International Development Research Centre; Styrelsen för Internationellt Utvecklingssamarbete","keywords":"Machine translation; Computer science; Translation (biology); Artificial intelligence; Natural language processing; Resource (disambiguation); Computer network; Chemistry","score_opus":0.011455373022265094,"score_gpt":0.2862024387224442,"score_spread":0.27474706570017915,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391421638","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025191477,0.004886567,0.88205737,0.007170787,0.0017303398,0.00015574454,0.0018295379,0.0044020233,0.072576225],"genre_scores_gemma":[0.35579288,0.0033598952,0.57600695,0.002093555,0.0006714828,0.00027401006,0.003983484,0.0009921134,0.056825552],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99936134,0.00017087722,0.000046649475,0.00018553199,0.00019707068,0.000038491784],"domain_scores_gemma":[0.99914813,0.00033603923,0.00007955925,0.00019678246,0.0002120397,0.000027478376],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007178522,0.0003792186,0.00044209117,0.0006616804,0.00093970925,0.0015817424,0.00065791764,0.000751195,0.009919559],"category_scores_gemma":[0.0035968602,0.00022661427,0.0006332733,0.00081645214,0.001106135,0.00262641,0.0011787883,0.001213229,0.0038092085],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023726332,0.00008400676,0.0007326836,0.0006246592,0.00008949622,0.00056687795,0.0003178574,0.028252894,0.012042517,0.56942093,0.061682325,0.3259485],"study_design_scores_gemma":[0.000058928686,0.00013682462,0.0006376418,0.00018955382,0.00007081052,0.00066485925,0.00010763512,0.30357915,0.021730356,0.4637071,0.20903589,0.00008126096],"about_ca_topic_score_codex":0.002283128,"about_ca_topic_score_gemma":0.0054628695,"teacher_disagreement_score":0.009919559,"about_ca_system_score_codex":0.0008466843,"about_ca_system_score_gemma":0.0015627266,"threshold_uncertainty_score":0.03318423},"labels":[],"label_agreement":null},{"id":"W4391556103","doi":"10.4230/lipics.fscd.2024.15","title":"Adjoint Natural Deduction","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Fonds de recherche du Québec – Nature et technologies","keywords":"Natural deduction; Natural (archaeology); Mathematics; Computer science; Mathematical economics; Programming language; Geology","score_opus":0.04104605341892123,"score_gpt":0.19785011631357002,"score_spread":0.1568040628946488,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391556103","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0053244196,0.00028165357,0.9628761,0.00096582394,0.00035602722,0.0001016828,0.00026840612,0.0016454582,0.028180555],"genre_scores_gemma":[0.24097721,0.0005294668,0.7375325,0.0014532171,0.00042307444,0.0002535056,0.0006970191,0.0006079477,0.017526038],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99763143,0.00068897835,0.00015223466,0.00052586244,0.0008122129,0.00018916122],"domain_scores_gemma":[0.9971244,0.0015693647,0.000120302524,0.000566349,0.00051570305,0.00010389534],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035192722,0.0004998003,0.00063551945,0.0014175423,0.0012780153,0.0026871955,0.0016579761,0.0008678102,0.0073361318],"category_scores_gemma":[0.0064047193,0.00048067805,0.0013958405,0.0011216678,0.0035613752,0.0038625968,0.0031569193,0.0028334968,0.0016988316],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00002583207,0.00002836259,0.0001847173,0.00010124598,0.000015675878,0.00013102354,0.00025599104,0.0009780457,0.0013672921,0.9578943,0.0039595296,0.035058003],"study_design_scores_gemma":[0.000022447643,0.000021351592,0.00009194839,0.00003465673,0.000016907308,0.00020147141,0.000045324607,0.011061875,0.0032063392,0.93525094,0.050030086,0.000016585636],"about_ca_topic_score_codex":0.00078875164,"about_ca_topic_score_gemma":0.0010352412,"teacher_disagreement_score":0.0073361318,"about_ca_system_score_codex":0.0013422532,"about_ca_system_score_gemma":0.0015073013,"threshold_uncertainty_score":0.024541855},"labels":[],"label_agreement":null},{"id":"W4391558193","doi":"10.1109/icdmw60847.2023.00147","title":"Exhaustive Evaluation of Dynamic Link Prediction","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Samsung; Canadian Institute for Advanced Research","keywords":"Computer science; Link (geometry); Computer network","score_opus":0.02696918588738054,"score_gpt":0.3326945274972388,"score_spread":0.30572534160985826,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391558193","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.54242027,0.018435238,0.39074764,0.002143728,0.0009403416,0.00059572887,0.008131404,0.019237192,0.017348384],"genre_scores_gemma":[0.8918346,0.0013991448,0.08864001,0.0005350053,0.00016239616,0.00021869304,0.013508069,0.00045060983,0.0032514173],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9927932,0.0027518342,0.0005051476,0.0018698984,0.0016612824,0.00041871436],"domain_scores_gemma":[0.97726715,0.013989282,0.0011985579,0.003665835,0.0032869014,0.0005922612],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010629955,0.0024033287,0.0019430991,0.0040059686,0.0010102667,0.002487812,0.0028400982,0.0025926607,0.001554094],"category_scores_gemma":[0.032844484,0.00048805645,0.0010904475,0.0030496293,0.0010626869,0.0070978,0.0022293981,0.0018553848,0.0010009687],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00096941926,0.0008119755,0.03361598,0.0011763845,0.00091086666,0.00036376074,0.0002463931,0.54496807,0.0054698694,0.006887576,0.029901158,0.37467858],"study_design_scores_gemma":[0.00003010487,0.0002710366,0.0028820855,0.00006975628,0.00008819298,0.00013885612,0.00009201246,0.98523116,0.0038692136,0.0049070013,0.0023811574,0.00003933739],"about_ca_topic_score_codex":0.012943303,"about_ca_topic_score_gemma":0.017987864,"teacher_disagreement_score":0.012943303,"about_ca_system_score_codex":0.0014951292,"about_ca_system_score_gemma":0.002209546,"threshold_uncertainty_score":0.056217253},"labels":[],"label_agreement":null},{"id":"W4391623961","doi":"10.1002/9781119131304.ch5","title":"Transported Dialect Features","year":2023,"lang":"en","type":"other","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Linguistics; Geography; History; Philosophy","score_opus":0.009528272302731022,"score_gpt":0.26046890251419025,"score_spread":0.25094063021145924,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391623961","genre_codex":"empirical","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5719121,0.0012191165,0.03352151,0.0006179056,0.00037160088,0.0002976136,0.044992775,0.0028440293,0.34422338],"genre_scores_gemma":[0.87133485,0.0004629723,0.015740508,0.00021046592,0.0000997956,0.00012106528,0.02499105,0.0011480221,0.08589131],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9996655,0.000039807117,0.0000392955,0.00009867551,0.000100860656,0.000055907563],"domain_scores_gemma":[0.99943584,0.000095676354,0.00008487295,0.0001693064,0.00017263775,0.000041711413],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00018440264,0.00035852363,0.00020648744,0.0017422653,0.0008418979,0.0016408392,0.0005259595,0.00038799914,0.043731943],"category_scores_gemma":[0.0014261826,0.00014499022,0.0002117874,0.0022975414,0.00044366633,0.0014395466,0.0011292966,0.0005728045,0.011595897],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010125489,0.00019792204,0.10674141,0.001192726,0.00014935524,0.008532487,0.02614093,0.0014243167,0.078727536,0.12116287,0.091575205,0.56314266],"study_design_scores_gemma":[0.00003033164,0.00012228863,0.16332462,0.00016679316,0.00010707328,0.015170775,0.008277329,0.0023011018,0.016312512,0.018788861,0.77529955,0.000098841854],"about_ca_topic_score_codex":0.004588311,"about_ca_topic_score_gemma":0.00642848,"teacher_disagreement_score":0.043731943,"about_ca_system_score_codex":0.0004350533,"about_ca_system_score_gemma":0.00035902555,"threshold_uncertainty_score":0.14629799},"labels":[],"label_agreement":null},{"id":"W4391760761","doi":"10.1145/3632754.3633076","title":"CIRAL at FIRE 2023: Cross-Lingual Information Retrieval for African Languages","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada); University of Waterloo","funders":"Universitas Brawijaya","keywords":"Yoruba; Swahili; Computer science; Hausa; Relevance (law); Natural language processing; Artificial intelligence; Information retrieval; Languages of Africa; Task (project management); Linguistics; Political science","score_opus":0.014959603819412236,"score_gpt":0.3202101600306263,"score_spread":0.30525055621121405,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391760761","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.285072,0.023194827,0.1290501,0.011327782,0.0118325595,0.022256345,0.32674116,0.08559676,0.104928434],"genre_scores_gemma":[0.16752942,0.0024459425,0.20481233,0.0029414906,0.001262209,0.00921316,0.5489761,0.0067153247,0.056104027],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9897377,0.0041898238,0.00047496118,0.0011691779,0.00292372,0.0015045974],"domain_scores_gemma":[0.9877401,0.002866715,0.00034380754,0.0024247612,0.0047361916,0.0018884384],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.021060035,0.0024398016,0.0023754498,0.006655069,0.0044975406,0.0042250827,0.0031891267,0.0026600903,0.02507153],"category_scores_gemma":[0.018261803,0.00070314674,0.0017823176,0.0043155914,0.0011905398,0.0075927726,0.0066515175,0.003945026,0.016878208],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0027325496,0.0022250044,0.002981691,0.0017332127,0.00026219356,0.00039017337,0.0011215303,0.0013802203,0.027282527,0.0036657348,0.69961685,0.25660843],"study_design_scores_gemma":[0.002717111,0.003972124,0.039263908,0.00077462586,0.00040939995,0.001724257,0.003651319,0.026547648,0.06402668,0.0059762085,0.8502991,0.00063763576],"about_ca_topic_score_codex":0.023599315,"about_ca_topic_score_gemma":0.029950596,"teacher_disagreement_score":0.02507153,"about_ca_system_score_codex":0.0023877919,"about_ca_system_score_gemma":0.0042977734,"threshold_uncertainty_score":0.11137742},"labels":[],"label_agreement":null},{"id":"W4391836017","doi":"10.48550/arxiv.2402.09299","title":"Trained Without My Consent: Detecting Code Inclusion In Language Models Trained on Code","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Code (set theory); Inclusion (mineral); Computer science; Programming language; Natural language processing; Artificial intelligence; Linguistics; Psychology; Social psychology; Philosophy","score_opus":0.07116136631781075,"score_gpt":0.2415754576448641,"score_spread":0.17041409132705335,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391836017","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.30446488,0.0017889054,0.5850523,0.0071408423,0.0010062641,0.0014219657,0.031156272,0.05133669,0.016631862],"genre_scores_gemma":[0.7059057,0.00031091526,0.23096392,0.002495975,0.00019341876,0.0012105309,0.04891935,0.0015633331,0.008436941],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9916829,0.0031164086,0.0005995659,0.002254194,0.0019021125,0.0004447975],"domain_scores_gemma":[0.9647907,0.017298311,0.0022215866,0.010940665,0.0040210513,0.0007276755],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0102823395,0.00093233166,0.00073472055,0.0011932993,0.0007239977,0.0018355263,0.0017824338,0.001584606,0.0044642542],"category_scores_gemma":[0.066010535,0.0005271716,0.0009071087,0.0008064824,0.0010486395,0.0029833463,0.0029221699,0.0026308554,0.0045905807],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016955157,0.00064644177,0.09812086,0.00074865506,0.00033863744,0.0018020999,0.001942862,0.039640576,0.019891057,0.009809454,0.122488104,0.7028758],"study_design_scores_gemma":[0.0002261358,0.00040786492,0.019460766,0.00040576005,0.000123543,0.0009637347,0.00064178556,0.8560562,0.02825541,0.03532188,0.057985045,0.00015188022],"about_ca_topic_score_codex":0.006459158,"about_ca_topic_score_gemma":0.012426167,"teacher_disagreement_score":0.0102823395,"about_ca_system_score_codex":0.0009078096,"about_ca_system_score_gemma":0.0034754255,"threshold_uncertainty_score":0.054378867},"labels":[],"label_agreement":null},{"id":"W4391923628","doi":"10.1109/icotl59758.2023.10435050","title":"Influence of Contextual Information on Bengali-English Forward and Backward Transliteration Using Binary Coding","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Saskatchewan","funders":"","keywords":"Bengali; Transliteration; Computer science; Natural language processing; Coding (social sciences); Artificial intelligence; Binary number; Speech recognition; Arithmetic; Mathematics; Statistics","score_opus":0.01589029054297399,"score_gpt":0.27052236377491834,"score_spread":0.25463207323194437,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391923628","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.968452,0.0018761248,0.017049842,0.00024837255,0.00023352301,0.00007833926,0.0023384595,0.001946166,0.007777163],"genre_scores_gemma":[0.98437506,0.00030116955,0.008213136,0.000046940473,0.00004831865,0.00003848663,0.004991672,0.0001358002,0.0018493247],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99868494,0.0005189765,0.00012678465,0.00034096514,0.00016262157,0.00016574154],"domain_scores_gemma":[0.9960198,0.002272858,0.00032041926,0.00046173425,0.0007883256,0.0001368335],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011448157,0.00084203196,0.00049877475,0.001181794,0.0006812568,0.0011682815,0.000358409,0.0003709341,0.002148346],"category_scores_gemma":[0.008175756,0.00009514987,0.0004990349,0.0009968142,0.0005166122,0.00090085046,0.0009769626,0.00064031425,0.0018495728],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0047825472,0.00033611016,0.11342968,0.0013795347,0.00030804664,0.001954025,0.0026636838,0.023295864,0.09708614,0.0020353154,0.013956159,0.73877287],"study_design_scores_gemma":[0.00013238982,0.0014366956,0.31250086,0.0005108726,0.00091186253,0.0031996127,0.008273753,0.31303355,0.307658,0.0036842383,0.04833188,0.00032624532],"about_ca_topic_score_codex":0.0141886985,"about_ca_topic_score_gemma":0.017763535,"teacher_disagreement_score":0.0141886985,"about_ca_system_score_codex":0.00046245192,"about_ca_system_score_gemma":0.000746693,"threshold_uncertainty_score":0.02821225},"labels":[],"label_agreement":null},{"id":"W4391930053","doi":"10.1109/oncon60463.2023.10430889","title":"Bidirectional Multi-Stack RNNs with Attention for Machine Translation","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University; University of Toronto","funders":"","keywords":"Computer science; Machine translation; Stack (abstract data type); Translation (biology); Artificial intelligence; Programming language","score_opus":0.042370378424999126,"score_gpt":0.3144449073504316,"score_spread":0.2720745289254325,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391930053","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021274649,0.0010249317,0.9665645,0.00024388805,0.00017718901,0.00006202268,0.000193076,0.0064349896,0.004024779],"genre_scores_gemma":[0.5547247,0.00083333405,0.42839143,0.0005359402,0.00017156196,0.00018557986,0.00117646,0.00087990175,0.013101119],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996326,0.000091436385,0.000022002485,0.0001115446,0.00007894866,0.00006334701],"domain_scores_gemma":[0.999537,0.00018070798,0.0000319303,0.0000722855,0.00015405015,0.00002408863],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009171828,0.0010912088,0.0005885207,0.0005443248,0.00047180807,0.0006673078,0.0014047407,0.00095132703,0.003482706],"category_scores_gemma":[0.0019519494,0.00051883433,0.00077972823,0.0006593877,0.0003764957,0.0018325775,0.0011245818,0.0017618731,0.0019862338],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00030625885,0.00012579141,0.00076662336,0.00017784083,0.00012979472,0.00018591013,0.00014593783,0.35347718,0.03218137,0.016277686,0.0075380537,0.58868766],"study_design_scores_gemma":[0.000008501377,0.000038690192,0.00013601893,0.000009820943,0.000018671568,0.000027637929,0.0000119978195,0.98319525,0.008119413,0.0062046875,0.0022204332,0.0000089509185],"about_ca_topic_score_codex":0.013835085,"about_ca_topic_score_gemma":0.024928916,"teacher_disagreement_score":0.013835085,"about_ca_system_score_codex":0.0010250619,"about_ca_system_score_gemma":0.0011724712,"threshold_uncertainty_score":0.027509093},"labels":[],"label_agreement":null},{"id":"W4391930129","doi":"10.55492/dhasa.v5i1.5027","title":"A Minimal Computing Approach to Southern African Language Resources","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Computer science; Programming language; Linguistics; Philosophy","score_opus":0.011456818297668436,"score_gpt":0.2622364919352548,"score_spread":0.25077967363758635,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391930129","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.057456207,0.0015005238,0.7433767,0.011638567,0.00019666611,0.0002767896,0.0004927127,0.0007856719,0.18427613],"genre_scores_gemma":[0.5879207,0.0006828928,0.38655755,0.0005379149,0.0001007179,0.00059655996,0.00040925844,0.00032105666,0.022873303],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9983944,0.00091481564,0.00006942041,0.00027022537,0.00021255814,0.00013854961],"domain_scores_gemma":[0.99849105,0.0008441113,0.00008563144,0.0003448654,0.00012404278,0.00011031036],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015659059,0.0005959285,0.00035840878,0.0021756287,0.0040067676,0.006755594,0.0020494403,0.0007950963,0.011398516],"category_scores_gemma":[0.0049952143,0.0004742255,0.00060671987,0.0025214613,0.010747592,0.006708885,0.0066399304,0.0015654885,0.0012363072],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000015315916,0.000010645308,0.00032629442,0.00007151807,0.000005196071,0.00010807045,0.0056553795,0.0021944272,0.0005260563,0.9743235,0.0012662961,0.0154973],"study_design_scores_gemma":[0.000013604445,0.000016351729,0.0003693805,0.0001132697,0.000009436833,0.00013197673,0.004793153,0.010030231,0.00109112,0.8439162,0.1394989,0.000016319418],"about_ca_topic_score_codex":0.0092878295,"about_ca_topic_score_gemma":0.011335867,"teacher_disagreement_score":0.011398516,"about_ca_system_score_codex":0.004397799,"about_ca_system_score_gemma":0.0030709344,"threshold_uncertainty_score":0.038131833},"labels":[],"label_agreement":null},{"id":"W4391955031","doi":"10.26615/issn.2683-0078.2023_029","title":"Lost in Innu-Aimun Translation - Re-dening Neural Machine Translation for Indigenous Interpreters and Translators Needs","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Cegep de Sept Iles; Université du Québec à Montréal","funders":"","keywords":"Interpreter; Indigenous; Machine translation; Linguistics; Documentation; Computer science; Process (computing); Citizen journalism; History; Natural language processing; Geography; Artificial intelligence; World Wide Web; Philosophy","score_opus":0.022498260889269366,"score_gpt":0.28691831260957434,"score_spread":0.26442005172030497,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391955031","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2198624,0.0067503946,0.5844126,0.027239839,0.0015922554,0.0005291315,0.0028727888,0.009598803,0.14714181],"genre_scores_gemma":[0.70862573,0.0020296231,0.23858294,0.0019729186,0.0002706373,0.00026533817,0.003569499,0.0018642879,0.04281897],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99482346,0.0024844648,0.00029092398,0.0007848553,0.0010706033,0.0005457301],"domain_scores_gemma":[0.98981243,0.0031670607,0.00033911562,0.0021791565,0.0041444707,0.00035772723],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0071797874,0.0010802812,0.0012055013,0.001648217,0.003548086,0.006505157,0.0017230215,0.0020801043,0.012103424],"category_scores_gemma":[0.021152163,0.000606388,0.0007531202,0.0027566666,0.0026656648,0.005772083,0.0033559005,0.0026527694,0.004092926],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008871864,0.0003655106,0.014272866,0.0017494645,0.00026302878,0.0018381701,0.015442431,0.030965982,0.029517824,0.20960405,0.048620578,0.6464729],"study_design_scores_gemma":[0.00017281502,0.000453975,0.014322543,0.0010141493,0.00034721868,0.002650682,0.018696135,0.33413216,0.052890364,0.14359675,0.43130898,0.0004142313],"about_ca_topic_score_codex":0.11560852,"about_ca_topic_score_gemma":0.2001192,"teacher_disagreement_score":0.11560852,"about_ca_system_score_codex":0.005546631,"about_ca_system_score_gemma":0.011444377,"threshold_uncertainty_score":0.22987121},"labels":[],"label_agreement":null},{"id":"W4391956911","doi":"10.26615/issn.2683-0078.2023_011","title":"The Proper Place of Men and Machines - Updated","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Montfort Hospital; Université de Montréal","funders":"","keywords":"Machine translation; Computer science; Variety (cybernetics); Meaning (existential); Semantics (computer science); Artificial intelligence; Natural language processing; Point (geometry); Quality (philosophy); Translation (biology); Linguistics; Programming language; Psychology; Epistemology; Mathematics","score_opus":0.008671590485849576,"score_gpt":0.2611207039081637,"score_spread":0.25244911342231413,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391956911","genre_codex":"methods","genre_gemma":"commentary","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"commentary","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.030510774,0.022441197,0.5236464,0.062274024,0.011230138,0.00008354974,0.0006000206,0.002116563,0.34709734],"genre_scores_gemma":[0.7886859,0.0048938217,0.11405097,0.0070945723,0.00509706,0.00015406517,0.00042226436,0.0012494755,0.07835186],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99647,0.0013344182,0.00021276213,0.0012040124,0.000553327,0.00022551659],"domain_scores_gemma":[0.99383485,0.0030508698,0.00022475554,0.0019860065,0.0006713597,0.00023214739],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003460632,0.0005220984,0.0008787748,0.0013220872,0.0026410348,0.0068540014,0.0015882743,0.0020090186,0.010122649],"category_scores_gemma":[0.01570253,0.000633177,0.0005757465,0.00092158286,0.01674611,0.020512363,0.0039459146,0.005089358,0.0030858098],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000034942263,0.0000054611996,0.000127918,0.000035016517,0.000004620816,0.00003097322,0.00043815526,0.00022112746,0.0002124632,0.98072904,0.0029524683,0.015207843],"study_design_scores_gemma":[0.000009157661,0.000012156357,0.00017049794,0.00004120211,0.000008546779,0.000100353005,0.00014281196,0.0023629053,0.0006690031,0.920345,0.076122455,0.000015794934],"about_ca_topic_score_codex":0.0022133642,"about_ca_topic_score_gemma":0.0019023727,"teacher_disagreement_score":0.010122649,"about_ca_system_score_codex":0.0022303297,"about_ca_system_score_gemma":0.0012290318,"threshold_uncertainty_score":0.033863664},"labels":[],"label_agreement":null},{"id":"W4391964032","doi":"10.26615/issn.2683-0078.2023_027","title":"Workbench for Post-editing of Translations from English and Hindi to Dravidian Languages","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"BC Research (Canada)","funders":"","keywords":"Computer science; Workbench; Machine translation; Natural language processing; Machine translation software usability; Hindi; Malayalam; Artificial intelligence; Computer-assisted translation; Rule-based machine translation; Tamil; Process (computing); Language translation; Example-based machine translation; Translation (biology); Programming language; Linguistics","score_opus":0.012591871910145182,"score_gpt":0.2911685354294152,"score_spread":0.27857666351927,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391964032","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01777915,0.00053248135,0.6949144,0.0005394352,0.0012992871,0.0021388328,0.014555084,0.23068619,0.03755505],"genre_scores_gemma":[0.065639704,0.0006346743,0.76694417,0.0004891777,0.00051285775,0.002093871,0.033105697,0.034498353,0.09608156],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9983752,0.00043592008,0.00024005154,0.00040866085,0.00042737328,0.00011280449],"domain_scores_gemma":[0.9916603,0.0038708549,0.00045121263,0.00211104,0.0015890192,0.0003175053],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029677476,0.0020319924,0.00085573597,0.0028042907,0.0011355771,0.002196259,0.0018690749,0.000773253,0.121789426],"category_scores_gemma":[0.008232944,0.0006935545,0.00092544395,0.0016783279,0.00062532525,0.0023805753,0.0020431294,0.001299953,0.048838533],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001016715,0.00035297452,0.0017566355,0.0016634803,0.00012140455,0.0024394665,0.002873979,0.0011083605,0.07297621,0.009868597,0.18068865,0.7251335],"study_design_scores_gemma":[0.00026347852,0.0008633519,0.003969495,0.0002565218,0.00009325561,0.003017849,0.0012360696,0.011630136,0.1351861,0.0055568204,0.83772093,0.00020596621],"about_ca_topic_score_codex":0.000717685,"about_ca_topic_score_gemma":0.0012810822,"teacher_disagreement_score":0.121789426,"about_ca_system_score_codex":0.00042482407,"about_ca_system_score_gemma":0.00096367626,"threshold_uncertainty_score":0.40742624},"labels":[],"label_agreement":null},{"id":"W4391987585","doi":"10.1093/jos/ffae001","title":"THE HELL with questions","year":2022,"lang":"en","type":"article","venue":"Journal of Semantics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Social Sciences and Humanities Research Council of Canada; National Outstanding Youth Science Fund Project of National Natural Science Foundation of China; Universität Konstanz","keywords":"Doxastic logic; Computer science; Semantics (computer science); Class (philosophy); Domain (mathematical analysis); Linguistics; Cognitive dissonance; Semantic property; Expression (computer science); Epistemology; Philosophy; Natural language processing; Artificial intelligence; Psychology; Mathematics; Programming language; Social psychology","score_opus":0.006740377960851957,"score_gpt":0.24271406674964655,"score_spread":0.2359736887887946,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391987585","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06209521,0.009657736,0.40505022,0.049239773,0.0019782865,0.00022187506,0.00069878995,0.00049557,0.47056258],"genre_scores_gemma":[0.93929136,0.0025235612,0.027891004,0.0058399923,0.00087383523,0.00016882058,0.0003371529,0.00023109876,0.022843119],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99475265,0.0034129785,0.00023803502,0.0006777175,0.0006452989,0.00027335627],"domain_scores_gemma":[0.99427986,0.0037232493,0.0003477301,0.0007400174,0.00072368677,0.00018540758],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006044522,0.0006143908,0.00066208415,0.0018629798,0.0022644901,0.004865406,0.0013331955,0.002764814,0.010166174],"category_scores_gemma":[0.0124197425,0.0003538204,0.0008806772,0.0014866769,0.014358892,0.017426131,0.0040592486,0.0040931534,0.0011210222],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000008859722,0.000003537687,0.000078918936,0.000034035445,0.000004769485,0.000035110083,0.0011542205,0.00009412619,0.00012427194,0.9950578,0.0008670765,0.002537219],"study_design_scores_gemma":[0.0000056219264,0.000007293522,0.00014326394,0.000061424376,0.0000065643794,0.000102629245,0.0011461711,0.00056667684,0.00031667252,0.95645034,0.041182876,0.00001046375],"about_ca_topic_score_codex":0.0013589192,"about_ca_topic_score_gemma":0.00077784766,"teacher_disagreement_score":0.010166174,"about_ca_system_score_codex":0.002110477,"about_ca_system_score_gemma":0.0010473626,"threshold_uncertainty_score":0.03400922},"labels":[],"label_agreement":null},{"id":"W4392305758","doi":"10.5220/0012567700003654","title":"Mitigating Outlier Activations in Low-Precision Fine-Tuning of Language Models","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; Huawei Technologies (Canada)","funders":"","keywords":"Computer science; Outlier; Language model; Artificial intelligence; Natural language processing","score_opus":0.016910568669447772,"score_gpt":0.2983856961191145,"score_spread":0.2814751274496667,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4392305758","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17587835,0.001567429,0.8090194,0.000782742,0.00041086343,0.00008637174,0.000318513,0.010078735,0.0018575669],"genre_scores_gemma":[0.8680943,0.00029112375,0.12770711,0.00042915271,0.0001248228,0.00007645795,0.0006243236,0.000653766,0.001998879],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9982085,0.00044546186,0.00011076886,0.0005360788,0.00047523086,0.00022391422],"domain_scores_gemma":[0.9951121,0.0027519965,0.0003603969,0.0009051049,0.0006815756,0.00018875318],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024490685,0.0012945037,0.0014447145,0.00084386003,0.0006855338,0.0018018924,0.0018806313,0.0015135485,0.0017311745],"category_scores_gemma":[0.017716,0.00065631967,0.0005074029,0.00083623297,0.0006325186,0.001910685,0.0019834228,0.0033158723,0.0009953376],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017861594,0.0006254482,0.012752543,0.0003488711,0.0003845877,0.0005439712,0.00043761267,0.21133523,0.08311164,0.004191001,0.009079949,0.675403],"study_design_scores_gemma":[0.000040296993,0.00014817106,0.0027794626,0.000023226377,0.0000595585,0.00016835854,0.00007506581,0.9691144,0.019533074,0.0065399185,0.0014976976,0.000020740601],"about_ca_topic_score_codex":0.0061605345,"about_ca_topic_score_gemma":0.012790741,"teacher_disagreement_score":0.0061605345,"about_ca_system_score_codex":0.000614034,"about_ca_system_score_gemma":0.0018235154,"threshold_uncertainty_score":0.01295203},"labels":[],"label_agreement":null},{"id":"W4392367355","doi":"10.1162/coli_a_00512","title":"A Novel Alignment-based Approach for PARSEVAL Measuress","year":2023,"lang":"en","type":"article","venue":"Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Parsing; Sentence; Natural language processing; Artificial intelligence; Lexical analysis; Parseval's theorem; Machine translation; Word (group theory); Linguistics","score_opus":0.055118956961616714,"score_gpt":0.3166822825463177,"score_spread":0.26156332558470097,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4392367355","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0030356308,0.000096251264,0.99321496,0.00005673958,0.000041871426,0.00007835103,0.00011804564,0.0025008132,0.0008572582],"genre_scores_gemma":[0.074866384,0.000058048874,0.9228679,0.00005621077,0.0000668062,0.00030440217,0.0003279857,0.00095134997,0.0005009735],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.97773224,0.008279352,0.0024966216,0.0030246873,0.007912829,0.0005543766],"domain_scores_gemma":[0.96939653,0.0110075725,0.0023374509,0.006130014,0.010565776,0.0005626985],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012469096,0.0018520508,0.0016846322,0.008922711,0.0014896762,0.006395544,0.0033589208,0.0017103471,0.00350377],"category_scores_gemma":[0.06463953,0.0009641323,0.0014111028,0.005906713,0.001942964,0.0057391836,0.0036089986,0.003224266,0.002086804],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00034309123,0.00023314399,0.0062838458,0.00042605426,0.000276684,0.00015342321,0.00083402864,0.037479386,0.027265357,0.17481688,0.008157616,0.7437305],"study_design_scores_gemma":[0.00004753155,0.00031385053,0.0035115806,0.00013935649,0.00011289986,0.00045894235,0.0001682673,0.78897065,0.03853088,0.14590667,0.021652197,0.00018720087],"about_ca_topic_score_codex":0.0016745699,"about_ca_topic_score_gemma":0.0015531679,"teacher_disagreement_score":0.012469096,"about_ca_system_score_codex":0.0018963577,"about_ca_system_score_gemma":0.0019688031,"threshold_uncertainty_score":0.06594366},"labels":[],"label_agreement":null},{"id":"W4392531254","doi":"","title":"A statistical, comparative analysis of the automotive industry segmentation discourse : implications for english > french translation","year":2017,"lang":"fr","type":"preprint","venue":"theses.fr (ABES)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Automotive industry; Translation (biology); Segmentation; Linguistics; Natural language processing; Discourse analysis; Computer science; Artificial intelligence; Engineering; Philosophy","score_opus":0.0993142122277714,"score_gpt":0.4071182537649624,"score_spread":0.307804041537191,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4392531254","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.86310375,0.004177535,0.014696842,0.0031047973,0.00021245542,0.0002497254,0.004004277,0.00015696646,0.110293664],"genre_scores_gemma":[0.98771274,0.0007946608,0.0045992667,0.0002357024,0.00012221235,0.00029009342,0.0016653081,0.000096125215,0.0044837897],"study_design_codex":"qualitative","study_design_gemma":"observational","domain_scores_codex":[0.99059886,0.0063045,0.0005916576,0.0007380551,0.00138432,0.00038260155],"domain_scores_gemma":[0.9565752,0.032850724,0.0027041358,0.001233765,0.0062907524,0.00034536206],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0064418335,0.00030190047,0.000292228,0.007447033,0.002182679,0.0044874963,0.0003852799,0.00055456575,0.010268456],"category_scores_gemma":[0.033333782,0.00016807627,0.00034534687,0.010473498,0.002221658,0.002240656,0.0018633088,0.00066368404,0.0015555249],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011178425,0.00019844809,0.12423594,0.0019569714,0.00019938826,0.0011970383,0.5435535,0.0007480484,0.014110042,0.084667,0.014985229,0.21303058],"study_design_scores_gemma":[0.00004414001,0.0004315069,0.5321085,0.00077267777,0.0001448697,0.00093458535,0.2853774,0.0033096205,0.0052797557,0.0064176996,0.16507013,0.000109139175],"about_ca_topic_score_codex":0.023198528,"about_ca_topic_score_gemma":0.025649052,"teacher_disagreement_score":0.023198528,"about_ca_system_score_codex":0.0044701835,"about_ca_system_score_gemma":0.0021090244,"threshold_uncertainty_score":0.04612696},"labels":[],"label_agreement":null},{"id":"W4392616692","doi":"","title":"CoSPLADE : Adaptation d'un Modèle Neuronal Basé sur des Représentations Parcimonieuses pour la Recherche d'Information Conversationnelle","year":2023,"lang":"fr","type":"article","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; Computer Research Institute of Montréal","funders":"Agence Nationale de la Recherche","keywords":"Computer science; Humanities; Philosophy","score_opus":0.11882449268525512,"score_gpt":0.3032067272744688,"score_spread":0.18438223458921366,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4392616692","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03457805,0.0007262831,0.9570875,0.00043437388,0.0005295748,0.00009028002,0.00027386632,0.0029949355,0.003285155],"genre_scores_gemma":[0.6874379,0.0009018297,0.29109025,0.00039972388,0.00018541305,0.00039541224,0.00081917533,0.00058928167,0.018181043],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99978,0.00005040429,0.000010728921,0.00008784287,0.00004444864,0.000026618607],"domain_scores_gemma":[0.999509,0.00025158003,0.00001704315,0.00007406286,0.00011379782,0.000034526092],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008204065,0.00095615775,0.00085559854,0.00026327898,0.0003819258,0.0011012255,0.0015349538,0.0017294498,0.0044084997],"category_scores_gemma":[0.002072761,0.00052997126,0.00069906435,0.00043435258,0.00053145696,0.0011215594,0.001091945,0.0015406726,0.0012304016],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00055846956,0.00016810103,0.0009562706,0.0002224591,0.00021052406,0.00025038657,0.00019135191,0.6737564,0.048820354,0.010874753,0.0060903677,0.25790057],"study_design_scores_gemma":[0.000016469823,0.000026953212,0.00016270904,0.000005871667,0.000011175424,0.000021204194,0.000006840155,0.99340725,0.0038582012,0.0016227863,0.00085273775,0.000007744128],"about_ca_topic_score_codex":0.014813538,"about_ca_topic_score_gemma":0.014088596,"teacher_disagreement_score":0.014813538,"about_ca_system_score_codex":0.00074047316,"about_ca_system_score_gemma":0.0010856242,"threshold_uncertainty_score":0.029454648},"labels":[],"label_agreement":null},{"id":"W4392669863","doi":"10.18653/v1/2023.findings-ijcnlp.14","title":"Multilingual Non-Autoregressive Machine Translation without Knowledge Distillation","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Alliance de recherche numérique du Canada; Alberta Innovates; Natural Sciences and Engineering Research Council of Canada; National Natural Science Foundation of China; Canadian Institute for Advanced Research","keywords":"Autoregressive model; Machine translation; Computer science; Transformer; Translation (biology); Distillation; Generalization; Artificial intelligence; Machine learning; Natural language processing; Speech recognition; Econometrics; Mathematics; Engineering","score_opus":0.02289192339714689,"score_gpt":0.32905212195646044,"score_spread":0.30616019855931353,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4392669863","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01334018,0.0006899738,0.96262515,0.00028498223,0.00020377254,0.000085334446,0.000816099,0.01590825,0.0060462356],"genre_scores_gemma":[0.47829363,0.0009161234,0.5010146,0.0005787318,0.00023080694,0.0001733225,0.005310788,0.0011094639,0.0123725515],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99914277,0.0002212053,0.00006415647,0.0003412972,0.00016000426,0.0000705868],"domain_scores_gemma":[0.99850595,0.0006001559,0.00011244438,0.00047837355,0.00025391576,0.000049141774],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009873257,0.0010838747,0.00087651063,0.00076768664,0.00059137295,0.0011733717,0.0014375,0.0007578322,0.005896293],"category_scores_gemma":[0.0032341657,0.00038583385,0.0009774361,0.0012890839,0.00055174535,0.0023208272,0.0017973405,0.001525583,0.0049614045],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00049228885,0.00033549566,0.0010812785,0.00060553406,0.00025256778,0.0004790082,0.00026071837,0.04747932,0.03801013,0.029021265,0.020450624,0.86153185],"study_design_scores_gemma":[0.0000666962,0.00022680525,0.0005933226,0.000043939366,0.00013815952,0.0004306251,0.00008696958,0.89370173,0.041318472,0.041457575,0.021877412,0.000058203757],"about_ca_topic_score_codex":0.0034401417,"about_ca_topic_score_gemma":0.008147682,"teacher_disagreement_score":0.005896293,"about_ca_system_score_codex":0.0004530773,"about_ca_system_score_gemma":0.0013400236,"threshold_uncertainty_score":0.019725084},"labels":[],"label_agreement":null},{"id":"W4392733308","doi":"10.1017/9781009210409.010","title":"The Statistics of Bilingualism","year":2024,"lang":"en","type":"book-chapter","venue":"Cambridge University Press eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Neuroscience of multilingualism; Psychology; Statistics; Linguistics; Mathematics; Philosophy; Neuroscience","score_opus":0.015518296900396933,"score_gpt":0.22643029298066855,"score_spread":0.2109119960802716,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4392733308","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025238048,0.11432292,0.04866176,0.034424677,0.004689889,0.00006759064,0.032751232,0.0013286876,0.7385152],"genre_scores_gemma":[0.4848776,0.11693871,0.039174136,0.0058579864,0.0057886974,0.00035161458,0.027139382,0.0018060239,0.31806588],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9982974,0.00047412512,0.00013646275,0.00030151053,0.00068512606,0.00010537686],"domain_scores_gemma":[0.9933781,0.0039202156,0.00045097683,0.0005691429,0.0014824823,0.00019914421],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020260592,0.00045890725,0.00059413584,0.008067962,0.0009006689,0.0033404771,0.00041939618,0.0005023731,0.013431566],"category_scores_gemma":[0.014082198,0.0003737632,0.00029866264,0.012863359,0.0020917132,0.0045980955,0.0011243038,0.0015145548,0.0045228205],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000055833654,0.000014658609,0.0063075577,0.00039534448,0.000040021114,0.00006723901,0.0015064862,0.0010506402,0.00028168884,0.44165874,0.27161032,0.2770115],"study_design_scores_gemma":[0.000008550935,0.000031606236,0.023472017,0.0005045663,0.000017753991,0.00038893006,0.00086857623,0.001654611,0.00042386912,0.2435482,0.72902185,0.000059414942],"about_ca_topic_score_codex":0.013200125,"about_ca_topic_score_gemma":0.014310122,"teacher_disagreement_score":0.013431566,"about_ca_system_score_codex":0.002827852,"about_ca_system_score_gemma":0.0019907078,"threshold_uncertainty_score":0.04493308},"labels":[],"label_agreement":null},{"id":"W4392913259","doi":"10.1111/cogs.13424","title":"Recursive Numeral Systems Optimize the Trade‐off Between Lexicon Size and Average Morphosyntactic Complexity","year":2022,"lang":"en","type":"article","venue":"Cognitive Science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Seventh Framework Programme; Azrieli Foundation; Tel Aviv University; École Normale Supérieure","keywords":"Lexicon; Numeral system; Computer science; Semantics (computer science); Natural language processing; Linguistics; Artificial intelligence; Domain (mathematical analysis); Simple (philosophy); Variation (astronomy); Principle of compositionality; Mathematics; Programming language","score_opus":0.03155150480088137,"score_gpt":0.28848358171073607,"score_spread":0.2569320769098547,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4392913259","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.37383848,0.00046442478,0.5792815,0.0009951231,0.000050819875,0.00017604057,0.00016806579,0.0040684645,0.040957056],"genre_scores_gemma":[0.8631329,0.00020355484,0.13041614,0.00017056665,0.000038209124,0.00014288093,0.0001736106,0.0006402502,0.0050820597],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9989796,0.00029246273,0.00009302814,0.0003148464,0.00018957352,0.0001304245],"domain_scores_gemma":[0.99523455,0.002830211,0.00050431717,0.00079060276,0.00045310287,0.00018716964],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013119527,0.00065162714,0.00094108866,0.0006762172,0.0008212165,0.0043857605,0.0013307483,0.0012745338,0.0055619203],"category_scores_gemma":[0.009646463,0.0008580634,0.0006633319,0.000509895,0.0021251827,0.00667448,0.002652184,0.0011020635,0.0019807064],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00039411016,0.00018229561,0.009083899,0.0005754923,0.00018504754,0.0005243881,0.00339815,0.16672899,0.24269505,0.31477204,0.0027624397,0.25869814],"study_design_scores_gemma":[0.00010245403,0.00025668537,0.0061851805,0.000043594024,0.00013999932,0.0006382389,0.0007772131,0.61525536,0.04705133,0.3183302,0.011082038,0.0001376855],"about_ca_topic_score_codex":0.0014712568,"about_ca_topic_score_gemma":0.003223725,"teacher_disagreement_score":0.0055619203,"about_ca_system_score_codex":0.0013181474,"about_ca_system_score_gemma":0.0008594569,"threshold_uncertainty_score":0.018606484},"labels":[],"label_agreement":null},{"id":"W4392914379","doi":"10.1111/cogs.13429","title":"Evaluating the Relative Importance of Wordhood Cues Using Statistical Learning","year":2024,"lang":"en","type":"article","venue":"Cognitive Science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Economic and Social Research Council; Social Sciences and Humanities Research Council of Canada","keywords":"Computer science; Mechanism (biology); Cognition; Cognitive psychology; Artificial intelligence; Trustworthiness; Perception; Immutability; Psychology; Statistical learning; Cognitive science; Natural language processing; Social psychology; Epistemology","score_opus":0.07019916433521724,"score_gpt":0.42921618287867247,"score_spread":0.35901701854345525,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4392914379","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8819816,0.00019914274,0.11014436,0.00039994653,0.000034620476,0.00012039011,0.00016126918,0.00046577526,0.0064929747],"genre_scores_gemma":[0.97501105,0.00008334573,0.024335068,0.00005331609,0.000014003414,0.00003360632,0.00015970325,0.00006298398,0.00024700697],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9964187,0.0015437726,0.00032211444,0.0005743055,0.0009580741,0.00018296066],"domain_scores_gemma":[0.9085281,0.0757015,0.0066669877,0.0036305923,0.003628984,0.0018438074],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0075685256,0.00052321306,0.0006864177,0.0025800746,0.00042301044,0.0033042324,0.0011493586,0.00094929896,0.0034722295],"category_scores_gemma":[0.096445605,0.00030992247,0.00047476104,0.0014545607,0.001769506,0.007667601,0.0024891756,0.0015209566,0.0006192957],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0024050192,0.0007082848,0.35398737,0.00082374504,0.000411579,0.000408364,0.0023994271,0.04365779,0.084842615,0.019308213,0.0011544428,0.4898932],"study_design_scores_gemma":[0.00018545616,0.003148653,0.20766932,0.00022213602,0.00030743188,0.00077682897,0.0036887128,0.541519,0.0650669,0.17399827,0.0030447994,0.00037244233],"about_ca_topic_score_codex":0.0011093117,"about_ca_topic_score_gemma":0.001599449,"teacher_disagreement_score":0.0075685256,"about_ca_system_score_codex":0.00061365665,"about_ca_system_score_gemma":0.0008982342,"threshold_uncertainty_score":0.040026665},"labels":[],"label_agreement":null},{"id":"W4392966345","doi":"10.21203/rs.3.rs-4109962/v1","title":"Detection of Banglish Slang in Social Media Comments Using a Hybrid Bidirectional Long Short-Term Memory (Bi-LSTM) Model","year":2024,"lang":"en","type":"preprint","venue":"Research Square","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Moncton","funders":"","keywords":"Slang; Term (time); Long short term memory; Computer science; Social media; Artificial intelligence; Speech recognition; Linguistics; World Wide Web; Philosophy; Physics; Recurrent neural network; Artificial neural network","score_opus":0.09671129433719242,"score_gpt":0.4066024667928772,"score_spread":0.30989117245568476,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4392966345","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8502668,0.0009120511,0.13437952,0.0007898827,0.00036706656,0.00027477535,0.0032856034,0.005249205,0.004475182],"genre_scores_gemma":[0.95681113,0.00019671973,0.035992917,0.00014846254,0.00006957901,0.000118814096,0.00271946,0.000055890556,0.0038870345],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99966,0.00005484495,0.000030469706,0.0001012876,0.000084071966,0.00006938945],"domain_scores_gemma":[0.99915373,0.00027712187,0.000111969915,0.000050589362,0.00035444865,0.000052071493],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00045885387,0.0010206243,0.0004171149,0.0011381079,0.0003375444,0.0005416643,0.000766211,0.0007218042,0.0009521536],"category_scores_gemma":[0.0012857462,0.00016425358,0.0005168755,0.00062335917,0.00022422003,0.0007041687,0.00058228505,0.00074537704,0.0012282722],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017660911,0.0012463744,0.045409355,0.00078153674,0.00025002056,0.0019405866,0.0011631588,0.06327171,0.13350832,0.00091975974,0.017325282,0.7324177],"study_design_scores_gemma":[0.00001543927,0.00016682937,0.010421965,0.000032951997,0.000042205334,0.0001318022,0.00047667813,0.9695742,0.01674151,0.00037651337,0.0019889104,0.00003092479],"about_ca_topic_score_codex":0.008681746,"about_ca_topic_score_gemma":0.013638581,"teacher_disagreement_score":0.008681746,"about_ca_system_score_codex":0.00047647953,"about_ca_system_score_gemma":0.0005578426,"threshold_uncertainty_score":0.0172624},"labels":[],"label_agreement":null},{"id":"W4393023501","doi":"10.48550/arxiv.2403.10758","title":"Rules still work for Open Information Extraction","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; National Office for Philosophy and Social Sciences","keywords":"Work (physics); Extraction (chemistry); Computer science; Information extraction; Data science; Information retrieval; Engineering; Chromatography; Mechanical engineering","score_opus":0.061929978844047225,"score_gpt":0.23798086349675035,"score_spread":0.17605088465270313,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393023501","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0054043257,0.0013503896,0.95337915,0.0029338764,0.0005082019,0.0003540328,0.002181868,0.0067109917,0.027177177],"genre_scores_gemma":[0.11016796,0.001929254,0.85090876,0.002024629,0.00053890696,0.0006457112,0.009564851,0.0032146003,0.021005312],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99065524,0.0023867947,0.0012760096,0.002669075,0.0025578511,0.00045508114],"domain_scores_gemma":[0.97236,0.009020362,0.0009504817,0.013662119,0.0034113077,0.00059571874],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0074105687,0.0015267266,0.0014437492,0.004308328,0.0027030471,0.008871527,0.0035164927,0.0026026801,0.019803347],"category_scores_gemma":[0.036911946,0.0011541767,0.0034767073,0.0049079596,0.0034259504,0.021677155,0.0059309425,0.0049432996,0.017772622],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017209313,0.00019949354,0.003528442,0.00080967013,0.00017988302,0.00034906404,0.00088590436,0.012088131,0.0028578406,0.48264307,0.044175223,0.45211112],"study_design_scores_gemma":[0.000038063623,0.000041624975,0.00049275917,0.0003088856,0.00010801722,0.00039345765,0.00030073355,0.075383924,0.005791263,0.6806952,0.23638263,0.000063448235],"about_ca_topic_score_codex":0.003914184,"about_ca_topic_score_gemma":0.005439115,"teacher_disagreement_score":0.019803347,"about_ca_system_score_codex":0.0015869531,"about_ca_system_score_gemma":0.0043117865,"threshold_uncertainty_score":0.066248775},"labels":[],"label_agreement":null},{"id":"W4393034946","doi":"10.1109/iwsc60764.2023.00011","title":"Unveiling the Potential of Large Language Models in Generating Semantic and Cross-Language Clones","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Saskatchewan","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Linguistics; Philosophy","score_opus":0.011430762508152705,"score_gpt":0.3022172625010232,"score_spread":0.2907864999928705,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393034946","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15678264,0.0009160276,0.7984324,0.0011061036,0.00024566252,0.0006755794,0.0016403942,0.034938417,0.0052627777],"genre_scores_gemma":[0.4039406,0.0003832091,0.5803509,0.00053801225,0.000038064907,0.00068110955,0.004903605,0.0055835107,0.003581044],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9961281,0.0015467348,0.00029058408,0.0007125597,0.0011746455,0.00014732602],"domain_scores_gemma":[0.9791038,0.013928642,0.00090513134,0.0032191207,0.0025722776,0.00027099974],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004696103,0.0015721605,0.00071910693,0.0013377048,0.00073492964,0.0026676762,0.0023853271,0.0019047608,0.0027981636],"category_scores_gemma":[0.032469593,0.00088402786,0.001877395,0.0009070591,0.0012146698,0.004390825,0.0027142793,0.0028106784,0.0017554557],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010167839,0.00064254634,0.028577244,0.0021499423,0.0004220455,0.0018436097,0.006488253,0.34643444,0.05511806,0.031298906,0.020057287,0.5059509],"study_design_scores_gemma":[0.000096204465,0.00027644954,0.0012797553,0.00013852968,0.00013781364,0.0005478124,0.0003710559,0.93908364,0.02082254,0.014460167,0.022710282,0.00007572167],"about_ca_topic_score_codex":0.00591927,"about_ca_topic_score_gemma":0.011150384,"teacher_disagreement_score":0.00591927,"about_ca_system_score_codex":0.0013096755,"about_ca_system_score_gemma":0.002399705,"threshold_uncertainty_score":0.024835646},"labels":[],"label_agreement":null},{"id":"W4393090168","doi":"10.1007/978-3-031-56069-9_15","title":"KnowFIRES: A Knowledge-Graph Framework for Interpreting Retrieved Entities from Search","year":2024,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Graph; Knowledge graph; Information retrieval; Theoretical computer science","score_opus":0.019236535897772425,"score_gpt":0.30452563723685183,"score_spread":0.2852891013390794,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393090168","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0014501009,0.00039308646,0.96841675,0.00037949425,0.00006096437,0.00015967342,0.0054329336,0.018578744,0.005128259],"genre_scores_gemma":[0.042241532,0.0009677422,0.9307394,0.00037804455,0.000061652776,0.00022137766,0.013767909,0.0033427428,0.00827961],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99884135,0.00025269415,0.000120141776,0.00031367005,0.00039899937,0.00007322524],"domain_scores_gemma":[0.9980969,0.00093592494,0.00011656542,0.00051607896,0.00024175706,0.00009277979],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016753298,0.00139649,0.00114379,0.0056442097,0.0011945833,0.0053996914,0.0043659704,0.0019955775,0.01830457],"category_scores_gemma":[0.006388054,0.0011628906,0.00309505,0.0049557113,0.0016906607,0.010788637,0.0033102902,0.0023686464,0.0050596744],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00032197713,0.00018409955,0.0014637515,0.0016185759,0.00032214393,0.0006143345,0.001326693,0.02988415,0.0055762697,0.41332468,0.06815927,0.4772041],"study_design_scores_gemma":[0.000049974027,0.00005307637,0.0005520511,0.00037324338,0.00026028362,0.0003189728,0.00034043365,0.18426287,0.012725446,0.60594434,0.1950196,0.00009967604],"about_ca_topic_score_codex":0.023026912,"about_ca_topic_score_gemma":0.043025587,"teacher_disagreement_score":0.023026912,"about_ca_system_score_codex":0.0015555716,"about_ca_system_score_gemma":0.0020131099,"threshold_uncertainty_score":0.06123483},"labels":[],"label_agreement":null},{"id":"W4393145735","doi":"10.1609/aaai.v38i21.30449","title":"Towards a Transformer-Based Reverse Dictionary Model for Quality Estimation of Definitions (Student Abstract)","year":2024,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Transformer; Computer science; Quality (philosophy); Engineering; Electrical engineering; Philosophy","score_opus":0.1585445641583282,"score_gpt":0.38410635970827417,"score_spread":0.22556179554994596,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393145735","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008740345,0.00009153606,0.9900601,0.00016742983,0.000015329893,0.000029607761,0.0000858675,0.00031574958,0.00049396744],"genre_scores_gemma":[0.59949625,0.0003531833,0.39560977,0.00023638328,0.000071190836,0.00019917317,0.0007050771,0.0002698856,0.003059151],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99823475,0.0007251566,0.00012579504,0.0004849468,0.00027378104,0.00015555437],"domain_scores_gemma":[0.99206537,0.00491006,0.000688264,0.00078655005,0.0012826413,0.00026702613],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0038910746,0.0008070936,0.0011148497,0.001838296,0.0003134455,0.0024041517,0.002614179,0.0013140754,0.004092854],"category_scores_gemma":[0.019554157,0.00064037676,0.0014946217,0.0016356418,0.0014003376,0.0050472263,0.0020731594,0.0025429947,0.0011403038],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00073942303,0.0003418428,0.00946365,0.00030971743,0.00025272713,0.00021419473,0.00065141334,0.45275614,0.008498164,0.17972136,0.005029407,0.34202194],"study_design_scores_gemma":[0.000009334723,0.000033701333,0.0002529206,0.0000093505105,0.000011873638,0.000025072879,0.000016442145,0.9744666,0.00048697356,0.024310457,0.00036850793,0.000008776037],"about_ca_topic_score_codex":0.00805811,"about_ca_topic_score_gemma":0.0071091363,"teacher_disagreement_score":0.00805811,"about_ca_system_score_codex":0.0010731845,"about_ca_system_score_gemma":0.0010600974,"threshold_uncertainty_score":0.020578265},"labels":[],"label_agreement":null},{"id":"W4393161476","doi":"10.1609/aaai.v38i21.30563","title":"Enhancing Machine Translation Experiences with Multilingual Knowledge Graphs","year":2024,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Machine translation; Knowledge graph; Computer science; Translation (biology); Natural language processing; Artificial intelligence; Linguistics; Chemistry; Philosophy","score_opus":0.05207619920806413,"score_gpt":0.32616867748507294,"score_spread":0.27409247827700883,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393161476","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1933384,0.0014586196,0.7368977,0.0026330103,0.00054462237,0.00036207362,0.0012902656,0.011852261,0.05162299],"genre_scores_gemma":[0.70411015,0.00096950075,0.28045142,0.00064651103,0.00010347882,0.00013730933,0.0023029868,0.0017723967,0.009506165],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99741936,0.0016460526,0.00012207309,0.00036798322,0.00036200363,0.000082404134],"domain_scores_gemma":[0.993001,0.0050341957,0.00020997581,0.0009560895,0.00066855276,0.00013013759],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023955968,0.00091546815,0.00046365446,0.0012182433,0.0009287065,0.0023706993,0.001030165,0.0009966262,0.009599519],"category_scores_gemma":[0.015420424,0.00035175317,0.0005263235,0.0014147386,0.00091988663,0.005416188,0.0030208863,0.001356075,0.0023572843],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011608577,0.0009271194,0.0045310343,0.0015226202,0.00023341777,0.003500928,0.02642564,0.06277312,0.04983506,0.0673655,0.02858012,0.7531447],"study_design_scores_gemma":[0.00027295915,0.00075798377,0.0040080464,0.00049308484,0.0005854081,0.0023160318,0.011139861,0.37449533,0.08644749,0.17161301,0.34753472,0.00033607736],"about_ca_topic_score_codex":0.0027006895,"about_ca_topic_score_gemma":0.0057968134,"teacher_disagreement_score":0.009599519,"about_ca_system_score_codex":0.00094418315,"about_ca_system_score_gemma":0.00090133894,"threshold_uncertainty_score":0.032113552},"labels":[],"label_agreement":null},{"id":"W4393228430","doi":"10.1002/ail2.92","title":"Building Text and Speech Benchmark Datasets and Models for Low‐Resourced East African Languages: Experiences and Lessons","year":2024,"lang":"en","type":"article","venue":"Applied AI Letters","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"FP7 International Cooperation; Deutsche Gesellschaft für Internationale Zusammenarbeit; International Development Research Centre; Rockefeller Foundation","keywords":"Benchmark (surveying); Computer science; Linguistics; Natural language processing; Artificial intelligence; Geography; Cartography","score_opus":0.011785325025830535,"score_gpt":0.27784629943561534,"score_spread":0.2660609744097848,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393228430","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.83092153,0.002671502,0.08031941,0.013316676,0.001234269,0.0013487437,0.04547391,0.014669299,0.0100447675],"genre_scores_gemma":[0.7306937,0.0011438273,0.10942721,0.0009365462,0.00033577555,0.0011498915,0.15011564,0.0009773356,0.0052199275],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99486923,0.0031675135,0.00035046352,0.00068691274,0.00057482894,0.00035102782],"domain_scores_gemma":[0.9860364,0.008059096,0.0003518895,0.0021809542,0.002577965,0.0007936477],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010788419,0.0019343052,0.0008807623,0.0018118989,0.0018086795,0.002288539,0.002837817,0.0022616747,0.0025057863],"category_scores_gemma":[0.017452914,0.0005229386,0.0012189251,0.002286032,0.0012250766,0.0047544236,0.0026827827,0.002830225,0.0026127521],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0022957046,0.006112504,0.061459817,0.0019373472,0.0007319697,0.0021333117,0.0037428006,0.24773699,0.024768407,0.011326871,0.19552113,0.44223323],"study_design_scores_gemma":[0.0004596282,0.0010871952,0.027862072,0.0004631795,0.00018642249,0.00048444225,0.0063762143,0.83674085,0.035792436,0.012365902,0.077941656,0.00024007831],"about_ca_topic_score_codex":0.036315747,"about_ca_topic_score_gemma":0.037239052,"teacher_disagreement_score":0.036315747,"about_ca_system_score_codex":0.0022818672,"about_ca_system_score_gemma":0.0020095904,"threshold_uncertainty_score":0.0722087},"labels":[],"label_agreement":null},{"id":"W4393315183","doi":"10.1007/978-3-031-57327-9_13","title":"Natural2CTL: A Dataset for Natural Language Requirements and Their CTL Formal Equivalents","year":2024,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Equivalent; Programming language; CTL*; Natural language; Natural language processing; Artificial intelligence","score_opus":0.020645002648048975,"score_gpt":0.3020611085821761,"score_spread":0.28141610593412714,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393315183","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012804048,0.0005684423,0.0028186801,0.0003380149,0.000063928164,0.0001544334,0.97167724,0.005147117,0.0064281872],"genre_scores_gemma":[0.008505577,0.00020000138,0.003944814,0.00015663564,0.000014070263,0.0003102619,0.9850815,0.0003836631,0.0014035247],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9973206,0.0005062843,0.00037255147,0.00047588002,0.0010655985,0.00025910125],"domain_scores_gemma":[0.98694706,0.007245763,0.0009935248,0.002054289,0.0019084583,0.00085084885],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014183755,0.0019060713,0.0008153337,0.0069285845,0.00075943384,0.001529157,0.0019126412,0.0024806173,0.018066147],"category_scores_gemma":[0.014210711,0.0005988985,0.0015529187,0.0053640516,0.0005197473,0.0018927518,0.0020056732,0.0018717501,0.015369532],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00072238635,0.0003696557,0.01042725,0.00358981,0.00011194393,0.00076661125,0.00030326005,0.004860875,0.004307513,0.0068974094,0.93251586,0.035127457],"study_design_scores_gemma":[0.0005811965,0.00019107044,0.025614332,0.00060901843,0.00009686831,0.0011480858,0.0006433438,0.015472362,0.0064429976,0.0066788225,0.94238573,0.00013605664],"about_ca_topic_score_codex":0.020532511,"about_ca_topic_score_gemma":0.031447366,"teacher_disagreement_score":0.020532511,"about_ca_system_score_codex":0.0019014549,"about_ca_system_score_gemma":0.0026195652,"threshold_uncertainty_score":0.060437262},"labels":[],"label_agreement":null},{"id":"W4393375535","doi":"10.31234/osf.io/wqsjc","title":"English Verbs Semantic Norms Database: Concreteness, Embodiment, Imageability, Valence and Arousal Ratings for 2,900 Verbs","year":2024,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University; University of Calgary","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Concreteness; Valence (chemistry); Psychology; Arousal; Natural language processing; Linguistics; Emotional valence; Cognitive psychology; Computer science; Social psychology; Cognition; Philosophy","score_opus":0.014149440572197223,"score_gpt":0.2863475727428367,"score_spread":0.27219813217063943,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393375535","genre_codex":"empirical","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.78782886,0.0006832973,0.01743471,0.0001914682,0.000110220964,0.0012149755,0.16762727,0.0022824719,0.022626631],"genre_scores_gemma":[0.61934453,0.00031331836,0.021548353,0.00010094916,0.00009747438,0.003932558,0.34666932,0.00060199614,0.007391613],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99815065,0.00040983647,0.0005882787,0.00021235348,0.000571688,0.000067055356],"domain_scores_gemma":[0.9905613,0.0035334842,0.0012793264,0.0012125243,0.0029686727,0.0004447275],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018187455,0.00050423475,0.0005247456,0.002200866,0.00027230306,0.0008502286,0.00046768546,0.00054352405,0.0066469787],"category_scores_gemma":[0.010891992,0.0001855043,0.00034294097,0.001615539,0.00036947848,0.0009868193,0.000746663,0.00034679804,0.003457098],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0061151134,0.0016797537,0.30064416,0.0034398793,0.00036315375,0.0010393328,0.004799593,0.0027816428,0.058063883,0.007840755,0.15755878,0.455674],"study_design_scores_gemma":[0.00060598133,0.0011652652,0.82215214,0.0001710485,0.00010400371,0.0017263426,0.0017897278,0.0077813626,0.012883754,0.0045979447,0.14684081,0.00018159686],"about_ca_topic_score_codex":0.00132489,"about_ca_topic_score_gemma":0.002279388,"teacher_disagreement_score":0.0066469787,"about_ca_system_score_codex":0.00039234685,"about_ca_system_score_gemma":0.00036336292,"threshold_uncertainty_score":0.022236407},"labels":[],"label_agreement":null},{"id":"W4393447236","doi":"10.5281/zenodo.5148952","title":"Artifact for the paper \"Abstract Interpretation of LLVM with a Region-Based Memory Model\"","year":2021,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Artifact (error); Interpretation (philosophy); Computer science; Artificial intelligence; Natural language processing; Programming language","score_opus":0.028732748717250012,"score_gpt":0.2600215847485285,"score_spread":0.23128883603127848,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393447236","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00041345,0.0001229987,0.0017796615,0.00019567141,0.00029148423,0.000069810194,0.97574687,0.018726768,0.0026533247],"genre_scores_gemma":[0.0011175934,0.00004608967,0.002279935,0.00013167973,0.000026573194,0.00016215675,0.9932848,0.001593397,0.0013576947],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9980971,0.00035511114,0.00023446791,0.0005548207,0.0005122696,0.00024627298],"domain_scores_gemma":[0.99572766,0.001134792,0.00022829366,0.0019043888,0.000768224,0.00023674486],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015642898,0.0036802588,0.0014232391,0.0024227202,0.00081221375,0.002962912,0.0037674033,0.0017321866,0.13545586],"category_scores_gemma":[0.008019255,0.0009503251,0.0018912329,0.003118908,0.00062738836,0.0021063597,0.0021471228,0.0025706578,0.1520755],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006824933,0.000030330226,0.00018954332,0.00030386815,0.000020435002,0.000011027895,0.0000071341187,0.00031513866,0.00019962683,0.00034268206,0.995363,0.0031489716],"study_design_scores_gemma":[0.0004420827,0.000041347903,0.0019241786,0.00019795689,0.000034193363,0.00009808253,0.000033016702,0.0037531222,0.0023747806,0.003457172,0.9876089,0.00003525584],"about_ca_topic_score_codex":0.011564443,"about_ca_topic_score_gemma":0.024241898,"teacher_disagreement_score":0.13545586,"about_ca_system_score_codex":0.0016972318,"about_ca_system_score_gemma":0.0024258199,"threshold_uncertainty_score":0.45314503},"labels":[],"label_agreement":null},{"id":"W4393454403","doi":"10.5281/zenodo.10342992","title":"Mile 0 Sign - Object Capture","year":2021,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Sign (mathematics); Object (grammar); Mile; Computer science; Geography; Cartography; Artificial intelligence; Geodesy; Mathematics","score_opus":0.02222393011007517,"score_gpt":0.25622660580343104,"score_spread":0.23400267569335587,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393454403","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0006439885,0.0002400096,0.0010101718,0.00010885869,0.00007979749,0.00006110274,0.9858782,0.009550451,0.0024273898],"genre_scores_gemma":[0.0009903712,0.00007879555,0.0013218917,0.000050372673,0.0000045992406,0.000095204516,0.9961966,0.00033279558,0.0009294002],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9988726,0.000109363915,0.000109725595,0.00044382346,0.00027689466,0.00018772096],"domain_scores_gemma":[0.99913436,0.00014277064,0.000059735325,0.0003558117,0.00022685181,0.00008036388],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005958771,0.0048302743,0.0016647468,0.0035668346,0.0009938561,0.0029331788,0.004556393,0.0027791339,0.05412523],"category_scores_gemma":[0.003413827,0.0010593241,0.0024910471,0.0051869466,0.00058444013,0.002965099,0.0029304828,0.0030725,0.10715647],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000090290916,0.000045185923,0.0009140862,0.00058855006,0.000034164674,0.00005740641,0.000034485834,0.00093910575,0.00033635143,0.00089080754,0.98883295,0.007236553],"study_design_scores_gemma":[0.00013359953,0.000027420516,0.00336072,0.00027150067,0.000033545894,0.00017787493,0.00012926677,0.0053049102,0.0016281405,0.0024532967,0.9864137,0.000066154156],"about_ca_topic_score_codex":0.069809556,"about_ca_topic_score_gemma":0.13580506,"teacher_disagreement_score":0.069809556,"about_ca_system_score_codex":0.0025687465,"about_ca_system_score_gemma":0.0022643814,"threshold_uncertainty_score":0.18106693},"labels":[],"label_agreement":null},{"id":"W4393456850","doi":"10.5281/zenodo.8122619","title":"Dataset for Automatic Refactoring Candidate Identification Leveraging Effective Code Representation","year":2023,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Code refactoring; Computer science; Code (set theory); Identification (biology); Representation (politics); Programming language; Software; Biology","score_opus":0.04331755302203935,"score_gpt":0.3214804849110235,"score_spread":0.2781629318889841,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393456850","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005315326,0.00071639265,0.0014662062,0.00025708484,0.00009461592,0.00011118867,0.9853356,0.0043694545,0.002334266],"genre_scores_gemma":[0.001639058,0.00009310519,0.0019166777,0.000052847507,0.00000694775,0.00009854466,0.9954269,0.000095341304,0.00067059585],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99767524,0.0004040856,0.0002926279,0.000615287,0.00079262134,0.00022029891],"domain_scores_gemma":[0.9962399,0.0010504345,0.00031613454,0.00095694023,0.0011635831,0.00027295994],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017210748,0.0032848106,0.0011687985,0.005986541,0.0010844122,0.0016565188,0.003553169,0.0025981746,0.008828558],"category_scores_gemma":[0.0063774306,0.0005861349,0.0016685232,0.0055972324,0.00061464706,0.0012585021,0.0019353192,0.0018696398,0.019377297],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029270316,0.00030957418,0.003853178,0.0022151854,0.0001250726,0.00025638693,0.00009060173,0.0031292615,0.0020837435,0.0012615301,0.9647514,0.0216313],"study_design_scores_gemma":[0.00066185405,0.0001854768,0.016533686,0.0006124761,0.00018334409,0.00067939906,0.00027947145,0.014336755,0.007187658,0.0030338524,0.9561854,0.00012054872],"about_ca_topic_score_codex":0.025069583,"about_ca_topic_score_gemma":0.05453433,"teacher_disagreement_score":0.025069583,"about_ca_system_score_codex":0.0022197848,"about_ca_system_score_gemma":0.0026645698,"threshold_uncertainty_score":0.049847364},"labels":[],"label_agreement":null},{"id":"W4393485174","doi":"10.5281/zenodo.10342991","title":"Mile 0 Sign - Object Capture","year":2021,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Sign (mathematics); Object (grammar); Mile; Computer science; Computer graphics (images); Geography; Artificial intelligence; Mathematics; Geodesy","score_opus":0.02222393011007517,"score_gpt":0.25622660580343104,"score_spread":0.23400267569335587,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393485174","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0006439885,0.0002400096,0.0010101718,0.00010885869,0.00007979749,0.00006110274,0.9858782,0.009550451,0.0024273898],"genre_scores_gemma":[0.0009903712,0.00007879555,0.0013218917,0.000050372673,0.0000045992406,0.000095204516,0.9961966,0.00033279558,0.0009294002],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9988726,0.000109363915,0.000109725595,0.00044382346,0.00027689466,0.00018772096],"domain_scores_gemma":[0.99913436,0.00014277064,0.000059735325,0.0003558117,0.00022685181,0.00008036388],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005958771,0.0048302743,0.0016647468,0.0035668346,0.0009938561,0.0029331788,0.004556393,0.0027791339,0.05412523],"category_scores_gemma":[0.003413827,0.0010593241,0.0024910471,0.0051869466,0.00058444013,0.002965099,0.0029304828,0.0030725,0.10715647],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000090290916,0.000045185923,0.0009140862,0.00058855006,0.000034164674,0.00005740641,0.000034485834,0.00093910575,0.00033635143,0.00089080754,0.98883295,0.007236553],"study_design_scores_gemma":[0.00013359953,0.000027420516,0.00336072,0.00027150067,0.000033545894,0.00017787493,0.00012926677,0.0053049102,0.0016281405,0.0024532967,0.9864137,0.000066154156],"about_ca_topic_score_codex":0.069809556,"about_ca_topic_score_gemma":0.13580506,"teacher_disagreement_score":0.069809556,"about_ca_system_score_codex":0.0025687465,"about_ca_system_score_gemma":0.0022643814,"threshold_uncertainty_score":0.18106693},"labels":[],"label_agreement":null},{"id":"W4393495083","doi":"10.1007/s10115-023-02059-2","title":"Entity linking for English and other languages: a survey","year":2024,"lang":"en","type":"article","venue":"Knowledge and Information Systems","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ontario Tech University","funders":"Aston University; UK Research and Innovation","keywords":"Linguistics; Computer science; Philosophy","score_opus":0.01679273697061635,"score_gpt":0.2900945133550044,"score_spread":0.2733017763843881,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393495083","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.030093648,0.7776113,0.08401592,0.0033932591,0.00078881084,0.00020756082,0.0043458804,0.003480224,0.09606334],"genre_scores_gemma":[0.085247256,0.81759363,0.069088675,0.0024648944,0.00086616835,0.00019022294,0.009555182,0.0011315802,0.013862388],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99844956,0.0003279442,0.00025559962,0.00038330382,0.00046154446,0.00012207967],"domain_scores_gemma":[0.9922983,0.0057428577,0.00045529762,0.00044019968,0.0009460681,0.00011729945],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031218615,0.0006968872,0.0010591939,0.010379954,0.0006299846,0.0030510416,0.0011793167,0.00090629794,0.008285613],"category_scores_gemma":[0.008048123,0.0005689901,0.00075628294,0.016086292,0.00065167865,0.010054532,0.0017133566,0.0009230351,0.0050911666],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000039219718,0.000079195044,0.0033462024,0.004933392,0.000051450013,0.00017784201,0.0006207338,0.00055832835,0.0017398552,0.008344359,0.020435285,0.9596742],"study_design_scores_gemma":[0.0000073612746,0.000058713522,0.0067338343,0.0031647454,0.00010468414,0.0015289149,0.0011273861,0.0021018211,0.0049227104,0.006710496,0.9734865,0.000052847536],"about_ca_topic_score_codex":0.0012589481,"about_ca_topic_score_gemma":0.0011632185,"teacher_disagreement_score":0.010379954,"about_ca_system_score_codex":0.0004450954,"about_ca_system_score_gemma":0.00097063335,"threshold_uncertainty_score":0.027718127},"labels":[],"label_agreement":null},{"id":"W4393584161","doi":"10.5281/zenodo.7599666","title":"What Makes Sentences Semantically Related? A Textual Relatedness Dataset and Empirical Study","year":2021,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"National Research Council Canada; University of Toronto","funders":"","keywords":"Natural language processing; Computer science; Artificial intelligence; Linguistics; Information retrieval; Philosophy","score_opus":0.036499614587439416,"score_gpt":0.3093099756010593,"score_spread":0.2728103610136199,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393584161","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.56984574,0.0022956438,0.009452406,0.0027305286,0.0002626784,0.0016697348,0.39155585,0.0015356953,0.02065177],"genre_scores_gemma":[0.2706737,0.00040347243,0.018054983,0.0006048484,0.00018498063,0.001983913,0.7033794,0.00014369395,0.0045709787],"study_design_codex":"not_applicable","study_design_gemma":"observational","domain_scores_codex":[0.9971004,0.0013016581,0.00043136906,0.0004511359,0.0005811338,0.00013423875],"domain_scores_gemma":[0.98832536,0.006011854,0.0013836273,0.001380493,0.002002865,0.00089579605],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020281698,0.00055436074,0.00038970102,0.0035400065,0.0013074084,0.000933437,0.00081251527,0.0011066704,0.005850604],"category_scores_gemma":[0.016028764,0.00017563115,0.0006510178,0.003361647,0.00056638476,0.0019280757,0.001709068,0.0010640667,0.004189051],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021592702,0.002939398,0.1267164,0.005426712,0.0004452357,0.0012848948,0.007264659,0.0026182295,0.012847997,0.007899148,0.68928957,0.14110853],"study_design_scores_gemma":[0.0010672199,0.0015305221,0.59773403,0.0010108408,0.00044912376,0.0030668487,0.014464045,0.026441867,0.009896577,0.009396808,0.33458528,0.00035683933],"about_ca_topic_score_codex":0.0041792523,"about_ca_topic_score_gemma":0.008336642,"teacher_disagreement_score":0.005850604,"about_ca_system_score_codex":0.0008126709,"about_ca_system_score_gemma":0.00069472176,"threshold_uncertainty_score":0.019572258},"labels":[],"label_agreement":null},{"id":"W4393587959","doi":"10.5281/zenodo.6390355","title":"Geographic Diversity in Public Code Contributions — Replication Package","year":2022,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Replication (statistics); Diversity (politics); Code (set theory); R package; Geography; Computer science; Programming language; Sociology; Mathematics; Anthropology; Statistics","score_opus":0.0399510739980967,"score_gpt":0.2771984470773367,"score_spread":0.23724737307924001,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393587959","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":"reproducibility","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":"reproducibility","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007885494,0.0003274927,0.12546682,0.0030378364,0.0015813218,0.0020913836,0.45043463,0.3443787,0.06479626],"genre_scores_gemma":[0.048709646,0.0004655599,0.2232774,0.0013597723,0.0006318289,0.0074576265,0.5101851,0.16775188,0.04016112],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99400914,0.0012065565,0.0007370518,0.0012283536,0.002408261,0.00041058924],"domain_scores_gemma":[0.959305,0.010357292,0.0021553617,0.016296575,0.010468038,0.00141781],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.00868771,0.0012932239,0.00096399506,0.0047678356,0.0012813427,0.0039723227,0.0029866674,0.0010602338,0.16573444],"category_scores_gemma":[0.068539254,0.001878439,0.0020728558,0.005234336,0.0008446412,0.0063703884,0.0060768193,0.0032860206,0.13558948],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002682776,0.00009725289,0.005912947,0.00052241393,0.000086357315,0.00009040231,0.00071299815,0.0011626129,0.00059903396,0.0050061066,0.9077157,0.07782596],"study_design_scores_gemma":[0.00027049228,0.00010262226,0.015592722,0.00052507885,0.000070859984,0.00026083365,0.0005171589,0.0058661215,0.002615766,0.015379963,0.95861876,0.00017949296],"about_ca_topic_score_codex":0.010309545,"about_ca_topic_score_gemma":0.0073298076,"teacher_disagreement_score":0.99131227,"about_ca_system_score_codex":0.0013640448,"about_ca_system_score_gemma":0.0049325805,"threshold_uncertainty_score":0.5544369},"labels":[],"label_agreement":null},{"id":"W4393594744","doi":"10.5281/zenodo.5139093","title":"TLMD: Tigrinya Language Modeling Dataset","year":2021,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence","score_opus":0.03338556646129205,"score_gpt":0.28261250703339086,"score_spread":0.24922694057209882,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393594744","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007673873,0.0004941049,0.001384992,0.00036077134,0.0001914685,0.00013893144,0.97935724,0.006151274,0.0042473488],"genre_scores_gemma":[0.004874124,0.000070454764,0.001659477,0.000085829786,0.000016983404,0.00015898985,0.9916727,0.00016890875,0.001292415],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99876,0.00028499676,0.0001288145,0.00037929567,0.00026888243,0.00017809198],"domain_scores_gemma":[0.9988071,0.00031039552,0.00007643656,0.0003295231,0.00033485892,0.00014172641],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008551523,0.0027880522,0.0013400288,0.0030892778,0.0016726681,0.0019079562,0.0031418635,0.002286107,0.02979463],"category_scores_gemma":[0.0035715252,0.0005851369,0.0017182267,0.003524423,0.0006096552,0.0022703758,0.002090011,0.002737378,0.055941973],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00026824092,0.00022468697,0.0035147408,0.00078445824,0.000104160026,0.00031845988,0.00015315496,0.00222109,0.0013889043,0.0009233472,0.97056735,0.019531345],"study_design_scores_gemma":[0.00056312454,0.00022803053,0.022687776,0.00046797667,0.00016572118,0.0008992777,0.0010630757,0.024242645,0.0045206062,0.0035841821,0.9413468,0.00023083844],"about_ca_topic_score_codex":0.048268043,"about_ca_topic_score_gemma":0.07358068,"teacher_disagreement_score":0.048268043,"about_ca_system_score_codex":0.002372034,"about_ca_system_score_gemma":0.0029495517,"threshold_uncertainty_score":0.09967297},"labels":[],"label_agreement":null},{"id":"W4393603794","doi":"10.5281/zenodo.7089050","title":"Tigrinya Analogy Test for evaluating Word Embeddings","year":2022,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Analogy; Word (group theory); Test (biology); Natural language processing; Computer science; Linguistics; Artificial intelligence; Philosophy; Biology","score_opus":0.043782621840588974,"score_gpt":0.3171485015133652,"score_spread":0.27336587967277626,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393603794","genre_codex":"empirical","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.70808065,0.0030992834,0.08250789,0.0013762178,0.0030098816,0.0026541906,0.11035707,0.035780687,0.05313407],"genre_scores_gemma":[0.62174344,0.0002992336,0.08492557,0.00094363175,0.00031568858,0.0038016664,0.2734066,0.0032294649,0.01133472],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9878022,0.004873587,0.0019586526,0.0021274441,0.0025354489,0.00070263725],"domain_scores_gemma":[0.975744,0.016115181,0.00086122274,0.0025145286,0.0037994925,0.00096556544],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0061414735,0.0031943105,0.0011344651,0.0047674263,0.0011403533,0.0023084362,0.0017551181,0.0028359783,0.019966068],"category_scores_gemma":[0.0411626,0.00040700004,0.0020713254,0.002735965,0.0009879312,0.002704859,0.0034469084,0.0021121146,0.017030511],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.005067872,0.0021106629,0.06577887,0.0031386209,0.0013901246,0.0024336874,0.001099181,0.022349022,0.022637729,0.004546304,0.47857395,0.390874],"study_design_scores_gemma":[0.003358137,0.0090839295,0.21374188,0.00092772127,0.0006432302,0.0069018872,0.0068728514,0.42151645,0.07860093,0.02107414,0.2365115,0.00076733093],"about_ca_topic_score_codex":0.0027924825,"about_ca_topic_score_gemma":0.004326567,"teacher_disagreement_score":0.019966068,"about_ca_system_score_codex":0.0007028376,"about_ca_system_score_gemma":0.0012407906,"threshold_uncertainty_score":0.066793144},"labels":[],"label_agreement":null},{"id":"W4393645132","doi":"10.5281/zenodo.3940705","title":"Multi-label Pathway Prediction based on Active Dataset Subsampling","year":2020,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Artificial intelligence; Machine learning; Data mining","score_opus":0.057678271878960276,"score_gpt":0.2859920711278265,"score_spread":0.22831379924886622,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393645132","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.052093916,0.0040098787,0.045089416,0.0011398884,0.0008293545,0.00087793363,0.87636787,0.014394816,0.0051968694],"genre_scores_gemma":[0.0418761,0.0005195776,0.040837515,0.0006031023,0.000077631645,0.00082284806,0.9122469,0.000640502,0.002375877],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99765044,0.00051406695,0.00014332772,0.0010494279,0.00043197555,0.00021083427],"domain_scores_gemma":[0.99634546,0.0017090107,0.00016064526,0.0010552023,0.00051043753,0.00021908781],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036178262,0.0024465048,0.0018482299,0.0031199802,0.0012341776,0.0020793176,0.002782081,0.002659248,0.010736073],"category_scores_gemma":[0.009554281,0.00058977486,0.003781653,0.0023670956,0.00087167,0.0015056704,0.0019139479,0.0023697938,0.007412747],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004022165,0.0014330611,0.03595658,0.005877458,0.002284088,0.0011276768,0.00023364645,0.029654546,0.019824252,0.008619478,0.7323351,0.1586319],"study_design_scores_gemma":[0.00486616,0.0016273794,0.052392665,0.0012809508,0.0032151917,0.0032282015,0.00057844166,0.27112284,0.050491586,0.072843745,0.5378244,0.00052854494],"about_ca_topic_score_codex":0.008418516,"about_ca_topic_score_gemma":0.014849935,"teacher_disagreement_score":0.010736073,"about_ca_system_score_codex":0.0010369612,"about_ca_system_score_gemma":0.0023880783,"threshold_uncertainty_score":0.035915732},"labels":[],"label_agreement":null},{"id":"W4393689030","doi":"10.5281/zenodo.3236068","title":"RUEG Corpus","year":2024,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Natural language processing","score_opus":0.02355233542740364,"score_gpt":0.26617775393678733,"score_spread":0.24262541850938368,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393689030","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0052509927,0.0012811746,0.0052801305,0.0011508297,0.00067886757,0.0006135722,0.8267364,0.0050891596,0.15391895],"genre_scores_gemma":[0.01355207,0.0007294719,0.010083392,0.0004471285,0.00020190001,0.0015606639,0.9270549,0.0040896065,0.04228095],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9970124,0.00079967716,0.00040887826,0.0008732783,0.0006809184,0.00022492114],"domain_scores_gemma":[0.99555993,0.0015152103,0.0002282378,0.0013508754,0.0012178621,0.00012785934],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028930192,0.0018139535,0.0016534595,0.009201081,0.0022889082,0.0048521855,0.00404712,0.0020929987,0.28496438],"category_scores_gemma":[0.008712118,0.0011084635,0.0008812455,0.010422105,0.0010936252,0.0048740734,0.004858346,0.0022413095,0.17281604],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035482348,0.000077105746,0.0011022103,0.0017009209,0.00007500242,0.00040122462,0.0012407277,0.00046574927,0.0014504292,0.024313617,0.90120506,0.06761302],"study_design_scores_gemma":[0.000056061286,0.000014110019,0.0014818633,0.00018894089,0.000020245043,0.00016911971,0.0003493702,0.00022539457,0.0005484304,0.0029641222,0.99396,0.000022234659],"about_ca_topic_score_codex":0.016969569,"about_ca_topic_score_gemma":0.02038453,"teacher_disagreement_score":0.28496438,"about_ca_system_score_codex":0.0022167324,"about_ca_system_score_gemma":0.0034803452,"threshold_uncertainty_score":0.95330083},"labels":[],"label_agreement":null},{"id":"W4393696029","doi":"10.5281/zenodo.2536217","title":"Dataset of discussion threads from Meneame","year":2019,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Information retrieval","score_opus":0.03412997641335556,"score_gpt":0.2837003058316073,"score_spread":0.24957032941825177,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393696029","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0052957865,0.0009716255,0.00047078996,0.0005784479,0.00020545883,0.00014667147,0.9805433,0.0009503745,0.010837488],"genre_scores_gemma":[0.0072610048,0.00017971269,0.0014459001,0.00014884475,0.000083294086,0.00050384033,0.9842131,0.0001179059,0.0060462803],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9982748,0.00060123444,0.00018640107,0.00036669648,0.00041107947,0.00015989103],"domain_scores_gemma":[0.9956026,0.0013923551,0.00053770194,0.00072645827,0.0010440855,0.0006968368],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011337564,0.001275924,0.0008112171,0.004516811,0.0016020043,0.0021675145,0.001708291,0.0018495438,0.041461192],"category_scores_gemma":[0.008278058,0.00038827612,0.0009271539,0.0049613197,0.00035028646,0.0017841349,0.0022436162,0.0013951787,0.04278691],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028885843,0.00012424534,0.00647044,0.00094995904,0.00006467407,0.00007676692,0.0005553139,0.00027408724,0.0003500177,0.0016955663,0.97560287,0.013547254],"study_design_scores_gemma":[0.00012348208,0.00003615203,0.018933099,0.00026888118,0.000024697028,0.00006154901,0.0005613775,0.0007310108,0.00047022055,0.0011842371,0.97757316,0.000032141772],"about_ca_topic_score_codex":0.02099059,"about_ca_topic_score_gemma":0.06157746,"teacher_disagreement_score":0.041461192,"about_ca_system_score_codex":0.0019431906,"about_ca_system_score_gemma":0.0020946804,"threshold_uncertainty_score":0.13870156},"labels":[],"label_agreement":null},{"id":"W4393712804","doi":"10.5281/zenodo.5988663","title":"UDPipe Models for Morphologically enhanced Universal Dependencies for Korean (morphUD)","year":2022,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Genealogy; Geography; History","score_opus":0.037035978394360396,"score_gpt":0.2605448126707389,"score_spread":0.22350883427637852,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393712804","genre_codex":"methods","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03168613,0.0029386822,0.5274992,0.0019736276,0.0018879861,0.0008947337,0.22700758,0.17644635,0.029665757],"genre_scores_gemma":[0.17057121,0.0017930564,0.23696193,0.0013024543,0.0001864927,0.002649988,0.50802636,0.021929193,0.056579303],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99949455,0.0001354382,0.00004813597,0.00017384921,0.00007256547,0.00007543204],"domain_scores_gemma":[0.99934965,0.00021999501,0.000024436498,0.00017123148,0.00018813419,0.000046609173],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011955821,0.0029112701,0.0008997095,0.0011956408,0.00070174993,0.0017700486,0.003150867,0.0014891131,0.060158618],"category_scores_gemma":[0.0042033275,0.0016013035,0.002348338,0.0011208903,0.00040422764,0.0031244282,0.002835559,0.0037097123,0.046791356],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00093525084,0.00034528758,0.0043330025,0.0012359282,0.00037105454,0.0005770222,0.00033197942,0.20931531,0.005701896,0.019147497,0.54859346,0.20911245],"study_design_scores_gemma":[0.00032055017,0.00018392924,0.0017908374,0.00020645394,0.00013312136,0.00044738883,0.00019455946,0.7316299,0.010125617,0.02691823,0.2278906,0.00015882206],"about_ca_topic_score_codex":0.014314784,"about_ca_topic_score_gemma":0.022446206,"teacher_disagreement_score":0.060158618,"about_ca_system_score_codex":0.0013306793,"about_ca_system_score_gemma":0.0019707032,"threshold_uncertainty_score":0.20125061},"labels":[],"label_agreement":null},{"id":"W4393732781","doi":"10.5281/zenodo.8122618","title":"Dataset for Automatic Refactoring Candidate Identification Leveraging Effective Code Representation","year":2023,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Code refactoring; Computer science; Code (set theory); Identification (biology); Representation (politics); Programming language; Software; Biology","score_opus":0.04331755302203935,"score_gpt":0.3214804849110235,"score_spread":0.2781629318889841,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393732781","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005315326,0.00071639265,0.0014662062,0.00025708484,0.00009461592,0.00011118867,0.9853356,0.0043694545,0.002334266],"genre_scores_gemma":[0.001639058,0.00009310519,0.0019166777,0.000052847507,0.00000694775,0.00009854466,0.9954269,0.000095341304,0.00067059585],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99767524,0.0004040856,0.0002926279,0.000615287,0.00079262134,0.00022029891],"domain_scores_gemma":[0.9962399,0.0010504345,0.00031613454,0.00095694023,0.0011635831,0.00027295994],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017210748,0.0032848106,0.0011687985,0.005986541,0.0010844122,0.0016565188,0.003553169,0.0025981746,0.008828558],"category_scores_gemma":[0.0063774306,0.0005861349,0.0016685232,0.0055972324,0.00061464706,0.0012585021,0.0019353192,0.0018696398,0.019377297],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029270316,0.00030957418,0.003853178,0.0022151854,0.0001250726,0.00025638693,0.00009060173,0.0031292615,0.0020837435,0.0012615301,0.9647514,0.0216313],"study_design_scores_gemma":[0.00066185405,0.0001854768,0.016533686,0.0006124761,0.00018334409,0.00067939906,0.00027947145,0.014336755,0.007187658,0.0030338524,0.9561854,0.00012054872],"about_ca_topic_score_codex":0.025069583,"about_ca_topic_score_gemma":0.05453433,"teacher_disagreement_score":0.025069583,"about_ca_system_score_codex":0.0022197848,"about_ca_system_score_gemma":0.0026645698,"threshold_uncertainty_score":0.049847364},"labels":[],"label_agreement":null},{"id":"W4393791072","doi":"10.5281/zenodo.3940706","title":"Multi-label Pathway Prediction based on Active Dataset Subsampling","year":2020,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Artificial intelligence; Pattern recognition (psychology)","score_opus":0.057678271878960276,"score_gpt":0.2859920711278265,"score_spread":0.22831379924886622,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393791072","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.052093916,0.0040098787,0.045089416,0.0011398884,0.0008293545,0.00087793363,0.87636787,0.014394816,0.0051968694],"genre_scores_gemma":[0.0418761,0.0005195776,0.040837515,0.0006031023,0.000077631645,0.00082284806,0.9122469,0.000640502,0.002375877],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99765044,0.00051406695,0.00014332772,0.0010494279,0.00043197555,0.00021083427],"domain_scores_gemma":[0.99634546,0.0017090107,0.00016064526,0.0010552023,0.00051043753,0.00021908781],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036178262,0.0024465048,0.0018482299,0.0031199802,0.0012341776,0.0020793176,0.002782081,0.002659248,0.010736073],"category_scores_gemma":[0.009554281,0.00058977486,0.003781653,0.0023670956,0.00087167,0.0015056704,0.0019139479,0.0023697938,0.007412747],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004022165,0.0014330611,0.03595658,0.005877458,0.002284088,0.0011276768,0.00023364645,0.029654546,0.019824252,0.008619478,0.7323351,0.1586319],"study_design_scores_gemma":[0.00486616,0.0016273794,0.052392665,0.0012809508,0.0032151917,0.0032282015,0.00057844166,0.27112284,0.050491586,0.072843745,0.5378244,0.00052854494],"about_ca_topic_score_codex":0.008418516,"about_ca_topic_score_gemma":0.014849935,"teacher_disagreement_score":0.010736073,"about_ca_system_score_codex":0.0010369612,"about_ca_system_score_gemma":0.0023880783,"threshold_uncertainty_score":0.035915732},"labels":[],"label_agreement":null},{"id":"W4393877878","doi":"10.5281/zenodo.6390354","title":"Geographic Diversity in Public Code Contributions — Replication Package","year":2022,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Replication (statistics); Diversity (politics); Code (set theory); R package; Geography; Computer science; Biology; Programming language; Sociology; Anthropology; Virology","score_opus":0.0399510739980967,"score_gpt":0.2771984470773367,"score_spread":0.23724737307924001,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393877878","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":"reproducibility","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":"reproducibility","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007885494,0.0003274927,0.12546682,0.0030378364,0.0015813218,0.0020913836,0.45043463,0.3443787,0.06479626],"genre_scores_gemma":[0.048709646,0.0004655599,0.2232774,0.0013597723,0.0006318289,0.0074576265,0.5101851,0.16775188,0.04016112],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99400914,0.0012065565,0.0007370518,0.0012283536,0.002408261,0.00041058924],"domain_scores_gemma":[0.959305,0.010357292,0.0021553617,0.016296575,0.010468038,0.00141781],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.00868771,0.0012932239,0.00096399506,0.0047678356,0.0012813427,0.0039723227,0.0029866674,0.0010602338,0.16573444],"category_scores_gemma":[0.068539254,0.001878439,0.0020728558,0.005234336,0.0008446412,0.0063703884,0.0060768193,0.0032860206,0.13558948],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002682776,0.00009725289,0.005912947,0.00052241393,0.000086357315,0.00009040231,0.00071299815,0.0011626129,0.00059903396,0.0050061066,0.9077157,0.07782596],"study_design_scores_gemma":[0.00027049228,0.00010262226,0.015592722,0.00052507885,0.000070859984,0.00026083365,0.0005171589,0.0058661215,0.002615766,0.015379963,0.95861876,0.00017949296],"about_ca_topic_score_codex":0.010309545,"about_ca_topic_score_gemma":0.0073298076,"teacher_disagreement_score":0.99131227,"about_ca_system_score_codex":0.0013640448,"about_ca_system_score_gemma":0.0049325805,"threshold_uncertainty_score":0.5544369},"labels":[],"label_agreement":null},{"id":"W4393903283","doi":"10.48550/arxiv.2404.00727","title":"A Controlled Reevaluation of Coreference Resolution Models","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"McGill University; Canadian Institute for Advanced Research; Nvidia","keywords":"Coreference; Resolution (logic); Computer science; Natural language processing; Artificial intelligence","score_opus":0.11287399143340084,"score_gpt":0.22816341770913945,"score_spread":0.11528942627573861,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393903283","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.64783764,0.0033191906,0.3268587,0.0015191037,0.00047919253,0.0004936812,0.0011027652,0.008898792,0.009490908],"genre_scores_gemma":[0.8947882,0.00049831555,0.099455036,0.00049456564,0.000064398104,0.00045186488,0.0010633403,0.00056354434,0.0026207485],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99556446,0.0019842067,0.00023751312,0.0014824672,0.00051186985,0.0002194985],"domain_scores_gemma":[0.9756153,0.016845033,0.00080188067,0.0047977264,0.0016398256,0.00030014542],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009315697,0.0015121526,0.0010838696,0.0007961949,0.00062916044,0.0020894501,0.0033890845,0.0014866416,0.0022663267],"category_scores_gemma":[0.042396065,0.0007621203,0.0008583929,0.00059549895,0.0011465761,0.0039983974,0.0020915968,0.0035618658,0.0010683698],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0025417388,0.0013678754,0.014965918,0.00140678,0.00081657176,0.00031617086,0.0010896048,0.44361603,0.061929777,0.013355242,0.009915293,0.44867897],"study_design_scores_gemma":[0.00024022037,0.0007497368,0.002357848,0.00008859651,0.00023350913,0.00015609153,0.00016142019,0.9501151,0.033246517,0.008158772,0.0044123074,0.00007981534],"about_ca_topic_score_codex":0.0056569516,"about_ca_topic_score_gemma":0.00855287,"teacher_disagreement_score":0.009315697,"about_ca_system_score_codex":0.0015203811,"about_ca_system_score_gemma":0.0018389007,"threshold_uncertainty_score":0.049266696},"labels":[],"label_agreement":null},{"id":"W4393968110","doi":"10.48550/arxiv.2404.02305","title":"Collapse of Self-trained Language Models","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Institute for Catastrophic Loss Reduction","keywords":"Linguistics; Computer science; Natural language processing; Psychology; Philosophy","score_opus":0.036866839979448424,"score_gpt":0.20205614572623784,"score_spread":0.16518930574678942,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393968110","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13547094,0.00048334186,0.8488787,0.00085688493,0.00019083491,0.00014073537,0.00039269286,0.010252401,0.003333445],"genre_scores_gemma":[0.83069,0.00020073356,0.15931478,0.00070632814,0.00008547944,0.00021410795,0.0016957802,0.0012729047,0.0058198404],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9977227,0.00086466694,0.000105081024,0.0007031603,0.00041419256,0.00019026428],"domain_scores_gemma":[0.9874493,0.006970429,0.00041988946,0.0034167727,0.0013820054,0.00036164158],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0040109544,0.0014488202,0.001462837,0.0008951404,0.00069055066,0.0018178929,0.0026220246,0.0019389404,0.0038844761],"category_scores_gemma":[0.026410857,0.0013551138,0.0014068406,0.0007440389,0.0017422596,0.006391633,0.0050596166,0.0051226304,0.0024699713],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00055044633,0.00026440833,0.0052750753,0.0002246321,0.0002709034,0.00040880506,0.0007615994,0.76563585,0.014184173,0.01353509,0.0063943113,0.19249468],"study_design_scores_gemma":[0.000009000137,0.00004873062,0.00017195895,0.000010914974,0.000013218146,0.000040682542,0.000025059697,0.98743594,0.003814412,0.007927205,0.0004921067,0.000010769009],"about_ca_topic_score_codex":0.0055786,"about_ca_topic_score_gemma":0.0072268904,"teacher_disagreement_score":0.0055786,"about_ca_system_score_codex":0.0013398833,"about_ca_system_score_gemma":0.0016154196,"threshold_uncertainty_score":0.02121222},"labels":[],"label_agreement":null},{"id":"W4394045852","doi":"10.5281/zenodo.7089244","title":"Tigrinya Analogy Test for evaluating Word Embeddings","year":2022,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Analogy; Word (group theory); Test (biology); Natural language processing; Linguistics; Computer science; Philosophy; Geology; Paleontology","score_opus":0.043782621840588974,"score_gpt":0.3171485015133652,"score_spread":0.27336587967277626,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4394045852","genre_codex":"empirical","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.70808065,0.0030992834,0.08250789,0.0013762178,0.0030098816,0.0026541906,0.11035707,0.035780687,0.05313407],"genre_scores_gemma":[0.62174344,0.0002992336,0.08492557,0.00094363175,0.00031568858,0.0038016664,0.2734066,0.0032294649,0.01133472],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9878022,0.004873587,0.0019586526,0.0021274441,0.0025354489,0.00070263725],"domain_scores_gemma":[0.975744,0.016115181,0.00086122274,0.0025145286,0.0037994925,0.00096556544],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0061414735,0.0031943105,0.0011344651,0.0047674263,0.0011403533,0.0023084362,0.0017551181,0.0028359783,0.019966068],"category_scores_gemma":[0.0411626,0.00040700004,0.0020713254,0.002735965,0.0009879312,0.002704859,0.0034469084,0.0021121146,0.017030511],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.005067872,0.0021106629,0.06577887,0.0031386209,0.0013901246,0.0024336874,0.001099181,0.022349022,0.022637729,0.004546304,0.47857395,0.390874],"study_design_scores_gemma":[0.003358137,0.0090839295,0.21374188,0.00092772127,0.0006432302,0.0069018872,0.0068728514,0.42151645,0.07860093,0.02107414,0.2365115,0.00076733093],"about_ca_topic_score_codex":0.0027924825,"about_ca_topic_score_gemma":0.004326567,"teacher_disagreement_score":0.019966068,"about_ca_system_score_codex":0.0007028376,"about_ca_system_score_gemma":0.0012407906,"threshold_uncertainty_score":0.066793144},"labels":[],"label_agreement":null},{"id":"W4394055529","doi":"10.5281/zenodo.5129226","title":"Artifact for the paper \"Abstract Interpretation of LLVM with a Region-Based Memory Model\"","year":2021,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Artifact (error); Interpretation (philosophy); Computer science; Natural language processing; Artificial intelligence; Programming language","score_opus":0.028732748717250012,"score_gpt":0.2600215847485285,"score_spread":0.23128883603127848,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4394055529","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00041345,0.0001229987,0.0017796615,0.00019567141,0.00029148423,0.000069810194,0.97574687,0.018726768,0.0026533247],"genre_scores_gemma":[0.0011175934,0.00004608967,0.002279935,0.00013167973,0.000026573194,0.00016215675,0.9932848,0.001593397,0.0013576947],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9980971,0.00035511114,0.00023446791,0.0005548207,0.0005122696,0.00024627298],"domain_scores_gemma":[0.99572766,0.001134792,0.00022829366,0.0019043888,0.000768224,0.00023674486],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015642898,0.0036802588,0.0014232391,0.0024227202,0.00081221375,0.002962912,0.0037674033,0.0017321866,0.13545586],"category_scores_gemma":[0.008019255,0.0009503251,0.0018912329,0.003118908,0.00062738836,0.0021063597,0.0021471228,0.0025706578,0.1520755],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006824933,0.000030330226,0.00018954332,0.00030386815,0.000020435002,0.000011027895,0.0000071341187,0.00031513866,0.00019962683,0.00034268206,0.995363,0.0031489716],"study_design_scores_gemma":[0.0004420827,0.000041347903,0.0019241786,0.00019795689,0.000034193363,0.00009808253,0.000033016702,0.0037531222,0.0023747806,0.003457172,0.9876089,0.00003525584],"about_ca_topic_score_codex":0.011564443,"about_ca_topic_score_gemma":0.024241898,"teacher_disagreement_score":0.13545586,"about_ca_system_score_codex":0.0016972318,"about_ca_system_score_gemma":0.0024258199,"threshold_uncertainty_score":0.45314503},"labels":[],"label_agreement":null},{"id":"W4394057177","doi":"10.5281/zenodo.7089051","title":"Tigrinya Analogy Test for evaluating Word Embeddings","year":2022,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Analogy; Word (group theory); Test (biology); Natural language processing; Computer science; Artificial intelligence; Linguistics; Philosophy; Biology; Paleontology","score_opus":0.043782621840588974,"score_gpt":0.3171485015133652,"score_spread":0.27336587967277626,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4394057177","genre_codex":"empirical","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7192661,0.0029961315,0.071302235,0.0015814354,0.0030941751,0.002907267,0.11002337,0.031799193,0.057030067],"genre_scores_gemma":[0.62054724,0.0002999166,0.07674195,0.0010652912,0.00035069807,0.0041315006,0.28125706,0.003035314,0.0125709465],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.98689574,0.005126572,0.002188773,0.00213413,0.0028999834,0.00075468386],"domain_scores_gemma":[0.9716329,0.01882979,0.0009882612,0.0028822315,0.0045385403,0.0011283155],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0064318185,0.003195583,0.001166538,0.0046031573,0.0011433162,0.0023063242,0.0018330346,0.00298539,0.020826532],"category_scores_gemma":[0.043903537,0.00042515766,0.0021098503,0.0027204251,0.0010172814,0.002800942,0.0035777262,0.0022375027,0.017796662],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0054399674,0.002400959,0.067554995,0.0029001194,0.0013633351,0.002290608,0.0010567474,0.021806441,0.02070831,0.0044693025,0.50114024,0.3688691],"study_design_scores_gemma":[0.003853971,0.010179368,0.22687228,0.0009630229,0.00065608724,0.006688498,0.0066621224,0.39872307,0.07839623,0.02053597,0.24562988,0.0008394625],"about_ca_topic_score_codex":0.0029342375,"about_ca_topic_score_gemma":0.0045726206,"teacher_disagreement_score":0.020826532,"about_ca_system_score_codex":0.0007520614,"about_ca_system_score_gemma":0.0012517972,"threshold_uncertainty_score":0.06967163},"labels":[],"label_agreement":null},{"id":"W4394137698","doi":"10.6084/m9.figshare.825709","title":"Toward an inclusive semantic interoperability: the case of Cree hydrographic features","year":2013,"lang":"en","type":"dataset","venue":"Figshare","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Hydrography; Interoperability; Semantic interoperability; Computer science; Geography; World Wide Web; Information retrieval; Cartography","score_opus":0.03054666412525984,"score_gpt":0.30142219813380466,"score_spread":0.2708755340085448,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4394137698","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.22547649,0.0015243363,0.076201886,0.010906633,0.00020294481,0.0005083687,0.6319103,0.0053111208,0.04795792],"genre_scores_gemma":[0.2461386,0.0006077616,0.12270468,0.000743338,0.000032080232,0.00058352813,0.6248427,0.000538591,0.0038087105],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.997928,0.00069378014,0.00022225315,0.0004870804,0.00049201865,0.00017687943],"domain_scores_gemma":[0.9948237,0.0018552819,0.00030524828,0.0019625644,0.00089836045,0.00015488184],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030428781,0.00029678104,0.00044690416,0.0044637118,0.0013013869,0.0023852307,0.001515567,0.0012750761,0.0019013028],"category_scores_gemma":[0.014112812,0.00020634259,0.000709357,0.010197121,0.0010865838,0.0044415155,0.0029728566,0.0015932514,0.00084860134],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00050447334,0.0003163713,0.116219364,0.002335303,0.00033532546,0.001657871,0.013625808,0.016669083,0.0040340642,0.17673892,0.5027178,0.16484575],"study_design_scores_gemma":[0.000077655895,0.000020983425,0.038580175,0.0003080379,0.000054237426,0.00045998694,0.006434972,0.015181176,0.002153696,0.03235905,0.90431124,0.000058798247],"about_ca_topic_score_codex":0.13562323,"about_ca_topic_score_gemma":0.24468513,"teacher_disagreement_score":0.8643768,"about_ca_system_score_codex":0.0025247987,"about_ca_system_score_gemma":0.0023493136,"threshold_uncertainty_score":0.26966763},"labels":[],"label_agreement":null},{"id":"W4394240925","doi":"10.6084/m9.figshare.19196744","title":"UTAUT2-based questionnaire: cross-cultural adaptation to Canadian French","year":2022,"lang":"en","type":"dataset","venue":"Figshare","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Adaptation (eye); Psychology; Geography; Neuroscience","score_opus":0.029621293707953904,"score_gpt":0.3127964494086247,"score_spread":0.2831751557006708,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4394240925","genre_codex":"empirical","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.82187283,0.0037414576,0.033113103,0.0038376276,0.0010669058,0.013765182,0.0280553,0.0008728225,0.093674764],"genre_scores_gemma":[0.8599831,0.00358235,0.068581775,0.0014571617,0.00008020907,0.019534,0.01606966,0.00029033452,0.030421413],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9958538,0.0012109474,0.00050615525,0.0003668248,0.0015334447,0.0005288195],"domain_scores_gemma":[0.98936427,0.0016694613,0.0003851056,0.00042202097,0.007644878,0.0005143485],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006582188,0.00082144066,0.0006113995,0.0019674927,0.0026980976,0.0016344439,0.0010204002,0.00069373107,0.0073867594],"category_scores_gemma":[0.012463638,0.00030880937,0.0011486439,0.002000461,0.00097386795,0.00079754915,0.0013521378,0.0011898987,0.0015912578],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007041283,0.0008746171,0.26707944,0.0028567393,0.00019302769,0.002231352,0.09204097,0.0025685083,0.010573213,0.008210099,0.12638345,0.48628438],"study_design_scores_gemma":[0.00012339541,0.0005854053,0.66659784,0.0010559397,0.00009964155,0.0018091178,0.026495697,0.0033038594,0.0029464618,0.0013157363,0.2953035,0.0003633925],"about_ca_topic_score_codex":0.63096255,"about_ca_topic_score_gemma":0.6860454,"teacher_disagreement_score":0.36903745,"about_ca_system_score_codex":0.011946105,"about_ca_system_score_gemma":0.025858391,"threshold_uncertainty_score":0.74242157},"labels":[],"label_agreement":null},{"id":"W4394509641","doi":"10.6084/m9.figshare.19969795","title":"Developing and implementing an English-Spanish literary parallel audio-textual corpus for data-driven ESL learning","year":2022,"lang":"en","type":"dataset","venue":"Figshare","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Computer science; Linguistics; Corpus linguistics; Natural language processing; Artificial intelligence; Philosophy","score_opus":0.06614477133137872,"score_gpt":0.33258394322356527,"score_spread":0.2664391718921866,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4394509641","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1212958,0.0022278856,0.053368527,0.0022426855,0.0017144792,0.0025338563,0.7666211,0.028954273,0.02104145],"genre_scores_gemma":[0.03614564,0.00021002334,0.048930865,0.00024633633,0.0000713468,0.0022418168,0.9067604,0.0006050986,0.004788546],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9976712,0.0007981935,0.00023055045,0.0007707379,0.0003863301,0.00014300493],"domain_scores_gemma":[0.99528885,0.001890154,0.0001677255,0.0011872594,0.0011075649,0.0003583919],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031799236,0.001589938,0.00087034836,0.003852429,0.0016324159,0.0019156685,0.002620935,0.001823302,0.010589336],"category_scores_gemma":[0.009047044,0.00041223597,0.0009539657,0.0027996765,0.0010298358,0.0020519188,0.0034225006,0.0020527723,0.013980471],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011151235,0.0017669353,0.011604677,0.003193929,0.00020566504,0.0012804478,0.0016301197,0.008941315,0.011984524,0.005214113,0.7024736,0.25058955],"study_design_scores_gemma":[0.00087240705,0.00045025465,0.02902047,0.0007131939,0.0001549805,0.0010753528,0.004811344,0.064202726,0.021415332,0.0070039704,0.870077,0.0002030692],"about_ca_topic_score_codex":0.012114803,"about_ca_topic_score_gemma":0.02382055,"teacher_disagreement_score":0.012114803,"about_ca_system_score_codex":0.0012985176,"about_ca_system_score_gemma":0.0018280796,"threshold_uncertainty_score":0.03542483},"labels":[],"label_agreement":null},{"id":"W4394571487","doi":"10.21203/rs.3.rs-4190039/v1","title":"Artificial Intelligence and the Spatial Documentation of Languages","year":2024,"lang":"en","type":"preprint","venue":"Research Square","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Documentation; Computer science; Field (mathematics); Process (computing); Artificial intelligence; Natural language processing; Multidisciplinary approach; Data science; Linguistics; World Wide Web; Programming language; Sociology","score_opus":0.055952854758477236,"score_gpt":0.4363178377246973,"score_spread":0.38036498296622007,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4394571487","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15838408,0.0045649637,0.6454817,0.017158167,0.00036075033,0.00014209205,0.0013009706,0.003449075,0.16915822],"genre_scores_gemma":[0.76753163,0.001902771,0.21710344,0.0002731361,0.000104500505,0.000070402355,0.00083296123,0.00043937322,0.011741838],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9977691,0.0015039411,0.00014009952,0.00020068792,0.0003323274,0.000053773467],"domain_scores_gemma":[0.98876756,0.007805069,0.0009877207,0.0014211438,0.00089491444,0.00012366289],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003061354,0.00028380138,0.0002166535,0.0035811227,0.0010428404,0.006391442,0.00081145676,0.0005662741,0.006448334],"category_scores_gemma":[0.017975727,0.00026293282,0.00041424346,0.004798504,0.0040658372,0.0072111026,0.002057773,0.0008573887,0.0010295621],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005198499,0.000030416471,0.0053266417,0.00032857512,0.000030589847,0.00029757913,0.017973356,0.009365645,0.0013254276,0.6866065,0.009384054,0.2692791],"study_design_scores_gemma":[0.000026373538,0.00002828081,0.007403078,0.00046341133,0.000031174914,0.00080399355,0.013583246,0.04978109,0.00427362,0.6923883,0.23114944,0.00006799474],"about_ca_topic_score_codex":0.007898095,"about_ca_topic_score_gemma":0.006688209,"teacher_disagreement_score":0.007898095,"about_ca_system_score_codex":0.0016392947,"about_ca_system_score_gemma":0.0018518127,"threshold_uncertainty_score":0.021571815},"labels":[],"label_agreement":null},{"id":"W4394673405","doi":"10.48550/arxiv.2404.05545","title":"Evaluating Interventional Reasoning Capabilities of Large Language Models","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Computer science","score_opus":0.07922288203065865,"score_gpt":0.27608229331438205,"score_spread":0.19685941128372342,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4394673405","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6127363,0.0031660432,0.35773233,0.0058712238,0.00022435968,0.0008240373,0.0035180307,0.008712031,0.007215611],"genre_scores_gemma":[0.8494945,0.0004626631,0.14615496,0.00057909905,0.000059299626,0.00043043462,0.002051785,0.00020654342,0.00056073925],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99022865,0.0069070174,0.00051781686,0.0012467146,0.0008645369,0.00023518856],"domain_scores_gemma":[0.70277387,0.2849973,0.003952688,0.005391756,0.0019544994,0.0009298562],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.018338367,0.0018705575,0.0009670463,0.0021237785,0.0006148018,0.0023726262,0.0024581193,0.0022732022,0.0031245782],"category_scores_gemma":[0.13249807,0.0007476367,0.0014244907,0.0013527967,0.0016573897,0.004251687,0.0019004018,0.0042033037,0.0005795366],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012825124,0.0010283492,0.016399948,0.0008336225,0.00052015495,0.00024471365,0.00071059127,0.87191147,0.0017488089,0.011130443,0.0031010753,0.091088325],"study_design_scores_gemma":[0.00010166447,0.00017082422,0.0007366671,0.0000349402,0.00005139453,0.000028258813,0.00007057333,0.9840119,0.00091688137,0.013391201,0.0004693029,0.000016266515],"about_ca_topic_score_codex":0.010419548,"about_ca_topic_score_gemma":0.012252458,"teacher_disagreement_score":0.018338367,"about_ca_system_score_codex":0.002582454,"about_ca_system_score_gemma":0.0026095083,"threshold_uncertainty_score":0.09698373},"labels":[],"label_agreement":null},{"id":"W4394753147","doi":"10.5430/wjel.v14n4p143","title":"Analyzing English Translation Studies of the Classic Chinese Novel Jin Ping Mei: A Critical Review and Reflection","year":2024,"lang":"en","type":"review","venue":"World Journal of English Language","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"Hubei Provincial Department of Education","keywords":"Intuition; Computer science; Translation studies; Translation (biology); Ping (video games); Data science; Natural language processing; Artificial intelligence; Linguistics; Epistemology; Chemistry; Philosophy","score_opus":0.05586022174215477,"score_gpt":0.3951306993286039,"score_spread":0.3392704775864491,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4394753147","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004832806,0.98205316,0.0010963092,0.0065018316,0.0013404785,0.00016661815,0.000095899275,0.000011870082,0.003901108],"genre_scores_gemma":[0.03900019,0.9512067,0.0020758347,0.0055124033,0.00064453896,0.00042712849,0.00012986679,0.000032094144,0.00097128144],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.98899406,0.0061480897,0.001998845,0.0007395247,0.0018650724,0.00025451463],"domain_scores_gemma":[0.9243431,0.06366421,0.002554013,0.0011739817,0.007949003,0.00031569324],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.020047849,0.0008554636,0.001711789,0.011445666,0.0018796669,0.0039832424,0.0015625368,0.0016919747,0.0021799868],"category_scores_gemma":[0.062226042,0.0005792406,0.0010092696,0.009932901,0.0041219303,0.0056078555,0.0020154621,0.0026341558,0.00051452924],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016720108,0.000081973914,0.0022750313,0.36636236,0.0008370194,0.0022947383,0.08459199,0.00026356414,0.0019922093,0.034911226,0.031755183,0.4744676],"study_design_scores_gemma":[0.000023282566,0.00013544018,0.005339755,0.36569157,0.001385761,0.001472298,0.046445966,0.00013798142,0.001310614,0.004698889,0.57327664,0.00008178294],"about_ca_topic_score_codex":0.0047370056,"about_ca_topic_score_gemma":0.01141173,"teacher_disagreement_score":0.020047849,"about_ca_system_score_codex":0.0046652644,"about_ca_system_score_gemma":0.014049545,"threshold_uncertainty_score":0.106024384},"labels":[],"label_agreement":null},{"id":"W4394773742","doi":"10.1162/tacl_a_00645","title":"To Diverge or Not to Diverge: A Morphosyntactic Perspective on Machine Translation vs Human Translation","year":2024,"lang":"en","type":"article","venue":"Transactions of the Association for Computational Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Google (Canada)","funders":"","keywords":"Divergence (linguistics); Machine translation; Perspective (graphical); Computer science; Translation (biology); Artificial intelligence; Natural language processing; Diversity (politics); Linguistics; Sociology; Biology","score_opus":0.0292461547114884,"score_gpt":0.3409946650393249,"score_spread":0.3117485103278365,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4394773742","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.91995776,0.004843834,0.0498286,0.0038518733,0.000088554516,0.000034560315,0.00058732426,0.00022923251,0.020578349],"genre_scores_gemma":[0.9933867,0.0003315083,0.0056309565,0.00017803433,0.000036784302,0.000010873398,0.00015859955,0.000064649896,0.0002018292],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99627304,0.0023306054,0.000156055,0.00048698616,0.0006082986,0.00014499112],"domain_scores_gemma":[0.98371243,0.012822358,0.0011469749,0.0009854997,0.0009758917,0.00035681576],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00486376,0.00036844844,0.0006066866,0.002959391,0.00089859887,0.0032786469,0.00046033773,0.0007479464,0.0026829767],"category_scores_gemma":[0.018674683,0.00018334456,0.00029307287,0.0028090577,0.003564146,0.0037500353,0.0015922482,0.0017407467,0.00042136337],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0029607126,0.0003920102,0.19736306,0.0012800625,0.00097285066,0.0022599834,0.027387133,0.01921145,0.15904878,0.2539157,0.0060013747,0.3292069],"study_design_scores_gemma":[0.0001295136,0.0009818107,0.48088744,0.00036926317,0.00038545686,0.0026759796,0.018509686,0.060270254,0.035178296,0.37685463,0.023523709,0.00023402339],"about_ca_topic_score_codex":0.0011833953,"about_ca_topic_score_gemma":0.0020424887,"teacher_disagreement_score":0.00486376,"about_ca_system_score_codex":0.00095232826,"about_ca_system_score_gemma":0.0005157721,"threshold_uncertainty_score":0.025722325},"labels":[],"label_agreement":null},{"id":"W4394793218","doi":"10.61091/ars158-10","title":"Counting Distinct Adjacent r-tuples in Words","year":2024,"lang":"en","type":"article","venue":"Ars Combinatoria","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Tuple; Mathematics; Arithmetic; Combinatorics; Discrete mathematics","score_opus":0.009912383422988145,"score_gpt":0.26386296322184283,"score_spread":0.25395057979885466,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4394793218","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.45558468,0.00054601347,0.49937296,0.0005022603,0.00025143722,0.0001900132,0.0009892342,0.0010719864,0.041491363],"genre_scores_gemma":[0.86245674,0.0004575275,0.11184385,0.000180383,0.00022765323,0.0004926623,0.0015432956,0.0007419905,0.022055916],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9981495,0.0004464185,0.00014724844,0.00035292946,0.00058001874,0.0003239585],"domain_scores_gemma":[0.9949185,0.0027709105,0.00040178845,0.0007171867,0.0008759595,0.00031551986],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015045954,0.00062693766,0.00072970707,0.003847245,0.0013566938,0.0026222952,0.001682173,0.0010439731,0.008283862],"category_scores_gemma":[0.009653188,0.0005919732,0.0010688652,0.0021621266,0.0018690039,0.0037143556,0.0017750966,0.00092706335,0.0026687675],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013525833,0.000055579723,0.005024656,0.00016428091,0.000020360185,0.0004902116,0.0012767388,0.0071859728,0.009752671,0.92542344,0.0036755372,0.046795238],"study_design_scores_gemma":[0.000020908388,0.00007772811,0.0030510537,0.00012621305,0.00004377985,0.001311966,0.00046246365,0.11330811,0.020042349,0.84585613,0.015590805,0.000108484586],"about_ca_topic_score_codex":0.00065372733,"about_ca_topic_score_gemma":0.0007134621,"teacher_disagreement_score":0.008283862,"about_ca_system_score_codex":0.0011122106,"about_ca_system_score_gemma":0.00054209866,"threshold_uncertainty_score":0.027712226},"labels":[],"label_agreement":null},{"id":"W4394930803","doi":"10.5430/wjel.v14n4p404","title":"Transliteration of Arabic Words/Phrase into English: An Exploration of Ambiguity Markers","year":2024,"lang":"en","type":"article","venue":"World Journal of English Language","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Transliteration; Arabic; Natural language processing; Ambiguity; Phrase; Computer science; Artificial intelligence; Linguistics; Philosophy","score_opus":0.013666126605276777,"score_gpt":0.28599401318755246,"score_spread":0.2723278865822757,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4394930803","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9489071,0.0014491163,0.028978622,0.001129946,0.000043932592,0.00017474486,0.00006567798,0.00006293413,0.019187903],"genre_scores_gemma":[0.98860264,0.00054047507,0.009453867,0.00009264998,0.000010883709,0.00004934019,0.000040533872,0.000029190722,0.0011804437],"study_design_codex":"qualitative","study_design_gemma":"not_applicable","domain_scores_codex":[0.9951493,0.0033384191,0.0002987212,0.00026566072,0.00076601736,0.00018189997],"domain_scores_gemma":[0.9828203,0.012907659,0.001898779,0.0005258064,0.0016667638,0.0001806535],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0039193016,0.0004471835,0.00028994182,0.0019105041,0.0018411382,0.002851036,0.00047999766,0.00052686484,0.0012734067],"category_scores_gemma":[0.02025744,0.00023330309,0.0002893084,0.0017213221,0.0023061957,0.0043616383,0.0019873108,0.00080774096,0.0003453462],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035878352,0.00011299385,0.04384984,0.00077486085,0.000028497072,0.004156839,0.78326356,0.0007544847,0.01145588,0.023713738,0.00095926365,0.13057128],"study_design_scores_gemma":[0.000031728403,0.00046593518,0.08999206,0.0013790283,0.00014657946,0.008693778,0.7516036,0.012369531,0.029093757,0.032818582,0.073235735,0.00016969464],"about_ca_topic_score_codex":0.002044958,"about_ca_topic_score_gemma":0.001970676,"teacher_disagreement_score":0.0039193016,"about_ca_system_score_codex":0.0012240013,"about_ca_system_score_gemma":0.0016907364,"threshold_uncertainty_score":0.020727515},"labels":[],"label_agreement":null},{"id":"W4394997761","doi":"10.1016/j.cviu.2024.104016","title":"Structure-aware feature stylization for domain generalization","year":2024,"lang":"en","type":"article","venue":"Computer Vision and Image Understanding","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Generalization; Feature (linguistics); Computer science; Artificial intelligence; Domain (mathematical analysis); Pattern recognition (psychology); Computer vision; Mathematics","score_opus":0.01787896364593283,"score_gpt":0.2960088582924437,"score_spread":0.2781298946465109,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4394997761","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02788749,0.00021903892,0.96614164,0.00016238597,0.000037198657,0.00006205292,0.00019941232,0.0041550254,0.0011357318],"genre_scores_gemma":[0.5034174,0.0004195066,0.48909664,0.00044818455,0.000090290676,0.00018022272,0.0018107768,0.0006941539,0.0038428213],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992318,0.00012125598,0.000053053915,0.00033155945,0.00018198392,0.00008029932],"domain_scores_gemma":[0.998164,0.00041535558,0.00018496417,0.0009662572,0.00019933589,0.00007020374],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012531249,0.00095567765,0.0010292974,0.0013067232,0.0004759961,0.00090898067,0.001417205,0.000884265,0.0019093326],"category_scores_gemma":[0.0038783352,0.00041682873,0.0013994409,0.0010589584,0.0010495465,0.0022431156,0.0023239562,0.0019797403,0.0010850152],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025370042,0.00015319212,0.0035074954,0.00010966224,0.00009241421,0.00023966671,0.00031899463,0.18385169,0.054252118,0.011982863,0.0075119184,0.7377263],"study_design_scores_gemma":[0.00001726962,0.0000556402,0.0008270281,0.000012274758,0.000016471684,0.00022856622,0.00006198955,0.9580479,0.020481184,0.015819553,0.004412341,0.000019725047],"about_ca_topic_score_codex":0.0018770135,"about_ca_topic_score_gemma":0.0029823312,"teacher_disagreement_score":0.0019093326,"about_ca_system_score_codex":0.0007719215,"about_ca_system_score_gemma":0.0007270097,"threshold_uncertainty_score":0.0066272616},"labels":[],"label_agreement":null},{"id":"W4395045955","doi":"10.3389/flang.2024.1327600","title":"Modeling the consequences of an L1 grammar for L2 production: simulations, variation, and predictions","year":2024,"lang":"en","type":"article","venue":"Frontiers in Language Sciences","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Variation (astronomy); Production (economics); Grammar; Computer science; Environmental science; Linguistics; Economics; Physics; Astrophysics; Philosophy","score_opus":0.015113820180145612,"score_gpt":0.29503376793833674,"score_spread":0.2799199477581911,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4395045955","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8783879,0.000048299466,0.11103362,0.0005280552,0.000013826012,0.000055364733,0.00022649308,0.00017415133,0.00953235],"genre_scores_gemma":[0.9805932,0.000028727642,0.018506762,0.000045369303,0.0000034066165,0.00007871775,0.000051028954,0.00004015486,0.0006525785],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996388,0.00017870485,0.00001533434,0.0000741223,0.000053540076,0.00003948558],"domain_scores_gemma":[0.9976593,0.0017553092,0.00014093865,0.00022573087,0.00013617423,0.000082515355],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010301931,0.00025253586,0.00029085716,0.000320517,0.0003290611,0.00088056346,0.0012038597,0.0009565171,0.0021432452],"category_scores_gemma":[0.0071164067,0.00024408732,0.00040488635,0.0003042302,0.0013174358,0.0009992556,0.00064873387,0.0007508733,0.00015405739],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006625066,0.000036419708,0.0076290886,0.00003943817,0.00002436347,0.00014955594,0.0004203773,0.93758315,0.003204803,0.04656903,0.00021465152,0.0040628156],"study_design_scores_gemma":[0.000014606915,0.00002461956,0.00081982586,0.00000573376,0.000006298361,0.000019477735,0.00004995505,0.98026615,0.00078135735,0.017733438,0.00026911608,0.000009513317],"about_ca_topic_score_codex":0.009633152,"about_ca_topic_score_gemma":0.006912645,"teacher_disagreement_score":0.009633152,"about_ca_system_score_codex":0.0013799111,"about_ca_system_score_gemma":0.0009002185,"threshold_uncertainty_score":0.019154191},"labels":[],"label_agreement":null},{"id":"W4395443601","doi":"10.48550/arxiv.2404.15004","title":"TAXI: Evaluating Categorical Knowledge Editing for Language Models","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"Arizona State University","keywords":"Categorical variable; Computer science; Natural language processing; Artificial intelligence; Machine learning","score_opus":0.11970686829732569,"score_gpt":0.27483631995752145,"score_spread":0.15512945166019576,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4395443601","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.50781024,0.015301444,0.1639439,0.0043892604,0.0023062269,0.0017681695,0.10835708,0.1659793,0.030144371],"genre_scores_gemma":[0.5810255,0.0014359487,0.24018858,0.0014983884,0.0003854712,0.00071807666,0.1642348,0.0050059655,0.005507334],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9841903,0.0061992286,0.0015061641,0.0036043203,0.00397524,0.0005247254],"domain_scores_gemma":[0.9289085,0.04908547,0.002470647,0.01390368,0.0042511495,0.0013806121],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013223687,0.0023704795,0.0012818347,0.0042013917,0.0013092988,0.0033943276,0.004883315,0.0033332424,0.0036113723],"category_scores_gemma":[0.08823721,0.0006161985,0.0016539232,0.003480109,0.0016198679,0.006733439,0.0031565065,0.003673006,0.0023771927],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.003469211,0.002149091,0.056702685,0.005539439,0.0019878543,0.00086113205,0.0017478808,0.21060853,0.016375678,0.012874047,0.27406648,0.41361794],"study_design_scores_gemma":[0.00077494996,0.0012211922,0.012034163,0.00028663938,0.00029436007,0.0010277714,0.0007845228,0.8911191,0.019306121,0.019596918,0.05334624,0.00020797987],"about_ca_topic_score_codex":0.016950043,"about_ca_topic_score_gemma":0.028847625,"teacher_disagreement_score":0.016950043,"about_ca_system_score_codex":0.002776038,"about_ca_system_score_gemma":0.0027369643,"threshold_uncertainty_score":0.06993443},"labels":[],"label_agreement":null},{"id":"W4395476790","doi":"","title":"Connecter les chapitres linguistiques de Programming Historian ?: Premières ébauches d'une table conceptuelle multilingue constituée semi-automatiquement","year":2024,"lang":"fr","type":"article","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science","score_opus":0.014866758955773931,"score_gpt":0.2582043866782232,"score_spread":0.24333762772244927,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4395476790","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.032312788,0.027055638,0.7923295,0.03458103,0.0051990654,0.0001634541,0.0020714237,0.0028233072,0.103463784],"genre_scores_gemma":[0.2382349,0.031254824,0.5972369,0.0052400534,0.005185947,0.00046791873,0.0047890455,0.0069752643,0.11061518],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.997586,0.0011261523,0.0002462894,0.00044498857,0.00045127576,0.0001452689],"domain_scores_gemma":[0.99300134,0.0045515504,0.00030110462,0.0008702712,0.0010659373,0.00020977455],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024846012,0.0006367089,0.0006921257,0.0036619648,0.002697088,0.00784964,0.0009449165,0.0013509548,0.015580743],"category_scores_gemma":[0.009156916,0.00073549926,0.00081490923,0.0049323933,0.0036930828,0.009509277,0.0026479734,0.0048736157,0.0037533964],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002084055,0.00006900165,0.002864233,0.00075676566,0.000045126806,0.0005055633,0.018295808,0.0020338385,0.005274234,0.60134065,0.060841538,0.30776477],"study_design_scores_gemma":[0.000011718831,0.000030304696,0.0019145169,0.0006109828,0.000029137711,0.000546274,0.0044295653,0.0034650466,0.002647639,0.08238707,0.9038853,0.000042453215],"about_ca_topic_score_codex":0.014307976,"about_ca_topic_score_gemma":0.021348063,"teacher_disagreement_score":0.015580743,"about_ca_system_score_codex":0.0030630731,"about_ca_system_score_gemma":0.003358096,"threshold_uncertainty_score":0.05212277},"labels":[],"label_agreement":null},{"id":"W4395666468","doi":"10.18280/mmep.110422","title":"Rule-Based Information Extraction from Multi-format Resumes for Automated Classification","year":2024,"lang":"en","type":"article","venue":"Mathematical Modelling and Engineering Problems","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Information retrieval; Information extraction; Data mining","score_opus":0.032775804960709636,"score_gpt":0.27086366904409886,"score_spread":0.23808786408338922,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4395666468","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08112526,0.0037394115,0.8233704,0.0012031529,0.0007843863,0.001589947,0.029938033,0.037157767,0.02109164],"genre_scores_gemma":[0.19001134,0.00176459,0.74110335,0.00028537298,0.0002608891,0.0007280371,0.054142635,0.000493559,0.011210229],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989184,0.0001176552,0.00020647595,0.000319874,0.00035811702,0.0000795745],"domain_scores_gemma":[0.9976719,0.0008877437,0.00026189268,0.0003775538,0.0007297932,0.00007115902],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00088219356,0.0012366864,0.0008559451,0.0063491357,0.0006477766,0.0017654208,0.0013064749,0.00087604683,0.0051707155],"category_scores_gemma":[0.0044042403,0.00028115374,0.0013638871,0.00425118,0.00035327405,0.0018506179,0.00077340705,0.0012088703,0.006905326],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028472423,0.00041428747,0.0057944204,0.0008731838,0.00009549262,0.0011648054,0.00040445323,0.005216834,0.036525425,0.003152447,0.03271833,0.91335565],"study_design_scores_gemma":[0.0001602123,0.00063352444,0.040674683,0.0010802957,0.00057344465,0.002856375,0.0027436893,0.5208401,0.18551144,0.027096298,0.21749276,0.00033717699],"about_ca_topic_score_codex":0.0040983194,"about_ca_topic_score_gemma":0.0055690217,"teacher_disagreement_score":0.0063491357,"about_ca_system_score_codex":0.0005194321,"about_ca_system_score_gemma":0.0013358868,"threshold_uncertainty_score":0.017297745},"labels":[],"label_agreement":null},{"id":"W4396220737","doi":"10.54254/2755-2721/57/20241325","title":"Advancements and challenges in AI-driven language technologies: From natural language processing to language acquisition","year":2024,"lang":"en","type":"article","venue":"Applied and Computational Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Thompson Rivers University","funders":"","keywords":"Computer science; Interpretability; Artificial intelligence; Machine translation; Natural language processing; Natural language; Language acquisition; Language model; Universal Networking Language; Language technology; Computational linguistics; Language industry; Comprehension approach; Linguistics","score_opus":0.0071218990408329925,"score_gpt":0.247057581249163,"score_spread":0.23993568220833,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4396220737","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.039193545,0.17863534,0.3799558,0.25835714,0.0019474596,0.00010507676,0.0003145465,0.0011318135,0.14035936],"genre_scores_gemma":[0.53367025,0.18258686,0.23428817,0.01734195,0.0046328786,0.0002800516,0.00039815018,0.000616366,0.026185254],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9966024,0.0016481202,0.00019834626,0.00037498644,0.0010070554,0.00016900907],"domain_scores_gemma":[0.9837219,0.01319586,0.0004450569,0.0010594367,0.0012052679,0.00037240816],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0070213107,0.0004634689,0.0005510729,0.0018596204,0.0011477071,0.0075199953,0.001458534,0.0024181353,0.00415897],"category_scores_gemma":[0.013850011,0.0003593493,0.00036338717,0.0017613214,0.00642954,0.018073194,0.0036639227,0.004323984,0.0021526467],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000047279947,0.000090662135,0.0015033936,0.0007877411,0.00001952468,0.00014407122,0.0024037694,0.0031342246,0.0017690596,0.6549178,0.00853458,0.3266479],"study_design_scores_gemma":[0.000008957111,0.00007553084,0.000785857,0.0007963418,0.000013712793,0.0004873437,0.0027787017,0.01840238,0.0036003576,0.74624354,0.22674187,0.000065387954],"about_ca_topic_score_codex":0.0011985644,"about_ca_topic_score_gemma":0.0010263924,"teacher_disagreement_score":0.0075199953,"about_ca_system_score_codex":0.0021650419,"about_ca_system_score_gemma":0.0022109593,"threshold_uncertainty_score":0.03713268},"labels":[],"label_agreement":null},{"id":"W4396531937","doi":"10.22215/etd/2024-15911","title":"Post-processing Techniques for Word Embedding","year":2024,"lang":"en","type":"dissertation","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"University of Ottawa","keywords":"Word embedding; Computer science; Word (group theory); Natural language processing; Embedding; Linguistics; Artificial intelligence; Arithmetic; Mathematics; Philosophy","score_opus":0.011466207255166085,"score_gpt":0.3307872894495249,"score_spread":0.31932108219435884,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4396531937","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0071240384,0.00021175673,0.98905796,0.00016853698,0.00010619119,0.00009086688,0.00022993605,0.0021935184,0.00081719924],"genre_scores_gemma":[0.11686128,0.0007290673,0.8705763,0.00019118194,0.00013080079,0.00036524286,0.0025148673,0.0008315224,0.00779974],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99873954,0.00028993312,0.00014601891,0.0003254438,0.0003930989,0.00010605307],"domain_scores_gemma":[0.99617285,0.0013518198,0.00024977565,0.0010061496,0.0011479615,0.00007142583],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001753323,0.0016623209,0.0009084685,0.0011878568,0.0005818432,0.001376903,0.001260927,0.00090176624,0.009924922],"category_scores_gemma":[0.008997589,0.00068665843,0.0012553086,0.0014498106,0.00087340554,0.0033619583,0.0020175665,0.0028563167,0.0067185517],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017586796,0.00014073019,0.00091158604,0.00038100098,0.00009629456,0.00013385646,0.00027800188,0.04251568,0.03996946,0.015776742,0.009090693,0.89053005],"study_design_scores_gemma":[0.000036944417,0.00020786992,0.0014814921,0.000053943142,0.00005725491,0.00017746261,0.00017642314,0.86508185,0.084342204,0.029308615,0.019032842,0.000043109434],"about_ca_topic_score_codex":0.0030262405,"about_ca_topic_score_gemma":0.0066243173,"teacher_disagreement_score":0.009924922,"about_ca_system_score_codex":0.00069802214,"about_ca_system_score_gemma":0.0015018722,"threshold_uncertainty_score":0.03320217},"labels":[],"label_agreement":null},{"id":"W4396600612","doi":"","title":"Détection automatique de propos misogynes en ligne: le cas de Reddit et des communautés Incels","year":2024,"lang":"fr","type":"article","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science","score_opus":0.012847674620103475,"score_gpt":0.28053553476220433,"score_spread":0.26768786014210083,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4396600612","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9080169,0.0010706902,0.07272386,0.0009730423,0.00025544743,0.00008612718,0.001105942,0.0040279054,0.011740081],"genre_scores_gemma":[0.93818295,0.00031945933,0.04168617,0.00013781835,0.000058152953,0.000038670747,0.0010117119,0.00038331936,0.01818177],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.999345,0.00004954694,0.0000318126,0.00019044557,0.00027782196,0.00010541523],"domain_scores_gemma":[0.9979888,0.00092250155,0.0001888059,0.00019100835,0.0005690309,0.00013977975],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00038800525,0.0007574286,0.00040300027,0.0018028403,0.0011223695,0.0015672864,0.0004780313,0.0015099583,0.0040474813],"category_scores_gemma":[0.0023778963,0.00035463274,0.00042304955,0.00086587336,0.00047542402,0.0007377083,0.0006773285,0.0009815645,0.0011980262],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018835581,0.00042038085,0.18418856,0.0006813535,0.00021542472,0.06309095,0.005314239,0.03288486,0.32324344,0.0131702395,0.021963853,0.35294315],"study_design_scores_gemma":[0.000119312994,0.00040472692,0.1377256,0.00021492751,0.00034268075,0.037774358,0.0041904976,0.40282646,0.3185843,0.005937294,0.09167388,0.00020599714],"about_ca_topic_score_codex":0.015961848,"about_ca_topic_score_gemma":0.025138577,"teacher_disagreement_score":0.015961848,"about_ca_system_score_codex":0.00056169176,"about_ca_system_score_gemma":0.00076325116,"threshold_uncertainty_score":0.031737924},"labels":[],"label_agreement":null},{"id":"W4396821301","doi":"10.48550/arxiv.2404.18923","title":"Holmes: A Benchmark to Assess the Linguistic Competence of Language Models","year":2024,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"University of Oxford; Institute for Catastrophic Loss Reduction","keywords":"Benchmark (surveying); Linguistics; Linguistic competence; Competence (human resources); Computer science; Linguistic description; Natural language processing; Psychology; Geography; Philosophy; Social psychology","score_opus":0.05603746822224436,"score_gpt":0.3264610893305728,"score_spread":0.2704236211083284,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4396821301","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.32764196,0.012261999,0.3459581,0.0033784872,0.0019444262,0.0016206704,0.120293535,0.13684124,0.050059598],"genre_scores_gemma":[0.50094956,0.0014515851,0.28660658,0.0010207586,0.00025106617,0.0013247409,0.19497834,0.0057003982,0.007717038],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99131835,0.0035034132,0.0013204535,0.0012974391,0.002147895,0.00041237558],"domain_scores_gemma":[0.9746766,0.015477478,0.0009186522,0.004946144,0.0032665886,0.0007145347],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007245083,0.0026260442,0.000981695,0.0059372573,0.0010099207,0.0041395095,0.002813656,0.002577884,0.005835937],"category_scores_gemma":[0.05195287,0.00062498386,0.0016313756,0.0032796457,0.0010051597,0.0066345106,0.0040634884,0.0021549535,0.005362447],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021214902,0.0010458002,0.050860584,0.0046220287,0.0014453818,0.00050494407,0.0016479851,0.07747753,0.013293858,0.019369856,0.27102244,0.55658823],"study_design_scores_gemma":[0.0006379615,0.0016352792,0.025708433,0.00072409253,0.00040350776,0.0008824917,0.0017782147,0.7201078,0.026867758,0.071427114,0.14950284,0.00032457686],"about_ca_topic_score_codex":0.0066666035,"about_ca_topic_score_gemma":0.009132595,"teacher_disagreement_score":0.007245083,"about_ca_system_score_codex":0.0012906798,"about_ca_system_score_gemma":0.0022510814,"threshold_uncertainty_score":0.03831607},"labels":[],"label_agreement":null},{"id":"W4396912844","doi":"10.48550/arxiv.2405.06665","title":"Enhancing Language Models for Financial Relation Extraction with Named Entities and Part-of-Speech","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Institute for Catastrophic Loss Reduction","keywords":"Relation (database); Relationship extraction; Computer science; Natural language processing; Extraction (chemistry); Artificial intelligence; Linguistics; Finance; Business; Data mining; Philosophy; Chemistry","score_opus":0.034908821738231545,"score_gpt":0.21069686015775566,"score_spread":0.17578803841952412,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4396912844","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07251985,0.0035955044,0.8600608,0.0023458258,0.0007390706,0.00033048788,0.015020567,0.03524661,0.010141354],"genre_scores_gemma":[0.3664724,0.0022264312,0.5414477,0.0013730976,0.0004663474,0.0005479128,0.072154574,0.0014598557,0.013851702],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.998901,0.00035462435,0.00008948813,0.00036167164,0.00019578,0.000097390286],"domain_scores_gemma":[0.9966614,0.0021383087,0.000176632,0.00047274271,0.00046108974,0.000089748326],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020323629,0.0019355641,0.00097467465,0.002442901,0.0007124725,0.0018779044,0.0018095202,0.0013155814,0.0046356595],"category_scores_gemma":[0.005667503,0.00062952575,0.0020706404,0.0021140086,0.00044151652,0.005372989,0.0015371002,0.002852058,0.009855899],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00074988336,0.00094349374,0.0077729155,0.0007403137,0.00041436966,0.0010110666,0.00049290986,0.12746973,0.025594885,0.012908204,0.088150814,0.7337514],"study_design_scores_gemma":[0.00004945509,0.00009583637,0.0013395713,0.00006440707,0.000116303905,0.00028623655,0.00017204545,0.9442818,0.014177963,0.01836095,0.02099844,0.00005696113],"about_ca_topic_score_codex":0.007963016,"about_ca_topic_score_gemma":0.0165292,"teacher_disagreement_score":0.007963016,"about_ca_system_score_codex":0.0009078718,"about_ca_system_score_gemma":0.0018175219,"threshold_uncertainty_score":0.015833318},"labels":[],"label_agreement":null},{"id":"W4396913016","doi":"10.48550/arxiv.2405.06806","title":"An Empirical Study on the Effectiveness of Large Language Models for SATD Identification and Classification","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Identification (biology); Computer science; Empirical research; Natural language processing; Artificial intelligence; Statistics; Mathematics; Biology","score_opus":0.07916967803166322,"score_gpt":0.28761757908907515,"score_spread":0.20844790105741193,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4396913016","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.84707445,0.020839809,0.0870686,0.0022464562,0.0011680648,0.0006437675,0.009486014,0.018828278,0.01264461],"genre_scores_gemma":[0.9010392,0.0017821637,0.07184645,0.00082817394,0.00023533794,0.0003136106,0.019696837,0.0006286959,0.0036295273],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99444675,0.0025060838,0.0005341633,0.0015330388,0.00066694146,0.0003129447],"domain_scores_gemma":[0.97748524,0.017172175,0.00089046743,0.0024855854,0.0014631622,0.00050333585],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0075934473,0.0032878346,0.0014373056,0.002885586,0.0010001708,0.0021682242,0.0020748847,0.0024839267,0.0029481256],"category_scores_gemma":[0.026608061,0.0006600479,0.0016929238,0.001972471,0.00087923213,0.005385645,0.0017836696,0.0037036757,0.0034550764],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0023443392,0.0022839976,0.05505833,0.0022162464,0.0012488513,0.0006922815,0.00067115633,0.12512164,0.010579116,0.002377827,0.049498808,0.74790734],"study_design_scores_gemma":[0.00021845472,0.0010602142,0.013917884,0.00028414096,0.00038663077,0.0007655826,0.00058522006,0.95537627,0.015203679,0.0032603545,0.008797695,0.00014379939],"about_ca_topic_score_codex":0.012291514,"about_ca_topic_score_gemma":0.01891925,"teacher_disagreement_score":0.012291514,"about_ca_system_score_codex":0.0020657738,"about_ca_system_score_gemma":0.0013783101,"threshold_uncertainty_score":0.04015851},"labels":[],"label_agreement":null},{"id":"W4396916472","doi":"10.3765/plsa.v9i1.5683","title":"A &lt;i&gt;wh&lt;/i&gt; discourse particle: Dutch &lt;i&gt;hoezo&lt;/i&gt;","year":2024,"lang":"en","type":"article","venue":"Proceedings of the Linguistic Society of America","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Psychology","score_opus":0.010462493358279857,"score_gpt":0.26840627540748635,"score_spread":0.2579437820492065,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4396916472","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.36091527,0.001814184,0.2971033,0.0021449153,0.00040849514,0.00047746286,0.0045148856,0.0005568859,0.33206466],"genre_scores_gemma":[0.9566038,0.00034178843,0.019869393,0.00016134506,0.00007319623,0.00015202191,0.0016825879,0.0005498311,0.020565927],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9988826,0.00036509594,0.00007817717,0.00035078754,0.00018566077,0.00013767355],"domain_scores_gemma":[0.99903226,0.00048295135,0.000106271786,0.00017502852,0.00017143197,0.000032144457],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00091220596,0.0005152043,0.00052128336,0.00083347096,0.0016329677,0.003143834,0.00071016466,0.001469711,0.012585836],"category_scores_gemma":[0.0019658296,0.00040629975,0.00057147065,0.0012429908,0.0032348481,0.0050117155,0.0017564907,0.0013651587,0.0016255946],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00033216347,0.00007506443,0.0075016455,0.0008013664,0.000071668255,0.0023321859,0.06591578,0.0011278393,0.042650025,0.83109695,0.008271898,0.03982339],"study_design_scores_gemma":[0.00011215538,0.000114282084,0.03187568,0.00025778534,0.00015055685,0.004539788,0.032373145,0.018835064,0.055904873,0.17048948,0.685168,0.00017919319],"about_ca_topic_score_codex":0.011005435,"about_ca_topic_score_gemma":0.01152709,"teacher_disagreement_score":0.012585836,"about_ca_system_score_codex":0.0024197567,"about_ca_system_score_gemma":0.00067968154,"threshold_uncertainty_score":0.042103887},"labels":[],"label_agreement":null},{"id":"W4396924654","doi":"10.1075/sibil.67.06ros","title":"Multilingual data coding and analysis within Phon","year":2024,"lang":"en","type":"book-chapter","venue":"Studies in bilingualism","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Memorial University of Newfoundland","funders":"","keywords":"Computer science; Coding (social sciences); Sociology; Social science","score_opus":0.10119838768139672,"score_gpt":0.39429807081351154,"score_spread":0.2930996831321148,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4396924654","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011849802,0.00029050795,0.9192276,0.00076614856,0.00016750317,0.00077810156,0.0024839635,0.0060000923,0.058436256],"genre_scores_gemma":[0.08899817,0.00035457735,0.8647703,0.00018281466,0.00007432338,0.001771593,0.0036609478,0.004105451,0.036081832],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9919186,0.0039831595,0.0007172638,0.0010730448,0.0021155933,0.00019237571],"domain_scores_gemma":[0.98465943,0.008454673,0.0005617234,0.003237362,0.0029516208,0.0001352563],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008332094,0.0005653541,0.00045726035,0.0039082225,0.0011147042,0.0043213936,0.0010070173,0.00041772155,0.019439975],"category_scores_gemma":[0.013901274,0.0005411245,0.00034746437,0.0054144156,0.0018914145,0.003566482,0.003650242,0.001511697,0.0044926843],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009804007,0.000057817226,0.0030259672,0.00071278005,0.000029591863,0.00047883054,0.02092593,0.0012705824,0.009797568,0.2871557,0.03501435,0.64143276],"study_design_scores_gemma":[0.00001888926,0.00005798786,0.005443143,0.0005481206,0.000027559856,0.00096562604,0.0094705215,0.02359103,0.025540758,0.13310812,0.8011582,0.00007005834],"about_ca_topic_score_codex":0.002023761,"about_ca_topic_score_gemma":0.003656053,"teacher_disagreement_score":0.019439975,"about_ca_system_score_codex":0.0019328007,"about_ca_system_score_gemma":0.0025772904,"threshold_uncertainty_score":0.06503314},"labels":[],"label_agreement":null},{"id":"W4397012280","doi":"10.1017/nlp.2024.7","title":"A survey of context in neural machine translation and its evaluation","year":2024,"lang":"en","type":"article","venue":"Natural language processing.","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"Dublin City University; Science Foundation Ireland","keywords":"Machine translation; Computer science; Artificial intelligence; Evaluation of machine translation; Context (archaeology); Natural language processing; Terminology; Machine translation software usability; Consistency (knowledge bases); Example-based machine translation; Sentence; Computer-assisted translation; Task (project management); Paragraph; Transfer-based machine translation; Machine learning; Linguistics; World Wide Web; Engineering","score_opus":0.030556959899479947,"score_gpt":0.33267595850217146,"score_spread":0.30211899860269154,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4397012280","genre_codex":"review","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17011558,0.66231364,0.12464162,0.0035201954,0.00090769015,0.0005511797,0.0013558522,0.0029700855,0.03362423],"genre_scores_gemma":[0.8122588,0.06769282,0.11174368,0.0010037926,0.000789605,0.00043112386,0.0026380115,0.0006064857,0.0028356737],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.98102933,0.012083333,0.0016058793,0.0015827799,0.0033802798,0.00031839617],"domain_scores_gemma":[0.97104836,0.019442126,0.0013356643,0.0017467352,0.006013337,0.00041394992],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0145920655,0.0015053155,0.0022158772,0.0054267994,0.0009528288,0.0029869883,0.0023208803,0.002165894,0.0024535498],"category_scores_gemma":[0.042875595,0.0005654492,0.0010717568,0.0049382616,0.0010596183,0.0033049393,0.0019217391,0.0012976509,0.00065929943],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016117473,0.00022742429,0.008395502,0.0038974353,0.00092337007,0.00009182491,0.0002090971,0.043230996,0.0029817754,0.005636111,0.0055218353,0.9272729],"study_design_scores_gemma":[0.0005812291,0.004859946,0.03096424,0.00749666,0.002693787,0.0010686708,0.0010997849,0.7670239,0.04423236,0.05328752,0.08629181,0.0004000997],"about_ca_topic_score_codex":0.0064816517,"about_ca_topic_score_gemma":0.0059772097,"teacher_disagreement_score":0.0145920655,"about_ca_system_score_codex":0.002650352,"about_ca_system_score_gemma":0.0015265252,"threshold_uncertainty_score":0.07717115},"labels":[{"model":"gemma","categories":[],"domain":null,"study_design":"simulation_or_modeling","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"low"},{"model":"gpt","categories":[],"domain":null,"study_design":"design_other","genre":"review","about_ca_system":false,"about_ca_topic":false,"confidence":"high"}],"label_agreement":"split"},{"id":"W4398231281","doi":"10.1103/physreve.109.054313","title":"Robustness of the random language model","year":2024,"lang":"en","type":"article","venue":"Physical review. E","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Robustness (evolution); Computer science; Natural language processing; Chemistry","score_opus":0.012809017430204022,"score_gpt":0.33426937049232824,"score_spread":0.3214603530621242,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4398231281","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.32417357,0.0032920293,0.5997179,0.01089114,0.0006941132,0.00022405105,0.0017960556,0.002353862,0.056857303],"genre_scores_gemma":[0.9787379,0.00060315564,0.014605091,0.0009221335,0.00031722477,0.00014958392,0.00065436657,0.00027799042,0.0037325865],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99494284,0.002591224,0.00019370159,0.0011174622,0.0006685659,0.00048614372],"domain_scores_gemma":[0.9698397,0.019735517,0.0022778083,0.0060075587,0.0013939328,0.0007454169],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0061355736,0.00074138504,0.0018825829,0.0012826768,0.001248393,0.0026761438,0.0024287503,0.002360688,0.0041272757],"category_scores_gemma":[0.044758756,0.00069662987,0.001677304,0.00043211857,0.0033012107,0.004976208,0.0032545913,0.003270269,0.0010622962],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00047538552,0.00007945762,0.0033773307,0.00020707745,0.00033869618,0.00039008274,0.0002893532,0.41198224,0.0040279184,0.55875796,0.005321185,0.014753316],"study_design_scores_gemma":[0.00005150976,0.00006643075,0.0009628478,0.000038117243,0.00004142772,0.00015457092,0.000040232888,0.59325504,0.00080945744,0.40247613,0.0020275565,0.00007665729],"about_ca_topic_score_codex":0.0050056605,"about_ca_topic_score_gemma":0.0016085092,"teacher_disagreement_score":0.0061355736,"about_ca_system_score_codex":0.0017138525,"about_ca_system_score_gemma":0.0013219877,"threshold_uncertainty_score":0.03244835},"labels":[],"label_agreement":null},{"id":"W4398288000","doi":"10.7910/dvn/66hucd/cbummy","title":"metadata.xml","year":2019,"lang":"en","type":"dataset","venue":"Harvard Dataverse","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Computer science; Metadata; Natural language processing; XML; Artificial intelligence; Information retrieval; World Wide Web","score_opus":0.018963922115092603,"score_gpt":0.2698429114328492,"score_spread":0.2508789893177566,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4398288000","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00013363383,0.00008660736,0.00029400244,0.00014436537,0.000065083725,0.000029036126,0.9925079,0.004312363,0.0024270662],"genre_scores_gemma":[0.0004175171,0.00010657219,0.000586151,0.00007862646,0.000013780311,0.0000616889,0.996636,0.00067425106,0.0014254455],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99787784,0.00032555833,0.0003389472,0.00058968255,0.00056506996,0.00030295522],"domain_scores_gemma":[0.99604475,0.00070861995,0.00029508842,0.001616602,0.0007521619,0.0005828137],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0018068736,0.004762641,0.0022586863,0.011860931,0.0016778974,0.005730662,0.0045864754,0.0027135643,0.22538571],"category_scores_gemma":[0.010339159,0.0013764048,0.0016078409,0.014443963,0.0008501787,0.0040679155,0.004576596,0.0027150745,0.30318722],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000056383768,0.000025353733,0.0002563549,0.00050212845,0.000017072642,0.00001525202,0.000028194318,0.00009747,0.00021862132,0.0012065929,0.9929391,0.004637432],"study_design_scores_gemma":[0.00013086546,0.000022461709,0.00086007325,0.00022388858,0.000018368273,0.000050109367,0.00007269229,0.0005166264,0.0009129576,0.0024427108,0.9947237,0.000025596117],"about_ca_topic_score_codex":0.017814266,"about_ca_topic_score_gemma":0.01793122,"teacher_disagreement_score":0.7746143,"about_ca_system_score_codex":0.0020797064,"about_ca_system_score_gemma":0.0042128707,"threshold_uncertainty_score":0.7539903},"labels":[],"label_agreement":null},{"id":"W4398341800","doi":"10.7910/dvn/66hucd","title":"Turk Talk hybrid natural language processing (NLP) virtual scenario example","year":2019,"lang":"en","type":"dataset","venue":"Harvard Dataverse","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Natural (archaeology); Linguistics; History","score_opus":0.014038222031516522,"score_gpt":0.2577051088327881,"score_spread":0.24366688680127155,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4398341800","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0074761985,0.00016570624,0.0018245252,0.00060373463,0.0002632872,0.0002727244,0.9757268,0.0024798128,0.011187263],"genre_scores_gemma":[0.012424266,0.00006678898,0.003077488,0.00017469957,0.00002687042,0.00057243666,0.97708124,0.0001159736,0.006460228],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99915886,0.00027023014,0.00008898625,0.00019252933,0.00016277554,0.00012655351],"domain_scores_gemma":[0.9981413,0.00066531287,0.000053952026,0.0005303598,0.00043813125,0.00017100798],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000999573,0.0013889868,0.00047669373,0.0013462283,0.001074319,0.0010587389,0.001756656,0.0016408305,0.06241789],"category_scores_gemma":[0.003460285,0.00030434565,0.0008880097,0.001511304,0.00043677198,0.0008946459,0.0015724995,0.0014725078,0.058247518],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00026692406,0.0001339874,0.0014739507,0.00038975582,0.000028514994,0.00019701013,0.00009330085,0.0007981967,0.00023375326,0.0008248185,0.9860044,0.009555341],"study_design_scores_gemma":[0.00053032755,0.00016037252,0.012725512,0.00031507283,0.000044697998,0.0006456042,0.0009148049,0.0074107824,0.0030283087,0.0035742852,0.9705757,0.00007450054],"about_ca_topic_score_codex":0.013709922,"about_ca_topic_score_gemma":0.05471781,"teacher_disagreement_score":0.06241789,"about_ca_system_score_codex":0.0009970755,"about_ca_system_score_gemma":0.0008947614,"threshold_uncertainty_score":0.20880866},"labels":[],"label_agreement":null},{"id":"W4398343469","doi":"10.7910/dvn/eaxeet/gim2my","title":"TopCitedWorksPoetry_Genre.tab","year":2019,"lang":"en","type":"dataset","venue":"Harvard Dataverse","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science","score_opus":0.016013502482509483,"score_gpt":0.26643407108090505,"score_spread":0.25042056859839557,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4398343469","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.000103049526,0.000067829336,0.000039098046,0.00005206586,0.000036496738,0.000008323693,0.9977946,0.00043716657,0.0014613308],"genre_scores_gemma":[0.00035489167,0.00008661812,0.0001898469,0.000037967246,0.000014558003,0.000047258705,0.997447,0.00012754144,0.0016943179],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9992213,0.000090571244,0.00009738092,0.0002383212,0.00018984466,0.00016246253],"domain_scores_gemma":[0.99794716,0.00045341026,0.0001956686,0.0005114093,0.00058903726,0.00030332175],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.00053347263,0.0030502265,0.0016752277,0.0089731235,0.001158627,0.0038692714,0.002883958,0.0019029143,0.21287276],"category_scores_gemma":[0.0044150045,0.000725193,0.0011742361,0.013044515,0.0005736111,0.0022770436,0.0027454644,0.0018187732,0.27349612],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000026729622,0.00001124231,0.00024029717,0.00048488966,0.000008455847,0.0000122950905,0.000017935223,0.00004985128,0.00009015408,0.00036860062,0.99664897,0.0020404642],"study_design_scores_gemma":[0.00009058094,0.000010556089,0.0014299364,0.00020524916,0.000012298987,0.000040527655,0.000096072545,0.00014996341,0.00035899895,0.00061002007,0.9969799,0.000015834828],"about_ca_topic_score_codex":0.015503793,"about_ca_topic_score_gemma":0.02447092,"teacher_disagreement_score":0.78712726,"about_ca_system_score_codex":0.0015015183,"about_ca_system_score_gemma":0.0020073405,"threshold_uncertainty_score":0.7121303},"labels":[],"label_agreement":null},{"id":"W4398347418","doi":"10.7910/dvn/eaxeet/kltm0y","title":"goetheUnread_ReadMe.txt","year":2019,"lang":"en","type":"dataset","venue":"Harvard Dataverse","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Mathematics; Computer science; Database; Statistics","score_opus":0.01801581548594115,"score_gpt":0.27210077579110337,"score_spread":0.2540849603051622,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4398347418","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00007704347,0.00004337376,0.000054501492,0.00006724602,0.000032329925,0.000011440728,0.99761707,0.0011455604,0.0009514638],"genre_scores_gemma":[0.00026865915,0.000041676132,0.00019464194,0.00004879599,0.000010908083,0.000068433736,0.99818194,0.00023811212,0.00094683666],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9986979,0.00020147234,0.00014093729,0.0004045312,0.000277479,0.00027763913],"domain_scores_gemma":[0.99703014,0.0006677226,0.00023054032,0.0010126502,0.000683056,0.0003757767],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.001184588,0.0038809152,0.0019239482,0.0055771605,0.001203298,0.0045792623,0.0048228004,0.0027500861,0.20731121],"category_scores_gemma":[0.007296334,0.0009708456,0.0017066143,0.008165538,0.00074965786,0.0023325612,0.0038680697,0.0021543098,0.31379217],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003730655,0.000011645207,0.00016303893,0.00031102213,0.000013738443,0.000009311707,0.000013066651,0.00007123724,0.00006097016,0.0002579952,0.99790597,0.0011446438],"study_design_scores_gemma":[0.00025442368,0.000022664926,0.0012842894,0.00021172555,0.000021323265,0.000039439354,0.00008275595,0.00034641405,0.0005212602,0.0015727874,0.99561584,0.000027161983],"about_ca_topic_score_codex":0.017352102,"about_ca_topic_score_gemma":0.026216812,"teacher_disagreement_score":0.7926888,"about_ca_system_score_codex":0.0016338059,"about_ca_system_score_gemma":0.0023605353,"threshold_uncertainty_score":0.6935251},"labels":[],"label_agreement":null},{"id":"W4398553853","doi":"10.7910/dvn/66hucd/7c6rqc","title":"An_assessment_of_the_influence_of_cueing_items_in.10.pdf","year":2019,"lang":"en","type":"dataset","venue":"Harvard Dataverse","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Natural language processing; Computer science; Artificial intelligence; Linguistics; Natural (archaeology); Speech recognition; History","score_opus":0.016192192326945107,"score_gpt":0.2840858691775789,"score_spread":0.26789367685063376,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4398553853","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0005337925,0.0001340103,0.000086673695,0.0001654292,0.000072051545,0.000031094252,0.99755317,0.00049160334,0.00093206775],"genre_scores_gemma":[0.0010261815,0.000058538986,0.0004938971,0.000054027994,0.000013515645,0.00014872153,0.99686325,0.00007535065,0.0012665704],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99826187,0.0004330899,0.00019463617,0.0004301115,0.00039058746,0.00028965157],"domain_scores_gemma":[0.9949626,0.0017073337,0.00032187873,0.0013028608,0.0013303248,0.00037504215],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0027274874,0.0033492725,0.0017126903,0.0044183973,0.0015870505,0.0032572327,0.0052928347,0.002670936,0.0662236],"category_scores_gemma":[0.013248145,0.0008359389,0.002942014,0.004579206,0.0006281398,0.0019550286,0.0034500447,0.0024770023,0.10365227],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013784421,0.000068332316,0.0016651853,0.00093115505,0.00007632813,0.000020358157,0.000031187526,0.0002229279,0.000100554236,0.00024329357,0.99284816,0.0036545815],"study_design_scores_gemma":[0.001348285,0.00010351318,0.0164869,0.0008192206,0.000258968,0.00017484809,0.00043319602,0.0021568323,0.0012733242,0.0018481002,0.97499245,0.00010435629],"about_ca_topic_score_codex":0.045273483,"about_ca_topic_score_gemma":0.11554159,"teacher_disagreement_score":0.9337764,"about_ca_system_score_codex":0.0021227843,"about_ca_system_score_gemma":0.0033800243,"threshold_uncertainty_score":0.22153997},"labels":[],"label_agreement":null},{"id":"W4398708237","doi":"10.7910/dvn/eaxeet/jm9gdx","title":"Corpus_ngram3.tab","year":2019,"lang":"fr","type":"dataset","venue":"Harvard Dataverse","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Natural language processing; Computer science; Linguistics; Philosophy","score_opus":0.027780468323702352,"score_gpt":0.2750121536315191,"score_spread":0.24723168530781675,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4398708237","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00033241577,0.00012373315,0.00017787999,0.000100267156,0.000051665706,0.000028604058,0.9942221,0.0035011852,0.0014620814],"genre_scores_gemma":[0.0005985424,0.00005087566,0.00056571,0.000041333857,0.000013503289,0.00010292123,0.9972293,0.00032982568,0.0010679858],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99866927,0.00021650373,0.00015498741,0.0004492779,0.00030089598,0.00020913097],"domain_scores_gemma":[0.99754,0.00079693703,0.00015180191,0.0007664332,0.0005073201,0.00023745732],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0011202913,0.0058531254,0.002304851,0.007353522,0.0019548973,0.0036174997,0.005376866,0.0036038358,0.16692896],"category_scores_gemma":[0.0066989404,0.0014889253,0.0018940016,0.007897997,0.000820907,0.0026583022,0.0027583214,0.0025948936,0.19584392],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006940034,0.000030038127,0.00020291317,0.0005335853,0.000023537812,0.000022570239,0.000019131763,0.00013921513,0.00022478549,0.00029676323,0.9954347,0.0030034345],"study_design_scores_gemma":[0.0006365343,0.000057495086,0.0018106113,0.00019651142,0.00004876432,0.00014705693,0.00010768759,0.0012867069,0.0019225843,0.0013315282,0.99239975,0.000054770604],"about_ca_topic_score_codex":0.023917772,"about_ca_topic_score_gemma":0.03102315,"teacher_disagreement_score":0.83307105,"about_ca_system_score_codex":0.002022366,"about_ca_system_score_gemma":0.0026110688,"threshold_uncertainty_score":0.558433},"labels":[],"label_agreement":null},{"id":"W4398747823","doi":"10.7910/dvn/66hucd/fl3zvg","title":"Turker Script for Scenario 92.docx","year":2019,"lang":"en","type":"dataset","venue":"Harvard Dataverse","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Natural language processing; Computer science; Artificial intelligence; Programming language; Linguistics","score_opus":0.021165662904531042,"score_gpt":0.2742400563038947,"score_spread":0.25307439339936366,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4398747823","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00013032898,0.000016538464,0.00014907234,0.00005052085,0.000032752818,0.000046641384,0.9963012,0.00170353,0.001569432],"genre_scores_gemma":[0.00046217468,0.000013395383,0.00045299082,0.000034993547,0.000009716312,0.00021556816,0.997626,0.00028153622,0.0009036113],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9985952,0.00027496362,0.00017339786,0.00039191506,0.00027435747,0.00029014927],"domain_scores_gemma":[0.9964457,0.0009320725,0.000203736,0.0013287524,0.0007559243,0.00033379722],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.0017144115,0.0032400263,0.0013350644,0.004347081,0.0014037163,0.0029818136,0.0032769376,0.0018650318,0.32676902],"category_scores_gemma":[0.008810143,0.0008909552,0.0016591658,0.0044977074,0.0006234561,0.0023066697,0.002669731,0.0022020997,0.3027542],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000034811503,0.00001550797,0.00017629124,0.00014649413,0.000008787569,0.000011857761,0.000011224785,0.000095556556,0.000043240776,0.00023804778,0.99813205,0.0010861756],"study_design_scores_gemma":[0.00039047934,0.000035599944,0.0022071314,0.00022496423,0.000020301075,0.000106121675,0.00017982951,0.001079223,0.000822208,0.0030846156,0.9918069,0.0000425883],"about_ca_topic_score_codex":0.012872816,"about_ca_topic_score_gemma":0.030361572,"teacher_disagreement_score":0.673231,"about_ca_system_score_codex":0.001491616,"about_ca_system_score_gemma":0.0023124458,"threshold_uncertainty_score":0.96028227},"labels":[],"label_agreement":null},{"id":"W4398751593","doi":"10.7910/dvn/eaxeet/wdhftj","title":"Corpus_ngram3_cleaned.csv","year":2019,"lang":"en","type":"dataset","venue":"Harvard Dataverse","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Natural language processing; Computer science","score_opus":0.0167009465814043,"score_gpt":0.26339928475765506,"score_spread":0.24669833817625075,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4398751593","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00043033255,0.00016173061,0.00028048886,0.00012160947,0.00007658462,0.000031476553,0.9927267,0.0043772836,0.0017938322],"genre_scores_gemma":[0.0006143626,0.000059441496,0.00071197376,0.00004004625,0.000013692342,0.00011296798,0.9967775,0.00042615295,0.0012439515],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99847966,0.0002882104,0.00016708855,0.00045558304,0.00036355882,0.0002459662],"domain_scores_gemma":[0.9972671,0.0007576825,0.00015182683,0.0009663865,0.00060254324,0.00025455805],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0015289765,0.004910557,0.0021433246,0.0066590616,0.002101307,0.0039234105,0.0054293657,0.0030171976,0.1284233],"category_scores_gemma":[0.00800466,0.0013226602,0.0018386382,0.007538713,0.0009669008,0.0022145293,0.0035751264,0.0027427007,0.18630785],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006511516,0.00002671602,0.0002251961,0.00045817316,0.000027854863,0.00002350799,0.00002541296,0.00013660592,0.00020033577,0.00038029358,0.9954353,0.0029953956],"study_design_scores_gemma":[0.00047954882,0.00003653607,0.0019087583,0.00017491462,0.000047090292,0.00011047077,0.00011094947,0.0009641673,0.0016318839,0.0015367119,0.9929508,0.000048295497],"about_ca_topic_score_codex":0.022844352,"about_ca_topic_score_gemma":0.035805892,"teacher_disagreement_score":0.87157667,"about_ca_system_score_codex":0.0017566268,"about_ca_system_score_gemma":0.0032052516,"threshold_uncertainty_score":0.42961878},"labels":[],"label_agreement":null},{"id":"W4398755567","doi":"10.7910/dvn/66hucd/2ghisq","title":"Turk-Talk-instruction-notes3.docx","year":2019,"lang":"en","type":"dataset","venue":"Harvard Dataverse","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Linguistics","score_opus":0.013442273266543387,"score_gpt":0.25717621254660517,"score_spread":0.2437339392800618,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4398755567","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0001758539,0.000027360327,0.000046995887,0.0000780498,0.00005121258,0.000020554335,0.9976623,0.00092050474,0.0010172306],"genre_scores_gemma":[0.00045420844,0.000022665008,0.00018939878,0.000041110954,0.000014805884,0.00009170429,0.99748695,0.0001337305,0.001565409],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9988356,0.0001763886,0.00013167426,0.00029140865,0.00025759923,0.00030739635],"domain_scores_gemma":[0.99702966,0.0005773576,0.00018928164,0.0010053089,0.0008259861,0.00037227874],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.001193729,0.003726942,0.0014149662,0.004118244,0.0013963139,0.003112288,0.003852486,0.0026948047,0.19889581],"category_scores_gemma":[0.0060711196,0.0007898325,0.0014592023,0.005015776,0.00073653104,0.0019032122,0.0026933914,0.0020402158,0.29614767],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000042991298,0.000021468657,0.00019271407,0.00016183007,0.0000064953597,0.000007997333,0.000010240125,0.00006325396,0.00004729837,0.00014666465,0.99811757,0.0011815681],"study_design_scores_gemma":[0.00045630606,0.000046533445,0.0028293713,0.00020419554,0.000019691106,0.000060470244,0.00018411815,0.0007543946,0.00092279655,0.0012630961,0.9932185,0.000040570863],"about_ca_topic_score_codex":0.024532698,"about_ca_topic_score_gemma":0.0479454,"teacher_disagreement_score":0.8011042,"about_ca_system_score_codex":0.0019977659,"about_ca_system_score_gemma":0.0026765943,"threshold_uncertainty_score":0.6653728},"labels":[],"label_agreement":null},{"id":"W4398805385","doi":"10.7910/dvn/sovpa4","title":"Replication Data for: Multi-label Prediction for Political Text-as-Data","year":2021,"lang":"en","type":"dataset","venue":"London School of Economics and Political Science Theses Online (London School of Economics and Political Science)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Replication (statistics); Computer science; Computational biology; Artificial intelligence; Biology; Virology","score_opus":0.0807064995258494,"score_gpt":0.37320601546287857,"score_spread":0.29249951593702916,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4398805385","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0019343903,0.00014301635,0.001259038,0.0005153828,0.0002336735,0.00012715833,0.99069655,0.003048049,0.0020426833],"genre_scores_gemma":[0.0022931923,0.00004264454,0.0027761166,0.00007991591,0.000023980807,0.00025010057,0.9930883,0.0001909618,0.0012547186],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99646384,0.001095587,0.00040691884,0.00081101706,0.0009324615,0.0002901923],"domain_scores_gemma":[0.9890676,0.0024180904,0.0007071651,0.004800754,0.0023504829,0.0006558603],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004384549,0.0021834415,0.0009915036,0.0030785745,0.0015999532,0.0023596012,0.0039461227,0.002452153,0.0363644],"category_scores_gemma":[0.018166156,0.0007272076,0.0013821484,0.0046429583,0.0009823106,0.0027091585,0.0029998359,0.00331004,0.06578612],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014099943,0.000121133904,0.0017690157,0.00034279187,0.000038344748,0.00006287923,0.00007117809,0.00071989547,0.0003260099,0.0010571287,0.9881535,0.0071971696],"study_design_scores_gemma":[0.00064206216,0.000096015174,0.008743748,0.00022071498,0.000036943966,0.00023080138,0.00034869483,0.007912727,0.0026346496,0.00462595,0.974414,0.00009363735],"about_ca_topic_score_codex":0.022976851,"about_ca_topic_score_gemma":0.04560216,"teacher_disagreement_score":0.0363644,"about_ca_system_score_codex":0.0019067598,"about_ca_system_score_gemma":0.0024780275,"threshold_uncertainty_score":0.12165105},"labels":[],"label_agreement":null},{"id":"W4399001331","doi":"10.7910/dvn/x8qjjv","title":"Corpus of North American Spoken English","year":2021,"lang":"en","type":"dataset","venue":"Harvard Dataverse","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"American English; Linguistics; Corpus linguistics; History; Natural language processing; Computer science; Philosophy","score_opus":0.009078403576490207,"score_gpt":0.23893571581011258,"score_spread":0.22985731223362238,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399001331","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013805628,0.00078972185,0.0013980184,0.0005046191,0.0002679759,0.0003502841,0.97248185,0.0013745268,0.009027483],"genre_scores_gemma":[0.004981909,0.00014494787,0.0014709666,0.000103782586,0.000031882755,0.0007470031,0.9896152,0.00014027863,0.0027639421],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9979717,0.0005952018,0.0002789245,0.0005206408,0.00043338168,0.00020004985],"domain_scores_gemma":[0.9966725,0.0008254517,0.00016103365,0.0006181705,0.0014463301,0.00027649666],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013370941,0.001629651,0.0011830267,0.0033267303,0.0024983955,0.001869582,0.002650565,0.0013327503,0.030682568],"category_scores_gemma":[0.0040768213,0.000533314,0.0006696258,0.0049514873,0.0009742458,0.0013785687,0.0027968518,0.002026684,0.04387107],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022787687,0.00015993738,0.0015363011,0.0010031628,0.000036674173,0.00037634306,0.00054063,0.00037174008,0.0012202978,0.0010881061,0.9765578,0.016881049],"study_design_scores_gemma":[0.00026372215,0.00004759902,0.024206273,0.00029188316,0.00005711038,0.0006195417,0.0014697531,0.0018073885,0.0016386791,0.0010507267,0.96845657,0.00009069971],"about_ca_topic_score_codex":0.04788924,"about_ca_topic_score_gemma":0.079890765,"teacher_disagreement_score":0.04788924,"about_ca_system_score_codex":0.0015990478,"about_ca_system_score_gemma":0.002858114,"threshold_uncertainty_score":0.10264343},"labels":[],"label_agreement":null},{"id":"W4399125980","doi":"10.1109/trustcom60117.2023.00347","title":"A Large-scale Non-standard English Database and Transformer-based Translation System","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Computer science; Natural language processing; Transformer; Readability; Artificial intelligence; Machine translation; Slang; Linguistics; Programming language","score_opus":0.012074384240633913,"score_gpt":0.2595191302919702,"score_spread":0.2474447460513363,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399125980","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09476096,0.00069140445,0.6455878,0.000803589,0.00049366197,0.0014491824,0.021003615,0.21999313,0.015216593],"genre_scores_gemma":[0.27774456,0.0004359088,0.6211239,0.0007899159,0.00012921427,0.00077122904,0.07872991,0.005769272,0.014506043],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99865544,0.00017834187,0.00021058465,0.0005628853,0.00031400129,0.00007866182],"domain_scores_gemma":[0.9979557,0.00043183172,0.000110426525,0.00061105314,0.0007287956,0.00016229373],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012338902,0.0010847693,0.0012380441,0.0028030856,0.00096035283,0.0015846421,0.002236414,0.0007685214,0.01229353],"category_scores_gemma":[0.003963072,0.00070061965,0.00079158094,0.002033666,0.000577884,0.003627937,0.0031229223,0.001067586,0.012733152],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010328176,0.0009902717,0.004302693,0.0011240485,0.00019631714,0.002448292,0.0013152817,0.0042352267,0.13350706,0.010033939,0.12398589,0.7168281],"study_design_scores_gemma":[0.0010563825,0.0010075402,0.009424072,0.00019803926,0.00050729816,0.0056355987,0.0028197335,0.39588293,0.3038658,0.017891005,0.26121476,0.000496893],"about_ca_topic_score_codex":0.00457584,"about_ca_topic_score_gemma":0.005203184,"teacher_disagreement_score":0.01229353,"about_ca_system_score_codex":0.00082965725,"about_ca_system_score_gemma":0.0020912248,"threshold_uncertainty_score":0.041125894},"labels":[],"label_agreement":null},{"id":"W4399158753","doi":"10.1075/kl.00008.par","title":"Word segmentation granularity in Korean","year":2024,"lang":"en","type":"article","venue":"Korean Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Granularity; Segmentation; Natural language processing; Word (group theory); Computer science; Text segmentation; Linguistics; Artificial intelligence; Philosophy","score_opus":0.014803721394184944,"score_gpt":0.2943217223496563,"score_spread":0.27951800095547136,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399158753","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7163243,0.0027573835,0.21354935,0.00064811774,0.0001664881,0.00022543332,0.0024241523,0.003017968,0.060886778],"genre_scores_gemma":[0.9280754,0.00046290344,0.0659352,0.00013879927,0.000018326824,0.00004625246,0.0014424701,0.00029757546,0.0035830888],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99930906,0.000118852986,0.000092734386,0.00026357322,0.00012177865,0.00009410863],"domain_scores_gemma":[0.9985656,0.00044426482,0.00024436906,0.00026335628,0.00040540777,0.00007700018],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00057830004,0.0003953405,0.00022728335,0.001857653,0.0006351416,0.0019773215,0.00039017346,0.00030272972,0.005339504],"category_scores_gemma":[0.0019751312,0.0003760624,0.00044181923,0.0022808893,0.0008854687,0.0029752702,0.0011476076,0.0005377221,0.0015378213],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001049042,0.00012752347,0.055086844,0.0011086079,0.00010157433,0.0019452955,0.014731662,0.012108243,0.16865484,0.14143655,0.00875942,0.5948905],"study_design_scores_gemma":[0.00011025936,0.00052015524,0.13924313,0.0007182343,0.00047126005,0.0031814952,0.017299594,0.121973954,0.21156664,0.18883522,0.3156051,0.00047496572],"about_ca_topic_score_codex":0.0028929105,"about_ca_topic_score_gemma":0.0033435186,"teacher_disagreement_score":0.005339504,"about_ca_system_score_codex":0.0007623952,"about_ca_system_score_gemma":0.00078899483,"threshold_uncertainty_score":0.01786244},"labels":[],"label_agreement":null},{"id":"W4399217282","doi":"10.14428/dvn/aauem2","title":"Core Metadata Schema for Learner Corpora (version 2)","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canarie","funders":"","keywords":"Metadata; Schema (genetic algorithms); Computer science; Information retrieval; Meta Data Services; Metadata repository; World Wide Web","score_opus":0.06591753522626567,"score_gpt":0.33272465975422877,"score_spread":0.26680712452796307,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399217282","genre_codex":"dataset","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0055119107,0.0008522656,0.36707956,0.0023867094,0.00094893295,0.0048095933,0.4403828,0.044185176,0.13384302],"genre_scores_gemma":[0.018944256,0.0011583306,0.25986484,0.0016205898,0.0002522197,0.005970436,0.6314587,0.014533253,0.06619737],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.994825,0.0011641195,0.0013036226,0.00057310105,0.0018231604,0.00031108933],"domain_scores_gemma":[0.98621327,0.0022781116,0.000836003,0.0037232942,0.00633912,0.0006100817],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008849333,0.00096396555,0.0009821946,0.005856306,0.0022744471,0.00756565,0.0026941844,0.0021801828,0.062876225],"category_scores_gemma":[0.018002316,0.001443923,0.0007709699,0.007204889,0.0011173941,0.009886887,0.0055643995,0.0031872476,0.09053168],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004244107,0.00019463607,0.0032228336,0.0017777932,0.000053167732,0.0003310412,0.0031798095,0.0018438225,0.008271477,0.11836271,0.7050194,0.15731882],"study_design_scores_gemma":[0.000026872453,0.000023955552,0.00081350486,0.000284678,0.000009956073,0.00019688987,0.0003704807,0.001036264,0.0030512977,0.008731587,0.98541,0.000044463406],"about_ca_topic_score_codex":0.012795318,"about_ca_topic_score_gemma":0.009880933,"teacher_disagreement_score":0.062876225,"about_ca_system_score_codex":0.002729074,"about_ca_system_score_gemma":0.008485969,"threshold_uncertainty_score":0.21034193},"labels":[],"label_agreement":null},{"id":"W4399264998","doi":"10.1007/s13042-024-02206-3","title":"DRA: dynamic routing attention for neural machine translation with low-resource languages","year":2024,"lang":"en","type":"article","venue":"International Journal of Machine Learning and Cybernetics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Computational intelligence; Computer science; Machine translation; Routing (electronic design automation); Artificial intelligence; Artificial neural network; Translation (biology); Resource (disambiguation); Natural language processing; Computer network","score_opus":0.005033158592483961,"score_gpt":0.28117202191372287,"score_spread":0.27613886332123894,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399264998","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.031018462,0.0016382454,0.8337514,0.0011767474,0.0013511872,0.00035917806,0.0026150416,0.116369285,0.01172039],"genre_scores_gemma":[0.36049232,0.00056921964,0.60534936,0.0012747004,0.0005602654,0.0005201908,0.006194766,0.005179664,0.019859511],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991697,0.00022428634,0.000046401397,0.00028899743,0.00013223747,0.00013840407],"domain_scores_gemma":[0.9988304,0.00054770464,0.000057315174,0.00030606845,0.0001846985,0.00007391523],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001056408,0.0016192733,0.0012808882,0.0013807681,0.0010863091,0.001718761,0.0024349147,0.0019963393,0.022682652],"category_scores_gemma":[0.0039783474,0.00073677424,0.0010280468,0.0015956872,0.00060145574,0.0026184008,0.0027984097,0.0024804026,0.008327886],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008201016,0.00034679731,0.0005118408,0.0004136476,0.00017846668,0.0003502891,0.00017994636,0.05142881,0.020443728,0.018648963,0.09683519,0.8098422],"study_design_scores_gemma":[0.00015549529,0.00018721572,0.0002519705,0.000031331736,0.000060332535,0.00012948689,0.00007190918,0.9364784,0.0132283615,0.036067314,0.013294956,0.00004323911],"about_ca_topic_score_codex":0.0070692454,"about_ca_topic_score_gemma":0.014640846,"teacher_disagreement_score":0.022682652,"about_ca_system_score_codex":0.0010628955,"about_ca_system_score_gemma":0.001596489,"threshold_uncertainty_score":0.075881004},"labels":[],"label_agreement":null},{"id":"W4399298003","doi":"10.1017/nlp.2024.5","title":"Calibration and context in human evaluation of machine translation","year":2024,"lang":"en","type":"article","venue":"Natural language processing.","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Calibration; Context (archaeology); Translation (biology); Machine translation; Computer science; Artificial intelligence; Natural language processing; Machine learning; Chemistry; Biology; Mathematics; Statistics; Biochemistry","score_opus":0.019952542382340694,"score_gpt":0.32952392468542374,"score_spread":0.30957138230308306,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399298003","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6657237,0.0113585945,0.27951953,0.0024298774,0.000580689,0.0014904718,0.00047263247,0.0013853206,0.037039302],"genre_scores_gemma":[0.9514693,0.00030714297,0.046351895,0.00034733795,0.000120580175,0.0004972448,0.00015809642,0.0001955351,0.000552889],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.6229384,0.33409697,0.00805994,0.013059717,0.020592835,0.0012521766],"domain_scores_gemma":[0.602674,0.29898867,0.03130128,0.02853955,0.035820194,0.0026762756],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.12338182,0.0012773442,0.0011317801,0.0031557623,0.0023810861,0.0051637734,0.0021042337,0.0021479414,0.0020985294],"category_scores_gemma":[0.38905817,0.0009923097,0.0006853218,0.0023246363,0.004251792,0.004760565,0.00757032,0.0022991695,0.00062380225],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.013265053,0.0010830809,0.121547,0.003837435,0.0027583987,0.00070531946,0.035930835,0.048922397,0.06267169,0.03181332,0.007383508,0.670082],"study_design_scores_gemma":[0.0015032112,0.008361804,0.29770315,0.005963982,0.002537868,0.002746942,0.013999833,0.2096984,0.17562464,0.21777008,0.062047374,0.0020426463],"about_ca_topic_score_codex":0.0013799558,"about_ca_topic_score_gemma":0.0020224468,"teacher_disagreement_score":0.87661815,"about_ca_system_score_codex":0.002286055,"about_ca_system_score_gemma":0.001582256,"threshold_uncertainty_score":0.65251327},"labels":[],"label_agreement":null},{"id":"W4399370190","doi":"10.3138/ctr.197.010","title":"<i>Some Must Watch While Some Must Sleep</i>: An Impromptu Text Thread","year":2024,"lang":"en","type":"article","venue":"Canadian Theatre Review","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Impromptu; Thread (computing); Sleep (system call); Visual arts; Performance art; Art; Art history; Psychology; Aesthetics; Computer science; Programming language","score_opus":0.017496054814235055,"score_gpt":0.2696887255823066,"score_spread":0.25219267076807156,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399370190","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022707306,0.13814968,0.04889703,0.32454023,0.040781297,0.0003454393,0.0009611215,0.0017060887,0.42191184],"genre_scores_gemma":[0.343448,0.109947816,0.02831709,0.0797757,0.019525157,0.00072459946,0.0021636176,0.0037252456,0.41237274],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99681705,0.0014097359,0.00013648905,0.00023016465,0.0011565549,0.0002500163],"domain_scores_gemma":[0.9861888,0.008718037,0.00063726766,0.0007069453,0.002558041,0.0011908299],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0055334377,0.00037180167,0.00031824704,0.00120162,0.0027542203,0.004519486,0.0015854903,0.001554731,0.013386954],"category_scores_gemma":[0.016638344,0.00022119925,0.00024304948,0.0014991116,0.006030703,0.0053105727,0.0023601602,0.0049621086,0.00450105],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016158333,0.00003567348,0.0003259858,0.0021781635,0.000016338041,0.00018898789,0.03251078,0.000093057395,0.002487944,0.084739394,0.6123942,0.264868],"study_design_scores_gemma":[0.000004956575,0.00002108553,0.0004841941,0.0008625369,0.0000057566403,0.00021769873,0.006062452,0.00007809013,0.0004995121,0.002134885,0.9896197,0.000009177267],"about_ca_topic_score_codex":0.02219356,"about_ca_topic_score_gemma":0.044736918,"teacher_disagreement_score":0.02219356,"about_ca_system_score_codex":0.0046187965,"about_ca_system_score_gemma":0.0053287367,"threshold_uncertainty_score":0.04478383},"labels":[],"label_agreement":null},{"id":"W4399488636","doi":"10.1007/s11049-023-09603-3","title":"Not all reconstruction effects are syntactic","year":2024,"lang":"en","type":"article","venue":"Natural Language & Linguistic Theory","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"McGill University","keywords":"Hindi; Mechanism (biology); Urdu; Argument (complex analysis); Computer science; Artificial intelligence; Natural language processing; Linguistics; Philosophy; Epistemology","score_opus":0.006094763569539101,"score_gpt":0.2723963261301065,"score_spread":0.2663015625605674,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399488636","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.58982265,0.0004369685,0.09960343,0.0018362267,0.00013755674,0.00006980749,0.00056914554,0.0014495291,0.30607474],"genre_scores_gemma":[0.99250925,0.000061315826,0.0034660955,0.0001801581,0.000027363454,0.000009701201,0.0000985785,0.00011578778,0.0035317845],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99912447,0.00018449123,0.000053699383,0.00022878165,0.0002997503,0.00010871394],"domain_scores_gemma":[0.99747294,0.0009168006,0.0002461246,0.00078577193,0.0005144124,0.00006389977],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00071285316,0.00037090937,0.0002866944,0.000576559,0.0007301617,0.0016434698,0.0005990958,0.00067754963,0.012031682],"category_scores_gemma":[0.002971469,0.00025227378,0.0002618119,0.00043213725,0.0035913305,0.003744718,0.0016425616,0.0012062351,0.0010512592],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024383419,0.000054281863,0.009030113,0.00030238624,0.00004635669,0.00077803805,0.003418047,0.00101582,0.038104787,0.85385406,0.002740575,0.09041169],"study_design_scores_gemma":[0.00011456543,0.00027631773,0.043047827,0.00013530695,0.00025665513,0.004463453,0.0033850395,0.010494272,0.07128372,0.7995372,0.06687554,0.00013014874],"about_ca_topic_score_codex":0.0014131214,"about_ca_topic_score_gemma":0.0013471602,"teacher_disagreement_score":0.012031682,"about_ca_system_score_codex":0.00080618815,"about_ca_system_score_gemma":0.00042213628,"threshold_uncertainty_score":0.040249944},"labels":[],"label_agreement":null},{"id":"W4399586015","doi":"10.1007/978-981-97-2958-6_9","title":"Plain Language in the Age of Neural Machine Translation: An Opportunity for Translators","year":2024,"lang":"en","type":"book-chapter","venue":"New frontiers in translation studies","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Machine translation; Translation (biology); Computer science; Linguistics; Natural language processing; Artificial intelligence; History; Philosophy; Chemistry","score_opus":0.07229075289294695,"score_gpt":0.3322333669470331,"score_spread":0.25994261405408614,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399586015","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013275904,0.13572931,0.3474612,0.13160053,0.011927583,0.00010592609,0.00061152125,0.0023840542,0.35690397],"genre_scores_gemma":[0.29083836,0.14130633,0.28732973,0.018837007,0.01876311,0.00044822315,0.0012545113,0.0025675572,0.23865514],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99830353,0.0010186997,0.00007892486,0.00014747032,0.00038119825,0.000070215705],"domain_scores_gemma":[0.9923149,0.005729521,0.00018783436,0.0009578817,0.0006264634,0.00018342739],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032511498,0.00053323293,0.00056172203,0.0012465549,0.0011481467,0.0064635295,0.0012366994,0.0023789576,0.014564049],"category_scores_gemma":[0.014174551,0.0004026134,0.00036301897,0.0018093875,0.004721373,0.020048851,0.0026198542,0.0036980112,0.005695596],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008543537,0.000052059808,0.00017677029,0.0005810558,0.000017357977,0.00017737586,0.0014150398,0.00077280955,0.0019835972,0.71968794,0.062266078,0.21278442],"study_design_scores_gemma":[0.000014074657,0.00004405049,0.00018851951,0.00036971667,0.000014948612,0.0002979896,0.00066316396,0.0039548413,0.001737951,0.6502027,0.34248897,0.000023113616],"about_ca_topic_score_codex":0.0008037454,"about_ca_topic_score_gemma":0.0012380447,"teacher_disagreement_score":0.014564049,"about_ca_system_score_codex":0.0012790157,"about_ca_system_score_gemma":0.0016578261,"threshold_uncertainty_score":0.048721552},"labels":[],"label_agreement":null},{"id":"W4399678864","doi":"10.1075/dt.24006.lom","title":"The rise of large language models informed by not so large corpora of training data","year":2024,"lang":"en","type":"article","venue":"Digital Translation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Standards Association","funders":"","keywords":"Training (meteorology); Computer science; Training set; Natural language processing; Artificial intelligence; Linguistics; Geography; Philosophy","score_opus":0.05178205753511441,"score_gpt":0.3195388255157452,"score_spread":0.2677567679806308,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399678864","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015887722,0.01359499,0.92497444,0.023368586,0.002055014,0.00017617964,0.005964939,0.007376247,0.0066019814],"genre_scores_gemma":[0.19338642,0.013865504,0.73871744,0.0074601932,0.0033353264,0.00081419986,0.028934395,0.0031334376,0.010353009],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9895102,0.0064902934,0.00035739725,0.0017486181,0.0017138378,0.00017969149],"domain_scores_gemma":[0.9059303,0.07315218,0.0013417641,0.01271843,0.006020216,0.0008371675],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.015872853,0.0017381352,0.0026460902,0.0034533497,0.0009734111,0.004898737,0.0032564895,0.0033546106,0.0063212137],"category_scores_gemma":[0.076271005,0.0022436867,0.0015572187,0.0045314655,0.0026498656,0.013949196,0.0046528527,0.008601792,0.0062789437],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012306673,0.0004681301,0.0059901113,0.0037173252,0.0012326848,0.0007239542,0.0008577607,0.11576013,0.019341476,0.09276563,0.12919568,0.6287164],"study_design_scores_gemma":[0.00014857925,0.00017733935,0.002007613,0.00058479834,0.00027124313,0.000373745,0.00032540952,0.6472107,0.007867011,0.24567881,0.09524623,0.00010855123],"about_ca_topic_score_codex":0.0030672492,"about_ca_topic_score_gemma":0.005226761,"teacher_disagreement_score":0.015872853,"about_ca_system_score_codex":0.0018546953,"about_ca_system_score_gemma":0.003281434,"threshold_uncertainty_score":0.08394468},"labels":[],"label_agreement":null},{"id":"W4399732990","doi":"10.22148/001c.116372","title":"Neither Corpus Nor Edition: Building a Pipeline to Make Data Analysis Possible on Medieval Arabic Commentary Traditions","year":2024,"lang":"en","type":"article","venue":"Journal of Cultural Analytics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Macro; Suite; Python (programming language); Macro level; Software; Representation (politics); Pipeline (software); Arabic; Natural language processing; Information retrieval; Artificial intelligence; Programming language; Linguistics; History","score_opus":0.04664510234599157,"score_gpt":0.3418212453957643,"score_spread":0.29517614304977274,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399732990","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02778457,0.000874502,0.48947993,0.0019824302,0.00051716284,0.0015384536,0.06993705,0.39302757,0.014858391],"genre_scores_gemma":[0.06663406,0.0005698718,0.7583896,0.0008426003,0.00017956205,0.0017996895,0.1318983,0.02743125,0.012255052],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9977118,0.00035605475,0.00018930242,0.0008975534,0.00064895244,0.00019633454],"domain_scores_gemma":[0.9953501,0.001630792,0.0002609639,0.001057646,0.0012457992,0.00045478594],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031985769,0.0017578657,0.0012102051,0.0049176966,0.0021038214,0.0042204205,0.001974647,0.0010301159,0.018719805],"category_scores_gemma":[0.014608132,0.0015562316,0.0018973878,0.0036156592,0.0015173597,0.006454154,0.0059469775,0.0030425931,0.02561128],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010065852,0.00039425754,0.0095816245,0.0016531327,0.00029263436,0.0011499696,0.0065809744,0.0051618316,0.049387917,0.014442249,0.3231137,0.58723515],"study_design_scores_gemma":[0.0003107877,0.00041592855,0.029217936,0.0005926526,0.0002127316,0.0012066797,0.004572414,0.1578067,0.09340655,0.061081093,0.6506872,0.00048938766],"about_ca_topic_score_codex":0.013230284,"about_ca_topic_score_gemma":0.015022725,"teacher_disagreement_score":0.018719805,"about_ca_system_score_codex":0.0016217437,"about_ca_system_score_gemma":0.0053665126,"threshold_uncertainty_score":0.06262392},"labels":[],"label_agreement":null},{"id":"W4399765771","doi":"10.32920/26052508.v1","title":"Discovering Related Terms and Detecting Trends in Software Engineering Using Word Embeddings","year":2024,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Word (group theory); Computer science; Software; Natural language processing; Data science; Artificial intelligence; Software engineering; Linguistics; Programming language; Philosophy","score_opus":0.011673111887770397,"score_gpt":0.27347946753105373,"score_spread":0.26180635564328336,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399765771","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.67158735,0.0035110463,0.30440837,0.0014190315,0.0002917391,0.00033105028,0.009963854,0.003062097,0.005425477],"genre_scores_gemma":[0.6693832,0.0016200369,0.3085425,0.00011568241,0.00010997785,0.00037533793,0.016828967,0.00023938496,0.00278497],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9987225,0.00034772026,0.0002513006,0.00032883757,0.0002700846,0.00007950959],"domain_scores_gemma":[0.99269694,0.0046757795,0.0009845194,0.00052580604,0.0009812565,0.00013557768],"candidate_categories":["bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.0014075388,0.00072321016,0.00039426016,0.008729736,0.0004919234,0.0017612635,0.00053073536,0.00085776707,0.0015360803],"category_scores_gemma":[0.009632642,0.0002984028,0.0007922261,0.00859915,0.00043089234,0.0037140432,0.0010752312,0.0008765207,0.00086453574],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00033295536,0.00042138938,0.07584992,0.0014887155,0.00023700783,0.0006952929,0.0037396604,0.029967017,0.02454514,0.010113761,0.012066385,0.84054273],"study_design_scores_gemma":[0.000117809956,0.0009282524,0.111114256,0.00057811505,0.00035262102,0.0017878783,0.0071383766,0.7361414,0.026358215,0.053151026,0.06217479,0.00015730111],"about_ca_topic_score_codex":0.0028353904,"about_ca_topic_score_gemma":0.0059533627,"teacher_disagreement_score":0.99127024,"about_ca_system_score_codex":0.0004943202,"about_ca_system_score_gemma":0.0007064857,"threshold_uncertainty_score":0.0074438453},"labels":[],"label_agreement":null},{"id":"W4399864810","doi":"10.5430/wjel.v14n5p515","title":"Challenges in Translation News Headlines: A Case of English Headlines Rendered into Arabic","year":2024,"lang":"en","type":"article","venue":"World Journal of English Language","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Arabic; Computer science; Translation (biology); Natural language processing; Linguistics; Artificial intelligence; Philosophy; Biology","score_opus":0.03202206710217024,"score_gpt":0.31129306442614724,"score_spread":0.279270997323977,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399864810","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9489568,0.0008906295,0.0076398244,0.010093073,0.00030800578,0.00027772685,0.0002598489,0.00025784862,0.031316243],"genre_scores_gemma":[0.9754647,0.0007388639,0.008643823,0.0016945003,0.00017665335,0.00015961158,0.00019398429,0.00026313993,0.012664725],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.99005365,0.0066482485,0.00053232646,0.0005461333,0.0015112472,0.0007083502],"domain_scores_gemma":[0.967474,0.023135765,0.0030567502,0.0012743433,0.003954463,0.0011046627],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00603521,0.0010948833,0.0005585241,0.0018287784,0.010490601,0.0061626276,0.0012826712,0.0044657863,0.0039618565],"category_scores_gemma":[0.029438645,0.00068568933,0.0005836202,0.0031826498,0.0053492654,0.0056721475,0.0023669403,0.003544795,0.0014801085],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00027569948,0.00017746307,0.0068993038,0.0006262107,0.0000214848,0.038321845,0.9136037,0.00066401163,0.0046781986,0.00626608,0.0071592242,0.021306809],"study_design_scores_gemma":[0.00004485021,0.00020940925,0.009340342,0.00042705575,0.000055765482,0.011921714,0.847176,0.0031454195,0.008462331,0.0025172657,0.1165918,0.00010800116],"about_ca_topic_score_codex":0.014157025,"about_ca_topic_score_gemma":0.021088539,"teacher_disagreement_score":0.014157025,"about_ca_system_score_codex":0.004788029,"about_ca_system_score_gemma":0.0025725537,"threshold_uncertainty_score":0.034739733},"labels":[],"label_agreement":null},{"id":"W4400027817","doi":"","title":"Measuring semantic specificities across corpora: looking for semantic shifts in Quebec English","year":2023,"lang":"en","type":"article","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Semantic similarity; Semantic change; Information retrieval","score_opus":0.0243168820608711,"score_gpt":0.24968619949234186,"score_spread":0.22536931743147076,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400027817","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.94407016,0.0029228374,0.008275522,0.0014754464,0.00010527102,0.00017127943,0.01946207,0.00054949574,0.02296786],"genre_scores_gemma":[0.96211827,0.0006759502,0.0057651903,0.0002690795,0.000035988774,0.00017240658,0.025549348,0.00027768215,0.005136141],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9975746,0.0007161367,0.00018301384,0.00073330087,0.0005013616,0.00029161113],"domain_scores_gemma":[0.98345596,0.0055889944,0.0011557156,0.0011986635,0.007942271,0.00065840007],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025223491,0.000669771,0.0006102898,0.0071557593,0.004218844,0.0033224104,0.00096301234,0.0008143255,0.0053563793],"category_scores_gemma":[0.014625822,0.00036585028,0.0003894349,0.009923147,0.0018575497,0.0031648534,0.0017492382,0.0012704567,0.0011190008],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016203918,0.0003299733,0.43132684,0.0028532187,0.0007370691,0.002460434,0.062418144,0.0058185123,0.08131782,0.012643269,0.086608745,0.31186563],"study_design_scores_gemma":[0.000056349643,0.00010740235,0.84975076,0.0004819022,0.00037724484,0.0007825601,0.037807766,0.013746336,0.010498024,0.002285854,0.08396253,0.00014329623],"about_ca_topic_score_codex":0.8125183,"about_ca_topic_score_gemma":0.91046554,"teacher_disagreement_score":0.1874817,"about_ca_system_score_codex":0.009845681,"about_ca_system_score_gemma":0.0064344495,"threshold_uncertainty_score":0.37717158},"labels":[],"label_agreement":null},{"id":"W4400055088","doi":"10.5430/wjel.v14n6p1","title":"The Use of Semi-automatic Annotation in Speech Acts Performed by Learners of English","year":2024,"lang":"en","type":"article","venue":"World Journal of English Language","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Annotation; Computer science; Natural language processing; Speech recognition; Artificial intelligence","score_opus":0.011685982314642284,"score_gpt":0.2647370985130471,"score_spread":0.2530511161984048,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400055088","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6818104,0.0007826522,0.29683244,0.0002930327,0.0002657396,0.0016642609,0.0023549523,0.0029068673,0.013089568],"genre_scores_gemma":[0.7446933,0.00031718277,0.24755229,0.00008736237,0.000070774455,0.0022439852,0.001860462,0.00019583617,0.0029789147],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.98776054,0.008441232,0.00087693136,0.0013775255,0.0013403052,0.00020338994],"domain_scores_gemma":[0.9477189,0.035363276,0.004021251,0.0042503364,0.007997223,0.00064898556],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007913267,0.00073876115,0.00044413688,0.0033853727,0.00073956914,0.0014323007,0.00092211424,0.0007158758,0.0016587614],"category_scores_gemma":[0.03376613,0.00027976144,0.00028324994,0.0014696703,0.00074650155,0.0015201133,0.0014277344,0.00055905274,0.0013656886],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012797453,0.00029190505,0.066659205,0.0021592067,0.00008952338,0.000696936,0.055995695,0.0016419658,0.1377005,0.0026079665,0.0029190877,0.72795826],"study_design_scores_gemma":[0.00021998775,0.0025665322,0.49753174,0.002001989,0.00035846198,0.0037530456,0.06585446,0.13282219,0.20065391,0.011115602,0.08209858,0.0010235013],"about_ca_topic_score_codex":0.001590168,"about_ca_topic_score_gemma":0.0025608551,"teacher_disagreement_score":0.007913267,"about_ca_system_score_codex":0.00036861442,"about_ca_system_score_gemma":0.0013779295,"threshold_uncertainty_score":0.04184985},"labels":[],"label_agreement":null},{"id":"W4400063360","doi":"10.1075/hts.2016.tra9","title":"Translation tools","year":2016,"lang":"en","type":"book-chapter","venue":"Handbook of translation studies","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Translation (biology); Computer science; Natural language processing; Biology; Genetics","score_opus":0.11334067744461528,"score_gpt":0.32474627473307854,"score_spread":0.21140559728846325,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400063360","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0018137037,0.01618969,0.49158812,0.0020010364,0.0024445467,0.00049321825,0.011939329,0.04106205,0.43246832],"genre_scores_gemma":[0.02397183,0.020113396,0.48405433,0.0016034597,0.0009548052,0.00089219003,0.050552793,0.018120108,0.39973715],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99828655,0.00039455024,0.00023424844,0.00031294947,0.0006553803,0.00011638716],"domain_scores_gemma":[0.9980039,0.00066716794,0.00006719272,0.0006725975,0.00052017096,0.000068938636],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015109861,0.0018787428,0.0013570323,0.005519216,0.0015981878,0.0067749345,0.0028321096,0.0015640619,0.17633219],"category_scores_gemma":[0.005607606,0.0012478663,0.0012716285,0.0075353393,0.0011868384,0.007193791,0.0037879744,0.0026176686,0.16987267],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000561064,0.00006173382,0.00011116626,0.0010233661,0.000028309256,0.00014599072,0.0004738418,0.00049620226,0.0027925519,0.08990977,0.2614703,0.6434307],"study_design_scores_gemma":[0.000015201188,0.000015829815,0.00013049254,0.0003435392,0.00001908674,0.0004387045,0.0001482153,0.0010839775,0.0031465127,0.042442136,0.9521944,0.000021826258],"about_ca_topic_score_codex":0.0011404076,"about_ca_topic_score_gemma":0.0015730672,"teacher_disagreement_score":0.17633219,"about_ca_system_score_codex":0.00097263523,"about_ca_system_score_gemma":0.0019462197,"threshold_uncertainty_score":0.58988994},"labels":[],"label_agreement":null},{"id":"W4400281600","doi":"10.32473/flairs.37.1.135596","title":"Decoding Complexity: A Mathematical Framework for Enhanced Translation Comprehension","year":2024,"lang":"en","type":"article","venue":"Proceedings of the ... International Florida Artificial Intelligence Research Society Conference","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Trois-Rivières","funders":"","keywords":"Decoding methods; Translation (biology); Computer science; Comprehension; Natural language processing; Theoretical computer science; Artificial intelligence; Cognitive science; Programming language; Psychology; Algorithm; Biology; Genetics","score_opus":0.27049055119903,"score_gpt":0.4406814160020632,"score_spread":0.1701908648030332,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400281600","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0055201557,0.0004461025,0.9810448,0.0012120438,0.00009645996,0.000056271572,0.00014442099,0.00019788412,0.011281906],"genre_scores_gemma":[0.32548153,0.0012159843,0.659814,0.000700181,0.0009114793,0.00057772506,0.0005177859,0.0004794056,0.010301941],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99434114,0.0025785153,0.000533207,0.0008060605,0.0014154211,0.0003256608],"domain_scores_gemma":[0.9824525,0.01215789,0.001444778,0.0017824651,0.0018982351,0.00026416878],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0064777136,0.0013640443,0.0010613509,0.004330984,0.0015133598,0.0066320403,0.0027137669,0.0023075445,0.006797223],"category_scores_gemma":[0.029177232,0.0008436084,0.0035838273,0.0025647203,0.0070828204,0.01568695,0.0040495736,0.0044305017,0.0022357672],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000019180525,0.000013798084,0.00018574622,0.00008578786,0.000015462969,0.00007189903,0.00030447435,0.012844358,0.0005764253,0.97455454,0.0007342536,0.010593942],"study_design_scores_gemma":[0.0000088387515,0.000025523217,0.000088125715,0.000027509675,0.000012901606,0.00007587664,0.000044680553,0.070575185,0.0004959204,0.92469364,0.0039297147,0.000022056296],"about_ca_topic_score_codex":0.0019569024,"about_ca_topic_score_gemma":0.0011890959,"teacher_disagreement_score":0.006797223,"about_ca_system_score_codex":0.0035389976,"about_ca_system_score_gemma":0.0018933078,"threshold_uncertainty_score":0.03425783},"labels":[],"label_agreement":null},{"id":"W4400284727","doi":"10.18413/2313-8912-2024-10-2-0-2","title":"Machine translation in hindsight","year":2024,"lang":"en","type":"article","venue":"RESEARCH RESULT Theoretical and Applied Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Predictability; Hindsight bias; Machine translation; Computer science; IBM; Artificial intelligence; Natural language processing; Operations research; Engineering; Mathematics; Statistics; Psychology","score_opus":0.029629528553337375,"score_gpt":0.3607876103260163,"score_spread":0.3311580817726789,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400284727","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005148284,0.009691026,0.40091684,0.020048458,0.013317626,0.0002314333,0.0030431564,0.008546831,0.5390563],"genre_scores_gemma":[0.30362064,0.008729406,0.33448115,0.013843018,0.010425461,0.0005047105,0.0082044955,0.0052134413,0.31497774],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9960657,0.0018120535,0.00029258966,0.00082957227,0.0008197538,0.00018036967],"domain_scores_gemma":[0.9933083,0.002666242,0.00039688524,0.0023956012,0.001066597,0.00016631767],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034102434,0.0010037611,0.0007185346,0.0016528668,0.0018059459,0.0057365126,0.001428926,0.0018883313,0.052750394],"category_scores_gemma":[0.015018585,0.00049318664,0.0005817187,0.0015789835,0.003058689,0.0054824897,0.0039110584,0.002710134,0.034586485],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011828086,0.000023176794,0.00051938515,0.0004118233,0.000039230064,0.00037134645,0.0009892076,0.0018379766,0.0010133354,0.5571922,0.17521901,0.26226497],"study_design_scores_gemma":[0.00002942882,0.000041585918,0.00030792254,0.00025019722,0.000013528629,0.00066121336,0.00024551564,0.004237209,0.0013080513,0.28077084,0.7121047,0.000029884191],"about_ca_topic_score_codex":0.0014838475,"about_ca_topic_score_gemma":0.0015501413,"teacher_disagreement_score":0.052750394,"about_ca_system_score_codex":0.001213463,"about_ca_system_score_gemma":0.0017812491,"threshold_uncertainty_score":0.17646766},"labels":[],"label_agreement":null},{"id":"W4400361588","doi":"10.69907/tbj.v1i1.68","title":"Bangla-Align: A forced-aligning toolkit for annotating Bangla speech","year":2024,"lang":"en","type":"article","venue":"TESOL Bangladesh journal","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Bengali; Computer science; Natural language processing; Artificial intelligence; Speech recognition; Programming language","score_opus":0.018127549648262402,"score_gpt":0.2985027038755927,"score_spread":0.2803751542273303,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400361588","genre_codex":"methods","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012074216,0.0012769446,0.7170287,0.00044120976,0.0010473676,0.0008525134,0.057470452,0.17740372,0.03240492],"genre_scores_gemma":[0.040863432,0.0004664385,0.827618,0.00043496696,0.0001184632,0.0012800848,0.07418671,0.032319993,0.022711841],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9975594,0.0005622421,0.00039696612,0.0008303458,0.0005063816,0.00014466944],"domain_scores_gemma":[0.99626726,0.0012879203,0.00025552275,0.0008692667,0.0011191323,0.00020085879],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024897156,0.0017839422,0.000972315,0.0019613823,0.0018417939,0.0027773557,0.0016733969,0.0011368954,0.06759978],"category_scores_gemma":[0.008181639,0.0012686799,0.001194086,0.0014927811,0.0007202001,0.0029423835,0.004057967,0.0020281582,0.064231515],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007929206,0.00012554713,0.0043007564,0.003207885,0.00022341403,0.0011273959,0.00442757,0.0019971041,0.10789728,0.007915467,0.29248247,0.57550216],"study_design_scores_gemma":[0.000109297136,0.00018604113,0.012268812,0.00048657646,0.00014160553,0.0033179668,0.0021216911,0.028710874,0.06356859,0.014038583,0.87459576,0.00045421545],"about_ca_topic_score_codex":0.0032843375,"about_ca_topic_score_gemma":0.007384358,"teacher_disagreement_score":0.06759978,"about_ca_system_score_codex":0.0006046794,"about_ca_system_score_gemma":0.0015422528,"threshold_uncertainty_score":0.22614378},"labels":[],"label_agreement":null},{"id":"W4400397332","doi":"","title":"Participation du CRIM à DEFT 2024 : Utilisation de petits modèles de Langue pour des QCMs dans le domaine médical","year":2024,"lang":"fr","type":"article","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Computer Research Institute of Montréal","funders":"","keywords":"Mod; Political science; Computer science; Artificial intelligence","score_opus":0.02346733149549102,"score_gpt":0.2704081809093147,"score_spread":0.24694084941382366,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400397332","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4064416,0.0018579826,0.44698012,0.01759447,0.0021550644,0.0016495467,0.01988223,0.048640817,0.054798108],"genre_scores_gemma":[0.59772944,0.0006798736,0.33339843,0.001255678,0.00020334312,0.00071553845,0.02898177,0.007301928,0.029733896],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9889587,0.0059839473,0.00046616938,0.0011360757,0.0029518302,0.0005032933],"domain_scores_gemma":[0.975182,0.014928189,0.0003819364,0.0032037632,0.005501728,0.00080252846],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0120691145,0.0013589911,0.0009387119,0.0016363701,0.0014454302,0.0038034227,0.002217684,0.0032151605,0.013540051],"category_scores_gemma":[0.042464785,0.0009627788,0.0019270978,0.0009202566,0.001352946,0.0043874825,0.0025709257,0.0030239453,0.0035421082],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.006420835,0.0024401417,0.02072331,0.0026108038,0.00059411617,0.0028481034,0.009506551,0.21725757,0.049271394,0.061467208,0.1684113,0.45844865],"study_design_scores_gemma":[0.00039719732,0.0006785495,0.004682398,0.0002535552,0.00015265917,0.00048558725,0.0016204396,0.7834531,0.040344708,0.009388829,0.15837763,0.00016535255],"about_ca_topic_score_codex":0.07118141,"about_ca_topic_score_gemma":0.059963565,"teacher_disagreement_score":0.07118141,"about_ca_system_score_codex":0.0046367547,"about_ca_system_score_gemma":0.007391268,"threshold_uncertainty_score":0.14153421},"labels":[],"label_agreement":null},{"id":"W4400477999","doi":"10.1017/9781108983624.002","title":"Experimental and Variationist Research on Heritage Languages","year":2024,"lang":"en","type":"book-chapter","venue":"Cambridge University Press eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Linguistics; Heritage language; Philosophy","score_opus":0.03692734001301878,"score_gpt":0.2937744914740275,"score_spread":0.2568471514610087,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400477999","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.35140982,0.018686164,0.01642396,0.008574699,0.0005861495,0.00020484837,0.00044717875,0.00007213848,0.603595],"genre_scores_gemma":[0.9191158,0.014468015,0.015401173,0.003721316,0.00038982977,0.00092093094,0.00048867427,0.0001371321,0.04535708],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99668175,0.0021775889,0.00011019861,0.0003821391,0.0005449952,0.00010336024],"domain_scores_gemma":[0.9859497,0.011909721,0.0004027103,0.0013082996,0.00028135075,0.00014819484],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004664631,0.00028378214,0.00029124896,0.0010878796,0.0013053468,0.00175064,0.0011660035,0.00075984467,0.009974234],"category_scores_gemma":[0.010635022,0.00023479575,0.00018542736,0.0015328629,0.009336099,0.0026781007,0.0018883258,0.0015866448,0.0006145715],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036647217,0.0014173074,0.0053992774,0.0011423117,0.0000719215,0.00033523998,0.028532822,0.0005299072,0.010831984,0.7706595,0.009718167,0.1709951],"study_design_scores_gemma":[0.0002531187,0.0013777118,0.059188556,0.001582985,0.00012309826,0.0016503036,0.030204173,0.0011891408,0.017466933,0.5048023,0.3820178,0.00014382423],"about_ca_topic_score_codex":0.0016121645,"about_ca_topic_score_gemma":0.0033572677,"teacher_disagreement_score":0.009974234,"about_ca_system_score_codex":0.0014581374,"about_ca_system_score_gemma":0.0007580294,"threshold_uncertainty_score":0.033367157},"labels":[],"label_agreement":null},{"id":"W4400478294","doi":"10.1017/9781108983624.009","title":"Working with Heritage Languages in Linguistics Classes","year":2024,"lang":"en","type":"book-chapter","venue":"Cambridge University Press eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Linguistics; Applied linguistics; Sociology; Philosophy","score_opus":0.020309910619629775,"score_gpt":0.22435562053670985,"score_spread":0.20404570991708007,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400478294","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09652595,0.0052804006,0.017023947,0.010053679,0.0018153784,0.0004707638,0.00070389255,0.0032371695,0.86488885],"genre_scores_gemma":[0.11382672,0.0033453524,0.024347087,0.002633206,0.0004270835,0.00026804337,0.0009394046,0.00097158155,0.8532415],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9989563,0.0002522222,0.000043827844,0.00023169546,0.0002814319,0.00023457623],"domain_scores_gemma":[0.9983529,0.00045312592,0.000060318114,0.00013252311,0.00017018348,0.0008309512],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013729229,0.00078594376,0.00065716746,0.0013123188,0.0050789574,0.006282003,0.0021003466,0.0015098231,0.07923314],"category_scores_gemma":[0.0020288706,0.0005257013,0.0006374424,0.0012232047,0.0014520821,0.004280852,0.004666604,0.0023934005,0.03880142],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012183299,0.0010542375,0.0044124797,0.0006126297,0.000010682464,0.0017474279,0.053847928,0.00037560248,0.008638314,0.02694218,0.2989102,0.60332656],"study_design_scores_gemma":[0.000013592909,0.00011324103,0.0033985062,0.0003100756,0.0000049824434,0.0010046795,0.016993675,0.00019296652,0.0017751819,0.0059633725,0.97020704,0.000022714094],"about_ca_topic_score_codex":0.0023994127,"about_ca_topic_score_gemma":0.012629373,"teacher_disagreement_score":0.07923314,"about_ca_system_score_codex":0.002741621,"about_ca_system_score_gemma":0.0034515467,"threshold_uncertainty_score":0.26506126},"labels":[],"label_agreement":null},{"id":"W4400488323","doi":"10.1080/00223891.2024.2375213","title":"Broader Issues in Test Translation and Validation: A Commentary Inspired by Macina et al. (2023)","year":2024,"lang":"en","type":"letter","venue":"Journal of Personality Assessment","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval; Université de Sherbrooke; Collège Laflèche; Université du Québec à Trois-Rivières","funders":"","keywords":"Psychology; Test (biology); Validation test; Test validity; Psychometrics; Clinical psychology; Ecology","score_opus":0.02943416655327385,"score_gpt":0.3528010667475053,"score_spread":0.3233669001942314,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400488323","genre_codex":"commentary","genre_gemma":"commentary","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"commentary","genre_consensus":"commentary","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.000055608947,0.0015476296,0.00010046861,0.9713147,0.026695566,0.000010397509,0.000026084277,0.000017094711,0.00023230267],"genre_scores_gemma":[0.00076124293,0.0010228318,0.0003100575,0.9706289,0.02650057,0.000059727798,0.000016645517,0.000031177715,0.00066886924],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.91176367,0.039903186,0.01509948,0.006826247,0.023640469,0.0027669317],"domain_scores_gemma":[0.6294129,0.26213107,0.009570272,0.0060728565,0.082649395,0.01016357],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.075628035,0.0017512792,0.0027980702,0.0027029044,0.008585514,0.0079764025,0.009443317,0.069550864,0.0034447745],"category_scores_gemma":[0.34384614,0.0016848936,0.003188728,0.002883578,0.019200655,0.011972988,0.0051047755,0.08560493,0.005101684],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000034341003,0.000018814715,0.00014744782,0.0001960426,0.00001026703,0.00023930382,0.0005534904,0.000027128868,0.00007176239,0.0019889278,0.98803073,0.008681834],"study_design_scores_gemma":[0.00012447378,0.00007286629,0.0009932357,0.0029816958,0.00004778364,0.0010772124,0.0015251234,0.00041467292,0.00028449218,0.0136996815,0.97864556,0.00013313955],"about_ca_topic_score_codex":0.03684233,"about_ca_topic_score_gemma":0.033199422,"teacher_disagreement_score":0.075628035,"about_ca_system_score_codex":0.016993608,"about_ca_system_score_gemma":0.028548216,"threshold_uncertainty_score":0.39996403},"labels":[],"label_agreement":null},{"id":"W4400488479","doi":"10.48550/arxiv.2407.06172","title":"On Speeding Up Language Model Evaluation","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Army Research Office; Division of Materials Research; Office of Naval Research; Multidisciplinary University Research Initiative; Materials Research Science and Engineering Center, Harvard University; Natural Sciences and Engineering Research Council of Canada; Cornell Center for Materials Research; Defense Advanced Research Projects Agency; National Institute of Food and Agriculture; National Science Foundation","keywords":"Computer science; Natural language processing","score_opus":0.0959286585888844,"score_gpt":0.25047465232152627,"score_spread":0.15454599373264188,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400488479","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0404929,0.00446063,0.8954949,0.0028114081,0.0005200093,0.0006213997,0.0012104384,0.045727212,0.008661109],"genre_scores_gemma":[0.23790807,0.00075856777,0.746025,0.0017280751,0.00028529656,0.0007293746,0.0031366914,0.004099443,0.0053294697],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.97908974,0.013673861,0.00097527937,0.0021300304,0.0033302698,0.0008008525],"domain_scores_gemma":[0.93902856,0.047948174,0.0014095531,0.006441703,0.004355982,0.0008160712],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.018665502,0.004188025,0.0026903462,0.0033026484,0.0016486864,0.004799916,0.0038786505,0.0036572167,0.019374054],"category_scores_gemma":[0.0950387,0.0015657977,0.0017920238,0.0024528122,0.0017957755,0.0090569155,0.0050825896,0.0060383882,0.011794016],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018367155,0.00077126984,0.004967382,0.0009448242,0.00038452566,0.00032041362,0.00041930497,0.18658875,0.009186564,0.022740701,0.04470708,0.72713244],"study_design_scores_gemma":[0.00023799023,0.0002064499,0.00056992157,0.000115173156,0.00007226513,0.00015442767,0.00015542083,0.95595264,0.00449151,0.031006088,0.0069958353,0.000042299504],"about_ca_topic_score_codex":0.008182959,"about_ca_topic_score_gemma":0.017054243,"teacher_disagreement_score":0.019374054,"about_ca_system_score_codex":0.0025376955,"about_ca_system_score_gemma":0.0044332664,"threshold_uncertainty_score":0.098713815},"labels":[],"label_agreement":null},{"id":"W4400510802","doi":"10.5430/wjel.v14n5p535","title":"A Comparative Study of the Error-Detection Accuracy of Grammarly and Microsoft Word Editor in Formal English Writing","year":2024,"lang":"en","type":"article","venue":"World Journal of English Language","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"Qassim University","keywords":"Computer science; Word (group theory); Microsoft excel; Natural language processing; Artificial intelligence; Linguistics; Operating system; Philosophy","score_opus":0.011045257708832064,"score_gpt":0.28913889810009114,"score_spread":0.27809364039125906,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400510802","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98297805,0.0013743168,0.009033479,0.00024162268,0.00008483664,0.00012780895,0.00026336426,0.0011998098,0.0046966574],"genre_scores_gemma":[0.9781698,0.0007566307,0.0183169,0.00012435598,0.000043548396,0.00007289674,0.0005196714,0.00038454257,0.0016117403],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9611862,0.01985003,0.006413385,0.003702749,0.008165118,0.00068252446],"domain_scores_gemma":[0.5005694,0.4178627,0.02843828,0.013311681,0.037437916,0.0023800929],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.025247175,0.0007162921,0.0012418282,0.0056466777,0.0006555632,0.0036411972,0.0012408764,0.0009130704,0.0010231278],"category_scores_gemma":[0.27799127,0.00046993166,0.0004525054,0.0032370877,0.0012949883,0.003879277,0.0021530688,0.00089184067,0.0008025094],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0030207578,0.0007342392,0.2726664,0.0020447415,0.00061283243,0.00069561467,0.050771296,0.002363305,0.013830065,0.0009741664,0.0027080919,0.64957845],"study_design_scores_gemma":[0.00031270014,0.0052016173,0.8241732,0.0016494012,0.0014165415,0.0047844993,0.03284045,0.045777638,0.05787706,0.003555684,0.021775758,0.00063549646],"about_ca_topic_score_codex":0.00255792,"about_ca_topic_score_gemma":0.0040808623,"teacher_disagreement_score":0.025247175,"about_ca_system_score_codex":0.00066006643,"about_ca_system_score_gemma":0.0011161211,"threshold_uncertainty_score":0.13352138},"labels":[],"label_agreement":null},{"id":"W4400526322","doi":"10.1145/3626772.3657952","title":"On Backbones and Training Regimes for Dense Retrieval in African Languages","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Training (meteorology); Computer science; Natural language processing; Artificial intelligence; Geography; Meteorology","score_opus":0.020338555014422575,"score_gpt":0.304658782273631,"score_spread":0.28432022725920847,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400526322","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5434727,0.01062036,0.39548618,0.0038210223,0.00043471236,0.0006226195,0.005205717,0.018460594,0.02187611],"genre_scores_gemma":[0.8181439,0.002121192,0.15826018,0.0009230501,0.00020683516,0.00046128614,0.013561432,0.0011582762,0.005163766],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9971468,0.0015162937,0.00023813346,0.00058961654,0.0002608338,0.00024842905],"domain_scores_gemma":[0.99133575,0.0056270156,0.000330448,0.0018601844,0.00063266535,0.00021386032],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0076253293,0.0013051697,0.0010745458,0.0017311547,0.0012015284,0.0017702591,0.0015654274,0.0016158796,0.0047871936],"category_scores_gemma":[0.021937579,0.0006556642,0.0010401,0.0014697207,0.0014246044,0.006231873,0.0033204437,0.0029335665,0.0042919624],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019872016,0.0008828722,0.013506734,0.0010940407,0.00033895357,0.0004137554,0.0011903213,0.19028583,0.02275395,0.0128758205,0.035995886,0.71867466],"study_design_scores_gemma":[0.0003986415,0.00076473167,0.004776386,0.00033568116,0.00019808533,0.0005463024,0.0009084308,0.92287254,0.025601605,0.027943801,0.015556558,0.00009724161],"about_ca_topic_score_codex":0.008901303,"about_ca_topic_score_gemma":0.012169362,"teacher_disagreement_score":0.008901303,"about_ca_system_score_codex":0.0010595042,"about_ca_system_score_gemma":0.0014710762,"threshold_uncertainty_score":0.040327072},"labels":[],"label_agreement":null},{"id":"W4400528580","doi":"10.1145/3626772.3657884","title":"CIRAL: A Test Collection for CLIR Evaluations in African Languages","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada); University of Waterloo","funders":"Universitas Brawijaya","keywords":"Computer science; Natural language processing; Test (biology); Artificial intelligence; Information retrieval","score_opus":0.02130701428789182,"score_gpt":0.36274277122373216,"score_spread":0.34143575693584033,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400528580","genre_codex":"empirical","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4579515,0.0129741635,0.0809793,0.0035569742,0.002618295,0.024090264,0.26699123,0.054004833,0.09683345],"genre_scores_gemma":[0.2765658,0.0017559573,0.17363358,0.0015723226,0.0006692875,0.013296094,0.50172883,0.006255684,0.02452242],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99087286,0.0046327873,0.00096962287,0.001047245,0.0019767159,0.0005007159],"domain_scores_gemma":[0.9753642,0.009530596,0.0008899773,0.0047256425,0.00813456,0.0013550612],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00834597,0.0019946566,0.0012494077,0.0067629735,0.003087886,0.0021081273,0.002998533,0.0018185499,0.014700952],"category_scores_gemma":[0.030061107,0.0006549247,0.0010557143,0.004471567,0.0018530972,0.0040265913,0.004538723,0.0019272795,0.013234145],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0027307568,0.0027886874,0.008948996,0.009326117,0.00024209954,0.0017620348,0.0071179983,0.003343219,0.042576376,0.004057298,0.6233329,0.2937734],"study_design_scores_gemma":[0.0022413512,0.0032235973,0.07499411,0.0016032357,0.00041441238,0.0047443905,0.009321927,0.024750687,0.06420666,0.0033903108,0.8103575,0.0007518885],"about_ca_topic_score_codex":0.009930022,"about_ca_topic_score_gemma":0.015172131,"teacher_disagreement_score":0.014700952,"about_ca_system_score_codex":0.001456468,"about_ca_system_score_gemma":0.0029923038,"threshold_uncertainty_score":0.049179554},"labels":[],"label_agreement":null},{"id":"W4400531953","doi":"10.1145/3626772.3657878","title":"C-Pack: Packed Resources For General Chinese Embeddings","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":269,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Packed bed; Computer science; Chemistry; Chromatography","score_opus":0.0077241796640625335,"score_gpt":0.297602640136452,"score_spread":0.28987846047238947,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400531953","genre_codex":"software","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0135748945,0.002095682,0.29578584,0.0011508387,0.0017899217,0.0010462109,0.2654861,0.3832396,0.03583102],"genre_scores_gemma":[0.042400282,0.0011008439,0.25622448,0.00095305266,0.00024095233,0.0026200102,0.6304893,0.04987352,0.016097518],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99760866,0.00043729984,0.00030254325,0.0005718104,0.0007512415,0.00032840975],"domain_scores_gemma":[0.9941643,0.0010892423,0.00018587145,0.00295329,0.0012792295,0.00032813346],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025414506,0.004819568,0.0016977176,0.004080434,0.0011573375,0.0027994185,0.005478612,0.0016123418,0.079639785],"category_scores_gemma":[0.021828145,0.0020987152,0.0021354894,0.0060536647,0.00095862884,0.008683732,0.0074778902,0.00336383,0.085750334],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00040090855,0.00020623939,0.0018780647,0.0011498542,0.00010570963,0.00022725333,0.00018937145,0.009778254,0.0035500256,0.012598538,0.795925,0.1739907],"study_design_scores_gemma":[0.0005327882,0.00035890928,0.0030546517,0.0004608603,0.00011362967,0.0006574704,0.00025801896,0.17065288,0.032967456,0.06097267,0.7296016,0.0003689622],"about_ca_topic_score_codex":0.010349185,"about_ca_topic_score_gemma":0.0180387,"teacher_disagreement_score":0.079639785,"about_ca_system_score_codex":0.0014414658,"about_ca_system_score_gemma":0.00406317,"threshold_uncertainty_score":0.26642168},"labels":[],"label_agreement":null},{"id":"W4400645859","doi":"10.1109/iv55156.2024.10588846","title":"SSL-Interactions: Pretext Tasks for Interactive Trajectory Prediction","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Pretext; Computer science; Trajectory; Human–computer interaction; Artificial intelligence","score_opus":0.014944216953087916,"score_gpt":0.30847650843581903,"score_spread":0.29353229148273113,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400645859","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.057065193,0.00057122513,0.89074993,0.00057146215,0.00021405566,0.00040911193,0.0056130816,0.041121997,0.0036840201],"genre_scores_gemma":[0.45197985,0.00020997098,0.51830804,0.00043375915,0.000188963,0.0008930927,0.021555323,0.0016721193,0.0047589494],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9981014,0.0006020796,0.00012426404,0.00057626737,0.00043972235,0.00015625554],"domain_scores_gemma":[0.9944613,0.0030799422,0.00036115004,0.0012559908,0.000498243,0.00034346426],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019597393,0.002311623,0.0010670049,0.0014316672,0.0007887336,0.0013632448,0.003052617,0.002327325,0.0068177916],"category_scores_gemma":[0.011430597,0.0005914679,0.0013419893,0.0010694957,0.000836011,0.003850883,0.0035612027,0.0030642413,0.0037734932],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018974248,0.0009178436,0.013525696,0.0005458546,0.00020656947,0.00039057896,0.00055606675,0.4060683,0.010699573,0.009307221,0.043998722,0.51188624],"study_design_scores_gemma":[0.00002745452,0.00010593805,0.00063395145,0.000015676942,0.0000075880344,0.000036546953,0.00005525677,0.9883691,0.002720748,0.0047644014,0.0032457027,0.000017632346],"about_ca_topic_score_codex":0.0067830314,"about_ca_topic_score_gemma":0.013487353,"teacher_disagreement_score":0.0068177916,"about_ca_system_score_codex":0.0010737083,"about_ca_system_score_gemma":0.0016432295,"threshold_uncertainty_score":0.022807777},"labels":[],"label_agreement":null},{"id":"W4400655051","doi":"10.5267/j.ijdns.2024.5.009","title":"Assessing the accuracy of MT and AI tools in translating humanities or social sciences Arabic research titles into English: Evidence from Google Translate, Gemini, and ChatGPT","year":2024,"lang":"en","type":"article","venue":"International Journal of Data and Network Science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Diction; Arabic; Computer science; Natural language processing; Linguistics; Syntax; Equivalence (formal languages); Artificial intelligence; Philosophy","score_opus":0.23886364574547767,"score_gpt":0.4978445780223662,"score_spread":0.25898093227688856,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400655051","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9659149,0.0051406007,0.0058044633,0.0016598755,0.0002198704,0.0003343892,0.0011991089,0.00017945711,0.019547377],"genre_scores_gemma":[0.98121923,0.0030264386,0.011637495,0.000531514,0.00015353494,0.00017470615,0.0016747375,0.00015911029,0.0014232335],"study_design_codex":"observational","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9548367,0.0257288,0.0055048144,0.0032505398,0.009939056,0.0007401733],"domain_scores_gemma":[0.62647384,0.26569283,0.032642927,0.020553228,0.052170638,0.0024665669],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.050689213,0.0009447576,0.0009001563,0.0061093,0.0019878612,0.0052729915,0.001556181,0.0017179062,0.0020002488],"category_scores_gemma":[0.2797272,0.00049758615,0.0008205477,0.007924259,0.0033850714,0.008398765,0.003421686,0.0016156466,0.0023671968],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004550073,0.0011655375,0.43227488,0.008160282,0.0009320286,0.0017064164,0.107485235,0.0032498061,0.0062373136,0.0042423294,0.0070090853,0.422987],"study_design_scores_gemma":[0.00057599816,0.0057294467,0.741522,0.007051697,0.0026728504,0.0054965746,0.089660436,0.029382959,0.03159497,0.0074656485,0.07824175,0.0006056204],"about_ca_topic_score_codex":0.007619277,"about_ca_topic_score_gemma":0.0076899943,"teacher_disagreement_score":0.050689213,"about_ca_system_score_codex":0.0013511492,"about_ca_system_score_gemma":0.0023720937,"threshold_uncertainty_score":0.26807332},"labels":[],"label_agreement":null},{"id":"W4400680832","doi":"10.1109/saner60148.2024.00021","title":"OppropBERL: A GNN and BERT-Style Reinforcement Learning-Based Type Inference","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Government of Ontario","keywords":"Reinforcement learning; Computer science; Style (visual arts); Inference; Artificial intelligence; Machine learning; Art","score_opus":0.014250119774445917,"score_gpt":0.2870010820885919,"score_spread":0.272750962314146,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400680832","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00699498,0.00020023825,0.96251553,0.00041293507,0.00016198553,0.00014457337,0.0004932488,0.02442934,0.00464708],"genre_scores_gemma":[0.26576936,0.00024335778,0.71355015,0.0012388612,0.000101841,0.0005282829,0.001616413,0.003067024,0.013884784],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991283,0.00016304167,0.000047795987,0.00032241363,0.00022710543,0.00011143125],"domain_scores_gemma":[0.99825937,0.00078986917,0.00014168881,0.00039980168,0.000306254,0.00010286987],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017685917,0.001362738,0.0010174264,0.00085887604,0.0005009892,0.0012851335,0.0049882093,0.0018744835,0.010788973],"category_scores_gemma":[0.0068501877,0.0011613596,0.0013734873,0.00054730475,0.0013710173,0.0024121695,0.0022302258,0.0037017039,0.0032311727],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004572087,0.00030138975,0.0043135732,0.00041103578,0.00017813516,0.00033189,0.00014770149,0.6054396,0.0069586216,0.035611477,0.022729157,0.32312012],"study_design_scores_gemma":[0.000023219556,0.000021439322,0.00008194927,0.00001554899,0.000011748037,0.000027172675,0.0000037592906,0.9873524,0.0014858539,0.008504518,0.002463529,0.000008929937],"about_ca_topic_score_codex":0.011956951,"about_ca_topic_score_gemma":0.021011803,"teacher_disagreement_score":0.011956951,"about_ca_system_score_codex":0.001944519,"about_ca_system_score_gemma":0.0027099794,"threshold_uncertainty_score":0.0360927},"labels":[],"label_agreement":null},{"id":"W4400898279","doi":"10.1111/jedm.12406","title":"Using Automated Procedures to Score Educational Essays Written in Three Languages","year":2024,"lang":"en","type":"article","venue":"Journal of Educational Measurement","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Mathematics education; Natural language processing; Computer science; Psychology; Linguistics; Artificial intelligence; Philosophy","score_opus":0.0603400789110201,"score_gpt":0.3634531269538653,"score_spread":0.30311304804284517,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400898279","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7465224,0.00038071183,0.22352445,0.00025225722,0.00020959828,0.001177902,0.00218063,0.011978339,0.013773814],"genre_scores_gemma":[0.818852,0.0001523614,0.17244616,0.00006588506,0.00007373051,0.00044008764,0.0021063606,0.0002581722,0.005605296],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99204636,0.0033618375,0.0008698298,0.0012341252,0.0022627325,0.00022518815],"domain_scores_gemma":[0.9688612,0.013045206,0.003167753,0.0021006821,0.011953738,0.0008714346],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0051577147,0.0010326157,0.00066085794,0.0036925934,0.00049697276,0.002299851,0.001045723,0.00060586317,0.0041193063],"category_scores_gemma":[0.03298691,0.00024844366,0.0003722198,0.0014650179,0.00036991216,0.00129249,0.0015576427,0.0006799394,0.0026914722],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008439936,0.0005127,0.040037744,0.00046916495,0.0001557472,0.00033054658,0.0013105095,0.0066188076,0.046464965,0.00097355363,0.004463994,0.8978182],"study_design_scores_gemma":[0.0005791047,0.0028786703,0.2662421,0.0004202122,0.0003353503,0.0021288912,0.005289431,0.47172955,0.21026033,0.0058657955,0.033754904,0.00051570777],"about_ca_topic_score_codex":0.0022502409,"about_ca_topic_score_gemma":0.004657552,"teacher_disagreement_score":0.0051577147,"about_ca_system_score_codex":0.00061444734,"about_ca_system_score_gemma":0.001010972,"threshold_uncertainty_score":0.027276874},"labels":[],"label_agreement":null},{"id":"W4400900006","doi":"","title":"Augmentation de jeux de données RI pour la recherche conversationnelle à initiative mixte","year":2023,"lang":"fr","type":"article","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; Computer Research Institute of Montréal","funders":"Sorbonne Université; Agence Nationale de la Recherche","keywords":"Computer science","score_opus":0.12678480537589584,"score_gpt":0.32339088209794864,"score_spread":0.1966060767220528,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400900006","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.102602586,0.0019585858,0.8514645,0.0011945489,0.0003393097,0.0011829514,0.0018320942,0.025613135,0.013812283],"genre_scores_gemma":[0.286497,0.0007205951,0.6861478,0.00043658752,0.00011119818,0.0012957935,0.0031951205,0.0022623278,0.019333573],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9836373,0.0052838833,0.0013490977,0.0036160804,0.005521294,0.00059247355],"domain_scores_gemma":[0.9681898,0.016543195,0.0009899334,0.0061074486,0.0073329476,0.0008367472],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009437096,0.0018906189,0.0016850333,0.0023229725,0.0015397165,0.004159237,0.0022100476,0.0017144964,0.009001706],"category_scores_gemma":[0.03340595,0.0010443814,0.0018974284,0.001515028,0.0009903706,0.005701388,0.0041621015,0.002837037,0.0034820095],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001864938,0.0006424427,0.013814868,0.0024692395,0.0006227823,0.00074914127,0.01275517,0.02164329,0.15487854,0.01070349,0.011258003,0.7685981],"study_design_scores_gemma":[0.00034844,0.0024109117,0.029642338,0.0009369013,0.0013392721,0.0015710269,0.0087551335,0.33856055,0.287423,0.019247027,0.3089876,0.00077788153],"about_ca_topic_score_codex":0.0111532975,"about_ca_topic_score_gemma":0.013829543,"teacher_disagreement_score":0.0111532975,"about_ca_system_score_codex":0.0013552255,"about_ca_system_score_gemma":0.0036866176,"threshold_uncertainty_score":0.049908757},"labels":[],"label_agreement":null},{"id":"W4400909634","doi":"10.1109/icde60146.2024.00430","title":"RAGE Against the Machine: Retrieval-Augmented LLM Explanations","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University; University of Waterloo","funders":"","keywords":"Computer science; Rage (emotion); Artificial intelligence; Machine learning; Information retrieval; Psychology; Neuroscience","score_opus":0.010793218787814326,"score_gpt":0.27239742964937036,"score_spread":0.26160421086155605,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400909634","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011486186,0.0006325384,0.9579782,0.0018154401,0.00008842763,0.00012348683,0.0009444063,0.02038007,0.006551148],"genre_scores_gemma":[0.23512222,0.00058773503,0.7533535,0.00071765744,0.000100927944,0.00018899821,0.0019167697,0.002178161,0.00583408],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9966756,0.0018636959,0.00012214959,0.00046855322,0.00074487145,0.00012521299],"domain_scores_gemma":[0.98523074,0.012280098,0.00045945877,0.0014375276,0.00045363195,0.00013860757],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036671252,0.0013987329,0.00072046235,0.0018809304,0.0008121031,0.0032977688,0.0025780199,0.002820556,0.020849766],"category_scores_gemma":[0.023516307,0.00070818176,0.0015274552,0.0007475074,0.0020900469,0.007634977,0.0054067,0.002467077,0.0024929363],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00096371747,0.00023360885,0.0034354657,0.0018345665,0.00026468677,0.0031304979,0.009803162,0.09977159,0.016420662,0.42144346,0.04765369,0.39504483],"study_design_scores_gemma":[0.00021966675,0.00014168555,0.0007739956,0.00045229506,0.00017023025,0.0010855998,0.0011136979,0.4243103,0.021577112,0.38727874,0.16270617,0.00017046784],"about_ca_topic_score_codex":0.0019996073,"about_ca_topic_score_gemma":0.004283554,"teacher_disagreement_score":0.020849766,"about_ca_system_score_codex":0.0009704609,"about_ca_system_score_gemma":0.0009940536,"threshold_uncertainty_score":0.069749415},"labels":[],"label_agreement":null},{"id":"W4401008167","doi":"","title":"La traduction automatique neuronale du chinois au français pour l'édition","year":2023,"lang":"fr","type":"article","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Art","score_opus":0.013048992610165172,"score_gpt":0.2346667076799369,"score_spread":0.2216177150697717,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401008167","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17022188,0.021261593,0.34303957,0.012624552,0.011792907,0.00029769956,0.0068862103,0.010626332,0.42324924],"genre_scores_gemma":[0.30413094,0.009017332,0.12388756,0.0007343259,0.0010597957,0.00017329522,0.0037330256,0.0027478002,0.55451596],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99912745,0.00015627765,0.000059709335,0.00021623401,0.00036033388,0.000080117694],"domain_scores_gemma":[0.99885607,0.00020911246,0.000038490525,0.00018973889,0.00065434986,0.000052311927],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001071719,0.0009394367,0.00040266002,0.0017699148,0.0015055442,0.004293947,0.00063635496,0.0010591969,0.027148085],"category_scores_gemma":[0.002413974,0.00031803834,0.00083315535,0.0020893945,0.0012782442,0.001798572,0.0010286552,0.0014918732,0.007883615],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00031917292,0.000057412813,0.004222799,0.00055392494,0.00014198279,0.0009701403,0.0030756711,0.007000145,0.033583947,0.13872753,0.093260415,0.71808684],"study_design_scores_gemma":[0.00003228622,0.00006742092,0.008912965,0.00021651556,0.0000765172,0.0007520426,0.0009047325,0.016313678,0.017637316,0.021413988,0.9335864,0.00008625339],"about_ca_topic_score_codex":0.11050159,"about_ca_topic_score_gemma":0.1307075,"teacher_disagreement_score":0.11050159,"about_ca_system_score_codex":0.0033689765,"about_ca_system_score_gemma":0.0032161297,"threshold_uncertainty_score":0.21971679},"labels":[],"label_agreement":null},{"id":"W4401013151","doi":"10.2307/jj.17681834.13","title":"Canadian Poetry and the Computational Concordance:","year":2023,"lang":"en","type":"book-chapter","venue":"Les Presses de l’Université d’Ottawa | University of Ottawa Press eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Concordance; Poetry; Computer science; History; Literature; Art; Medicine; Internal medicine","score_opus":0.011561627118923492,"score_gpt":0.19715878186472213,"score_spread":0.18559715474579863,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401013151","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014222677,0.0063582547,0.005762017,0.014324197,0.00062601117,0.0000250649,0.000567594,0.00009989863,0.9580142],"genre_scores_gemma":[0.82993335,0.006140792,0.008929258,0.0013405982,0.00041679438,0.00004645861,0.00082623295,0.00019208943,0.15217443],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99842453,0.00046073782,0.00004704875,0.0002235455,0.0006014095,0.00024281212],"domain_scores_gemma":[0.99784076,0.000925659,0.00007877317,0.0002805359,0.00070928386,0.00016497361],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010827241,0.00042009598,0.00055202836,0.0032160287,0.010007404,0.009389152,0.0014832513,0.0015523753,0.026640967],"category_scores_gemma":[0.00836017,0.0004202476,0.00031115822,0.009175594,0.01364551,0.004978881,0.0023303884,0.0025493598,0.0014830403],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000011340285,0.0000030023443,0.00017375757,0.000025078578,0.0000026559908,0.00003647809,0.0017966928,0.00016862407,0.000020370626,0.97183675,0.014139721,0.011785472],"study_design_scores_gemma":[0.00001821159,0.0000072012335,0.0027428647,0.00015262591,0.000017659464,0.00016880476,0.004956877,0.0012871828,0.00018005917,0.5637045,0.42672315,0.00004089336],"about_ca_topic_score_codex":0.86266226,"about_ca_topic_score_gemma":0.90328246,"teacher_disagreement_score":0.13733774,"about_ca_system_score_codex":0.027996872,"about_ca_system_score_gemma":0.034449138,"threshold_uncertainty_score":0.27629304},"labels":[],"label_agreement":null},{"id":"W4401041961","doi":"10.18653/v1/2024.starsem-1.10","title":"Lexical Substitution as Causal Language Modeling","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Machine Intelligence Institute","keywords":"Substitution (logic); Computer science; Natural language processing; Artificial intelligence; Linguistics; Programming language; Philosophy","score_opus":0.01690305588437108,"score_gpt":0.3108748777878408,"score_spread":0.2939718219034697,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401041961","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007599977,0.0002525311,0.9816768,0.00050096615,0.000102673286,0.000069587266,0.000556155,0.005467335,0.003774021],"genre_scores_gemma":[0.44328848,0.00050946604,0.5441485,0.0005981775,0.00021520315,0.0002628239,0.0025153428,0.0013628823,0.007099156],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9980679,0.0010611987,0.00009779161,0.00041246574,0.0002742835,0.00008643325],"domain_scores_gemma":[0.996223,0.0024494296,0.00023156666,0.00069960894,0.00030999162,0.000086413376],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018020663,0.0008705279,0.0007168982,0.0012912239,0.0005888433,0.0021877924,0.0016647524,0.0011511708,0.009877467],"category_scores_gemma":[0.009470994,0.00061512826,0.0011187408,0.0013957031,0.0012707254,0.0039417543,0.0020248313,0.001888548,0.0038068697],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00031558578,0.00017548217,0.0042543486,0.0006854165,0.00017106031,0.0007767664,0.001113314,0.1377013,0.015210499,0.30076063,0.021822535,0.51701313],"study_design_scores_gemma":[0.000020789832,0.00004348957,0.00030670708,0.000049099388,0.000031860796,0.00022941371,0.00012639028,0.7897636,0.0058038514,0.18707055,0.0165247,0.000029517158],"about_ca_topic_score_codex":0.0020818282,"about_ca_topic_score_gemma":0.0039501823,"teacher_disagreement_score":0.009877467,"about_ca_system_score_codex":0.0007531852,"about_ca_system_score_gemma":0.001707505,"threshold_uncertainty_score":0.033043444},"labels":[],"label_agreement":null},{"id":"W4401042369","doi":"10.18653/v1/2024.findings-naacl.41","title":"Solving Data-centric Tasks using Large Language Models","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Data modeling; Programming language; Human–computer interaction; Natural language processing; Software engineering","score_opus":0.040679397785736926,"score_gpt":0.3286579711128834,"score_spread":0.2879785733271465,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401042369","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.027110016,0.0013980551,0.9400016,0.0027329214,0.00018768436,0.0003797853,0.0023709773,0.01951757,0.006301327],"genre_scores_gemma":[0.21426156,0.0010986497,0.76611537,0.0006389221,0.0001714735,0.000598692,0.010199816,0.001726041,0.005189459],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99708956,0.0013264321,0.0002596905,0.0007917633,0.00039138927,0.00014119776],"domain_scores_gemma":[0.98433566,0.012208244,0.00039882987,0.002074219,0.0006113746,0.00037162483],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032630046,0.0019912168,0.0018604322,0.001081212,0.0014426428,0.0052041886,0.00347322,0.002248215,0.0060523767],"category_scores_gemma":[0.014301993,0.0018234599,0.0028524909,0.001690511,0.0011504029,0.010622401,0.0037862074,0.004149728,0.0038512452],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010175921,0.0011187891,0.0029812786,0.0019751708,0.00093576615,0.001025854,0.0014082819,0.3939875,0.015447193,0.056291226,0.08642095,0.43739045],"study_design_scores_gemma":[0.00017980346,0.00007170568,0.00022345106,0.000049595707,0.00010090755,0.0001071015,0.00032038486,0.88872117,0.005153758,0.09300179,0.012028765,0.00004166106],"about_ca_topic_score_codex":0.010193292,"about_ca_topic_score_gemma":0.029640548,"teacher_disagreement_score":0.010193292,"about_ca_system_score_codex":0.0016360639,"about_ca_system_score_gemma":0.0034922825,"threshold_uncertainty_score":0.020267904},"labels":[],"label_agreement":null},{"id":"W4401042477","doi":"10.18653/v1/2024.semeval-1.75","title":"Team AT at SemEval-2024 Task 8: Machine-Generated Text Detection with Semantic Embeddings","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"St. Francis Xavier University","funders":"","keywords":"SemEval; Computer science; Task (project management); Natural language processing; Artificial intelligence; Information retrieval","score_opus":0.006096407110127132,"score_gpt":0.24466585729853868,"score_spread":0.23856945018841155,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401042477","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.32372308,0.0067819837,0.28528795,0.0134213,0.012961683,0.00853917,0.108764,0.17257191,0.067949],"genre_scores_gemma":[0.40007532,0.00069899805,0.33711654,0.003391047,0.0010990623,0.0043942416,0.18689966,0.0076480946,0.058677018],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99027306,0.004536791,0.00043695635,0.0026145948,0.0015994324,0.0005392219],"domain_scores_gemma":[0.9808594,0.0068260795,0.0005474089,0.005238568,0.004730249,0.0017982231],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.016047599,0.0030939467,0.0020173362,0.002529275,0.0022889008,0.0033893452,0.0028426705,0.0038905516,0.015300958],"category_scores_gemma":[0.030973423,0.00088596577,0.0015384202,0.0014151826,0.001193155,0.006444288,0.0048424928,0.005583128,0.026620839],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0024825959,0.002534735,0.0060280524,0.0009955564,0.00048085564,0.0005213992,0.0009847247,0.007669351,0.021382295,0.0035724975,0.45875853,0.4945894],"study_design_scores_gemma":[0.002622203,0.0054776534,0.020691998,0.00040433038,0.0004613141,0.002267732,0.0026419517,0.3230165,0.13791773,0.020559799,0.48342755,0.00051122607],"about_ca_topic_score_codex":0.0062142476,"about_ca_topic_score_gemma":0.007839175,"teacher_disagreement_score":0.016047599,"about_ca_system_score_codex":0.0013511068,"about_ca_system_score_gemma":0.0025928088,"threshold_uncertainty_score":0.08486885},"labels":[],"label_agreement":null},{"id":"W4401042627","doi":"10.18653/v1/2024.naacl-demo.1","title":"TOPICAL: TOPIC Pages AutomagicaLly","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada","keywords":"Computer science; Information retrieval; World Wide Web","score_opus":0.012829161563518054,"score_gpt":0.2814461780181651,"score_spread":0.2686170164546471,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401042627","genre_codex":"software","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006522943,0.0044639846,0.26509878,0.0040430785,0.0057643875,0.0010868482,0.12908809,0.42680344,0.15712844],"genre_scores_gemma":[0.061132316,0.004259213,0.3034694,0.0023037202,0.0023428453,0.0011864858,0.2603229,0.10522397,0.25975913],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9988066,0.00019606305,0.000105435036,0.0003129289,0.00043617206,0.00014276615],"domain_scores_gemma":[0.9976337,0.00044300104,0.00011306631,0.0009702284,0.00066590647,0.00017411883],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012850441,0.0021659774,0.0012888919,0.0064564995,0.0021307282,0.0059753293,0.0015322968,0.0015989663,0.14643073],"category_scores_gemma":[0.005110821,0.0016330773,0.0022008638,0.0047066365,0.00054720044,0.01006175,0.006006445,0.0019155744,0.1865218],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013823404,0.000054686152,0.0005885335,0.00044095377,0.0000615043,0.00006708728,0.0002224654,0.00019552141,0.0035083971,0.0077160564,0.8661149,0.120891556],"study_design_scores_gemma":[0.00004197103,0.000023689312,0.0005436666,0.000115427996,0.00004815707,0.00015225587,0.0001806301,0.0028546983,0.007786328,0.012185512,0.97602624,0.000041434723],"about_ca_topic_score_codex":0.0039047734,"about_ca_topic_score_gemma":0.009864134,"teacher_disagreement_score":0.14643073,"about_ca_system_score_codex":0.00087912567,"about_ca_system_score_gemma":0.0018131601,"threshold_uncertainty_score":0.48985958},"labels":[],"label_agreement":null},{"id":"W4401042732","doi":"10.18653/v1/2024.starsem-1.11","title":"Paraphrase Identification via Textual Inference","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Machine Intelligence Institute","keywords":"Paraphrase; Computer science; Inference; Artificial intelligence; Identification (biology); Natural language processing","score_opus":0.013481027687619265,"score_gpt":0.30711194750791676,"score_spread":0.2936309198202975,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401042732","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.079239726,0.004507602,0.8406598,0.002131476,0.0005034754,0.0012059241,0.013604811,0.03501402,0.023133123],"genre_scores_gemma":[0.39646712,0.0012033649,0.5522487,0.0009741058,0.0003403407,0.0004541837,0.036684263,0.0011144015,0.010513557],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.996427,0.00092830835,0.00036467143,0.001122039,0.0009810404,0.00017698818],"domain_scores_gemma":[0.9925512,0.0033463046,0.00062369066,0.0017157195,0.001542444,0.0002206546],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024727632,0.0017549386,0.001166768,0.00368082,0.0010332647,0.0028474065,0.00430568,0.0021888814,0.0079541495],"category_scores_gemma":[0.022029772,0.0004946827,0.0011577989,0.0027864142,0.0009139547,0.006483172,0.0032548665,0.003063871,0.0066180667],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004636292,0.00077729597,0.0046911896,0.0027641829,0.0002984354,0.00094313297,0.00076061185,0.028539034,0.04865456,0.018220535,0.07007128,0.8238161],"study_design_scores_gemma":[0.00011987106,0.0003104111,0.0046038763,0.00026662793,0.00020139919,0.0014368617,0.00071581727,0.8142916,0.07297487,0.060568996,0.044378895,0.00013074934],"about_ca_topic_score_codex":0.0032369508,"about_ca_topic_score_gemma":0.007706338,"teacher_disagreement_score":0.0079541495,"about_ca_system_score_codex":0.0010749834,"about_ca_system_score_gemma":0.002099577,"threshold_uncertainty_score":0.026609302},"labels":[],"label_agreement":null},{"id":"W4401042851","doi":"10.18653/v1/2024.findings-naacl.101","title":"TagDebias: Entity and Concept Tagging for Social Bias Mitigation in Pretrained Language Models","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Alliance de recherche numérique du Canada; Mitacs","keywords":"Computer science; Natural language processing; Artificial intelligence; Language model","score_opus":0.0276593721056513,"score_gpt":0.3086277423095566,"score_spread":0.2809683702039053,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401042851","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0880203,0.0019952967,0.8833852,0.0012764718,0.00095098454,0.00052572123,0.0032511312,0.016561512,0.004033323],"genre_scores_gemma":[0.54905283,0.0008787638,0.4181583,0.0021521915,0.00055809517,0.00096890697,0.01535193,0.0015501439,0.011328817],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.997552,0.0010831612,0.00014003378,0.0006603625,0.00035587177,0.00020853037],"domain_scores_gemma":[0.99345064,0.0035925722,0.00033488753,0.0016256548,0.00078606425,0.00021013782],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00535435,0.0020936332,0.0011108917,0.0018133984,0.0010344993,0.0019532645,0.0024789965,0.002168545,0.0031604983],"category_scores_gemma":[0.013980569,0.0006698259,0.0012536137,0.0013790119,0.0009156003,0.0043406505,0.0028849011,0.0039107297,0.0036887936],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011844604,0.0008438245,0.015113226,0.0006668804,0.0005997944,0.00050876226,0.0009363011,0.09524331,0.025561756,0.013077839,0.06196479,0.784299],"study_design_scores_gemma":[0.00012052995,0.00030292757,0.0023660453,0.00011255914,0.0001260171,0.00031557577,0.00027179022,0.93563354,0.017287683,0.026744802,0.01662477,0.000093762166],"about_ca_topic_score_codex":0.0040815673,"about_ca_topic_score_gemma":0.011392737,"teacher_disagreement_score":0.00535435,"about_ca_system_score_codex":0.00097415515,"about_ca_system_score_gemma":0.0019504088,"threshold_uncertainty_score":0.028316796},"labels":[],"label_agreement":null},{"id":"W4401042975","doi":"10.18653/v1/2024.naacl-long.420","title":"Interplay of Machine Translation, Diacritics, and Diacritization","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Machine translation; Computer science; Translation (biology); Artificial intelligence; Natural language processing; Biology; Messenger RNA","score_opus":0.0069337225399700355,"score_gpt":0.28555567715292024,"score_spread":0.2786219546129502,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401042975","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2884431,0.02799273,0.34467426,0.02631121,0.0047061243,0.0002373813,0.0018355586,0.006509746,0.29928997],"genre_scores_gemma":[0.92319953,0.0033962948,0.054608095,0.00069134607,0.0008034929,0.00006603486,0.000711954,0.001460343,0.015062821],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99829775,0.000894785,0.00010373032,0.00030814958,0.00026272357,0.00013282083],"domain_scores_gemma":[0.99078256,0.005425137,0.0005786934,0.0016398494,0.0013249253,0.00024871712],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024889875,0.0006702208,0.0007717972,0.0014729517,0.0018316583,0.0072243116,0.0007932928,0.0011831614,0.011531139],"category_scores_gemma":[0.01623089,0.00074919115,0.0004746135,0.001919877,0.0022618382,0.010026137,0.0031000813,0.0020608117,0.0055505354],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008109795,0.00030434507,0.0049662553,0.001397281,0.00015685163,0.0007727141,0.0041275765,0.0075884676,0.032132383,0.54750866,0.033060253,0.3671743],"study_design_scores_gemma":[0.00008788173,0.00014874767,0.0043748217,0.00018887129,0.00016559206,0.0007289255,0.0015564744,0.06957316,0.02688387,0.7969898,0.09918255,0.00011925727],"about_ca_topic_score_codex":0.0012099976,"about_ca_topic_score_gemma":0.0020209413,"teacher_disagreement_score":0.011531139,"about_ca_system_score_codex":0.000912429,"about_ca_system_score_gemma":0.0011774567,"threshold_uncertainty_score":0.03857553},"labels":[],"label_agreement":null},{"id":"W4401043166","doi":"10.18653/v1/2024.findings-naacl.123","title":"Methods, Applications, and Directions of Learning-to-Rank in NLP Research","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Government of Canada","keywords":"Artificial intelligence; Rank (graph theory); Computer science; Natural language processing; Learning to rank; Machine learning; Ranking (information retrieval); Mathematics","score_opus":0.04320733918438564,"score_gpt":0.45186669717001327,"score_spread":0.40865935798562764,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401043166","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0027729725,0.079673424,0.8928894,0.013829273,0.0006926701,0.00015497713,0.00028525025,0.00044756068,0.009254589],"genre_scores_gemma":[0.13319737,0.10741327,0.73703337,0.0037483487,0.0072697806,0.0009025563,0.00078945083,0.00044063476,0.009205241],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9691478,0.02123921,0.0013996451,0.0027485094,0.0049501956,0.00051462156],"domain_scores_gemma":[0.88241404,0.09782219,0.0028699564,0.008174182,0.007672798,0.0010468342],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03786913,0.0021865591,0.0031361803,0.008119496,0.0021288411,0.011006805,0.0038743678,0.004416055,0.005532953],"category_scores_gemma":[0.08716519,0.0013501908,0.0018755195,0.015161335,0.009766314,0.015323707,0.0046921973,0.007232668,0.00407643],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000108112974,0.00015707096,0.0029414906,0.0016054534,0.00017600191,0.00014700854,0.00081752613,0.023443839,0.00045863152,0.64281136,0.010658337,0.3166752],"study_design_scores_gemma":[0.000044435765,0.00012364722,0.0007016766,0.0005767625,0.000044259148,0.00023567663,0.0003696208,0.11176604,0.00061167363,0.84604275,0.039373428,0.00010998334],"about_ca_topic_score_codex":0.0053812284,"about_ca_topic_score_gemma":0.004616491,"teacher_disagreement_score":0.03786913,"about_ca_system_score_codex":0.0029061795,"about_ca_system_score_gemma":0.0038421606,"threshold_uncertainty_score":0.20027345},"labels":[],"label_agreement":null},{"id":"W4401043207","doi":"10.18653/v1/2024.semeval-1.90","title":"OtterlyObsessedWithSemantics at SemEval-2024 Task 4: Developing a Hierarchical Multi-Label Classification Head for Large Language Models","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"Bayerische Akademie der Wissenschaften","keywords":"SemEval; Computer science; Task (project management); Head (geology); Natural language processing; Artificial intelligence; Engineering; Biology","score_opus":0.06040173108412855,"score_gpt":0.3564111599994934,"score_spread":0.29600942891536486,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401043207","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.27645135,0.0032955299,0.4592685,0.0073954575,0.003604499,0.002302153,0.037487123,0.13927452,0.07092092],"genre_scores_gemma":[0.5533009,0.00036792093,0.31283498,0.001471793,0.00032964212,0.0016225213,0.06622392,0.0038947253,0.059953563],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.995789,0.0017350401,0.00021365378,0.0011969528,0.00067557575,0.00038983062],"domain_scores_gemma":[0.9951054,0.0017907189,0.00019746329,0.0016101798,0.0009408264,0.00035536548],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0053433753,0.0034994504,0.0014298891,0.0019123681,0.0021450592,0.004175652,0.0027673363,0.0030275485,0.02405067],"category_scores_gemma":[0.01176514,0.00089295965,0.0018409779,0.0009837169,0.00094648305,0.0062377495,0.004995866,0.004266463,0.0147595275],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012167239,0.001081042,0.010322751,0.0008401798,0.00036957493,0.0005545729,0.0012655642,0.019411841,0.018959742,0.009995072,0.26416484,0.6718181],"study_design_scores_gemma":[0.00037586177,0.0006763605,0.011299487,0.0003209457,0.00019009529,0.000624819,0.0018476743,0.59197634,0.09483568,0.038002312,0.25961274,0.00023765741],"about_ca_topic_score_codex":0.01171304,"about_ca_topic_score_gemma":0.022198787,"teacher_disagreement_score":0.02405067,"about_ca_system_score_codex":0.002332893,"about_ca_system_score_gemma":0.002417497,"threshold_uncertainty_score":0.08045757},"labels":[],"label_agreement":null},{"id":"W4401043678","doi":"10.18653/v1/2024.semeval-1.254","title":"UAlberta at SemEval-2024 Task 1: A Potpourri of Methods for Quantifying Multilingual Semantic Textual Relatedness and Similarity","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Machine Intelligence Institute","keywords":"SemEval; Computer science; Potpourri; Natural language processing; Semantic similarity; Task (project management); Similarity (geometry); Artificial intelligence; Information retrieval; Botany","score_opus":0.052564131892954856,"score_gpt":0.411826296017001,"score_spread":0.35926216412404616,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401043678","genre_codex":"dataset","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.25817057,0.025432128,0.22191243,0.010517696,0.0070044105,0.007760738,0.2700672,0.10367362,0.095461145],"genre_scores_gemma":[0.23369543,0.0015560482,0.20832106,0.001822399,0.00068810664,0.005489226,0.5096047,0.0076851267,0.031137919],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.97882473,0.012597336,0.0014742237,0.0030867036,0.0032561275,0.00076101895],"domain_scores_gemma":[0.9743852,0.013454242,0.0008259871,0.0048595383,0.0049445736,0.0015304827],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0168535,0.004300268,0.0031626278,0.011186013,0.004826061,0.0067536295,0.0044191335,0.0059332713,0.021065447],"category_scores_gemma":[0.041195925,0.0012237774,0.0023598212,0.004464477,0.0017804556,0.013945131,0.012247676,0.004376918,0.01948629],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.003420117,0.0019114979,0.0070676715,0.003364708,0.0008981252,0.00078218075,0.0021809016,0.003273122,0.0135009885,0.008136854,0.58683264,0.36863127],"study_design_scores_gemma":[0.0034796472,0.0015198713,0.035073448,0.0017365804,0.0010467446,0.0025532516,0.0103137465,0.22158922,0.045736086,0.044575624,0.6315573,0.00081843877],"about_ca_topic_score_codex":0.021629363,"about_ca_topic_score_gemma":0.026392154,"teacher_disagreement_score":0.021629363,"about_ca_system_score_codex":0.0029063798,"about_ca_system_score_gemma":0.0034314247,"threshold_uncertainty_score":0.08913088},"labels":[],"label_agreement":null},{"id":"W4401047440","doi":"10.2139/ssrn.4907583","title":"Multi-Granularity Prosodic Speech Synthesis with Grammar Information","year":2024,"lang":"en","type":"preprint","venue":"SSRN Electronic Journal","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Windsor","funders":"","keywords":"Granularity; Computer science; Grammar; Natural language processing; Artificial intelligence; Speech recognition; Linguistics; Programming language","score_opus":0.007283251231385494,"score_gpt":0.24522203356638594,"score_spread":0.23793878233500046,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401047440","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.037576184,0.0002954715,0.9545468,0.00011157588,0.0001146796,0.000041816504,0.00030409318,0.00303202,0.003977414],"genre_scores_gemma":[0.5697564,0.00023442824,0.42487243,0.000087902794,0.00011692156,0.00010242906,0.00076907856,0.00069783535,0.0033625483],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9997236,0.00006614703,0.000022407992,0.00009676022,0.00005938017,0.00003174563],"domain_scores_gemma":[0.9994294,0.00032653502,0.000028236447,0.00012808887,0.00005741686,0.000030333267],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00035840383,0.0007228657,0.0006754227,0.00049341493,0.00027550387,0.0007623988,0.0004492351,0.0006124288,0.0066092666],"category_scores_gemma":[0.001112928,0.000525526,0.000524939,0.00045441097,0.00021137908,0.0009430539,0.0013914893,0.00096455857,0.0015395179],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014646475,0.00012822326,0.00062829256,0.00033071186,0.00011637288,0.0005154409,0.00022139595,0.07195384,0.35145944,0.0099389795,0.0023500705,0.5608925],"study_design_scores_gemma":[0.0001470297,0.00028620148,0.0013774192,0.000046516063,0.00013294027,0.0002480233,0.00012617177,0.8837206,0.09086779,0.017494345,0.0055152336,0.00003780748],"about_ca_topic_score_codex":0.00049696286,"about_ca_topic_score_gemma":0.0011424331,"teacher_disagreement_score":0.0066092666,"about_ca_system_score_codex":0.00017336883,"about_ca_system_score_gemma":0.00029922384,"threshold_uncertainty_score":0.022110164},"labels":[],"label_agreement":null},{"id":"W4401132887","doi":"10.1007/978-3-031-66329-1_41","title":"Large Language User Interfaces: Voice Interactive User Interfaces Powered by LLMs","year":2024,"lang":"en","type":"book-chapter","venue":"Lecture notes in networks and systems","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ontario Tech University","funders":"","keywords":"Computer science; Human–computer interaction","score_opus":0.0071225763394385435,"score_gpt":0.2526935945584568,"score_spread":0.24557101821901825,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401132887","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018872641,0.005912554,0.76556736,0.0006634363,0.0005431354,0.0001731717,0.0012021473,0.05499916,0.15206642],"genre_scores_gemma":[0.23211937,0.0035704551,0.19230951,0.0019214835,0.0008761036,0.0006105769,0.003061875,0.0064273095,0.5591034],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99983025,0.000038719973,0.000009537565,0.000034583267,0.00007320992,0.000013714462],"domain_scores_gemma":[0.9996958,0.00016863404,0.000015669651,0.000045112392,0.000050531242,0.000024312489],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00030917197,0.0007140658,0.00039247217,0.0004870018,0.00026979047,0.0014704907,0.001183246,0.0008014292,0.12081686],"category_scores_gemma":[0.0008359947,0.00037925655,0.00022836728,0.00062995276,0.00028803412,0.0026156327,0.0014071749,0.0006687202,0.030789252],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00046926265,0.00009160747,0.0003095186,0.00071763736,0.000034180124,0.00032099566,0.0007523724,0.00062165596,0.14814584,0.023367736,0.13402775,0.6911415],"study_design_scores_gemma":[0.000120026925,0.00026511008,0.0016553847,0.00024165174,0.00010335558,0.0016589587,0.00032589387,0.03149194,0.081813686,0.021174919,0.8610626,0.00008650815],"about_ca_topic_score_codex":0.0002567975,"about_ca_topic_score_gemma":0.00067276176,"teacher_disagreement_score":0.12081686,"about_ca_system_score_codex":0.00022532244,"about_ca_system_score_gemma":0.000115764255,"threshold_uncertainty_score":0.40417266},"labels":[],"label_agreement":null},{"id":"W4401201550","doi":"10.48550/arxiv.2407.19299","title":"The Impact of LoRA Adapters on LLMs for Clinical Text Classification Under Computational and Data Constraints","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Institut de Valorisation des Données; Aristotle University of Thessaloniki; Università degli Studi di Parma; Fonds de Recherche du Québec - Santé; University of Surrey; Fonds de recherche du Québec – Nature et technologies; Université de Montréal; Norges Teknisk-Naturvitenskapelige Universitet; Université du Luxembourg","keywords":"Artificial intelligence; Computer science; Natural language processing","score_opus":0.2683584329024094,"score_gpt":0.3461425138004021,"score_spread":0.07778408089799271,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401201550","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.545741,0.0034411736,0.38044918,0.0017314557,0.00068236946,0.0005759966,0.0019821622,0.055148076,0.010248504],"genre_scores_gemma":[0.8179113,0.00062809343,0.16867882,0.0009395538,0.00010763068,0.00043108064,0.0049982676,0.0010356582,0.005269583],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99876165,0.00041476745,0.000103141916,0.0003782729,0.00018936828,0.00015280256],"domain_scores_gemma":[0.9971499,0.0015929067,0.00013570717,0.00057488267,0.00039877693,0.0001477993],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002926114,0.0015804125,0.00058942306,0.0006762539,0.0004185913,0.0010740417,0.0017234464,0.0010434098,0.0037683127],"category_scores_gemma":[0.012301366,0.00044663926,0.00076308177,0.00056761666,0.0005631249,0.0022870123,0.0017461876,0.0020178172,0.002403509],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013620724,0.00064716855,0.012723704,0.00046139632,0.0003671741,0.0003948145,0.00030491911,0.26754776,0.021710278,0.0021054042,0.01852255,0.6738528],"study_design_scores_gemma":[0.00010751783,0.0004259337,0.002719994,0.000041542004,0.000070626666,0.00015601455,0.00017655884,0.9760685,0.0148683265,0.0016507495,0.003679067,0.000035207224],"about_ca_topic_score_codex":0.010683284,"about_ca_topic_score_gemma":0.019358495,"teacher_disagreement_score":0.010683284,"about_ca_system_score_codex":0.0008534233,"about_ca_system_score_gemma":0.0016055601,"threshold_uncertainty_score":0.021242201},"labels":[],"label_agreement":null},{"id":"W4401412882","doi":"10.4230/lipics.sea.2024.10","title":"Taxonomic Classification with Maximal Exact Matches in KATKA Kernels and Minimizer Digests","year":2024,"lang":"en","type":"article","venue":"PubMed","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"National Institute of General Medical Sciences; Natural Sciences and Engineering Research Council of Canada; National Human Genome Research Institute; Gruppo Nazionale per il Calcolo Scientifico; Agencia Nacional de Investigación y Desarrollo; Ministero della Salute; Istituto Nazionale di Alta Matematica \"Francesco Severi\"; National Institutes of Health; National Science Foundation","keywords":"Mathematics; Combinatorics","score_opus":0.02307681910436923,"score_gpt":0.22936124112503362,"score_spread":0.20628442202066438,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401412882","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.25012636,0.0007188531,0.7323964,0.0004390363,0.000119954944,0.00019264207,0.0028053417,0.009818113,0.0033832218],"genre_scores_gemma":[0.5027436,0.00021709688,0.48648107,0.00009248456,0.0000334106,0.00021814204,0.006494349,0.0007279505,0.0029918852],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99854565,0.0003563275,0.0002041531,0.0004073061,0.0003315992,0.00015492568],"domain_scores_gemma":[0.9953349,0.0018554965,0.00039074878,0.0014733168,0.00079160446,0.00015394698],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026681228,0.0007068911,0.0007714454,0.0021508418,0.00062355056,0.0020747208,0.0013991359,0.001240852,0.0039299065],"category_scores_gemma":[0.017521841,0.00046832496,0.0011616782,0.0020151946,0.00059979194,0.003543096,0.0017062065,0.0013692942,0.0029218073],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.003092855,0.0005401234,0.02134189,0.0011968508,0.00034171747,0.00039351455,0.001235334,0.13111901,0.053829554,0.03374332,0.020247594,0.7329182],"study_design_scores_gemma":[0.00006223024,0.00023940411,0.005924128,0.00008966386,0.00006985985,0.00047476668,0.00044516358,0.9187243,0.02278253,0.04363461,0.00747858,0.00007473171],"about_ca_topic_score_codex":0.0021868765,"about_ca_topic_score_gemma":0.0027256901,"teacher_disagreement_score":0.0039299065,"about_ca_system_score_codex":0.000753486,"about_ca_system_score_gemma":0.0011195382,"threshold_uncertainty_score":0.014110565},"labels":[],"label_agreement":null},{"id":"W4401419681","doi":"10.37236/11270","title":"A Note on the Maximum Number of k-Powers in a Finite Word","year":2024,"lang":"en","type":"article","venue":"The Electronic Journal of Combinatorics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"Uniwersytet Warszawski","keywords":"Word (group theory); Mathematics; Arithmetic; Geometry","score_opus":0.007622824280421959,"score_gpt":0.2723886229256945,"score_spread":0.26476579864527255,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401419681","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1746871,0.031338368,0.44865334,0.020074774,0.0052990466,0.00023583345,0.0012356829,0.001762392,0.31671342],"genre_scores_gemma":[0.80051327,0.012593977,0.13618359,0.0045110635,0.005881298,0.00067725097,0.00077552994,0.001222295,0.037641726],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9942731,0.0012017052,0.000568465,0.0017723419,0.0012050826,0.0009793134],"domain_scores_gemma":[0.96533495,0.028093763,0.0010559178,0.0034041896,0.0010763959,0.001034764],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0040439214,0.0025741493,0.0019906112,0.0026937195,0.0038873632,0.005203381,0.0032786794,0.002428214,0.010471413],"category_scores_gemma":[0.023142511,0.0021488888,0.0024429325,0.002796099,0.0091325175,0.021643821,0.0078021307,0.010127275,0.0040945434],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012198476,0.00019548334,0.0046275496,0.0010815819,0.00012458998,0.0015140144,0.0019815431,0.013331778,0.017990353,0.84143096,0.02334019,0.09316214],"study_design_scores_gemma":[0.000063122265,0.00016831994,0.0011409714,0.00021021279,0.00010637838,0.0011678239,0.00018571784,0.020105055,0.007155648,0.9378823,0.03170661,0.000107859625],"about_ca_topic_score_codex":0.0011916636,"about_ca_topic_score_gemma":0.0012123049,"teacher_disagreement_score":0.010471413,"about_ca_system_score_codex":0.0025202637,"about_ca_system_score_gemma":0.0011327809,"threshold_uncertainty_score":0.035030305},"labels":[],"label_agreement":null},{"id":"W4401453798","doi":"10.1109/hsi61632.2024.10613583","title":"Enhancing Neural Machine Translation of Indigenous Languages through Gender Debiasing and Named Entity Recognition","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Debiasing; Computer science; Indigenous; Machine translation; Artificial intelligence; Natural language processing; Translation (biology); Psychology; Cognitive science","score_opus":0.029106753169971387,"score_gpt":0.30278875695496654,"score_spread":0.27368200378499513,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401453798","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.56475186,0.0013101483,0.40175316,0.0009298227,0.00031083557,0.00017550164,0.00055520446,0.005048907,0.025164492],"genre_scores_gemma":[0.8116176,0.0005408995,0.17603888,0.0002708949,0.000055109434,0.000057485864,0.0013640008,0.00019250496,0.009862592],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99971706,0.00008772533,0.000018916378,0.000070461254,0.00006428837,0.00004150773],"domain_scores_gemma":[0.9993381,0.00022562174,0.00006633058,0.000109475215,0.000242166,0.00001821499],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00075660617,0.00038998047,0.00028931777,0.000494961,0.00046860205,0.00070934754,0.0004424601,0.0003739507,0.0028412677],"category_scores_gemma":[0.002336425,0.00011529671,0.00026590607,0.00069276814,0.00032832558,0.0015214578,0.00074490905,0.0005206225,0.0015382268],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023465599,0.00020458143,0.00405929,0.00020822168,0.00005008284,0.00030657728,0.00081615127,0.027296359,0.07020842,0.008025387,0.005053721,0.8835366],"study_design_scores_gemma":[0.000055491197,0.000474428,0.0095622735,0.000068125206,0.00014387694,0.0006569084,0.0015282042,0.7602477,0.17346328,0.018191248,0.035532314,0.00007625123],"about_ca_topic_score_codex":0.008256872,"about_ca_topic_score_gemma":0.0154614905,"teacher_disagreement_score":0.008256872,"about_ca_system_score_codex":0.0005287389,"about_ca_system_score_gemma":0.00080629915,"threshold_uncertainty_score":0.016417623},"labels":[],"label_agreement":null},{"id":"W4401500059","doi":"10.5539/elt.v17n9p26","title":"Construction of a Corpus for L2 Derivational Morphology: Based on Construction","year":2024,"lang":"en","type":"article","venue":"English Language Teaching","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Psychology; Linguistics; Morphology (biology); Natural language processing; Computer science; Philosophy; Geology","score_opus":0.007918195315693738,"score_gpt":0.2671008478312024,"score_spread":0.25918265251550865,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401500059","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.42714694,0.0007567378,0.40862912,0.00084060733,0.0005069378,0.007439323,0.033874422,0.0030061305,0.117799856],"genre_scores_gemma":[0.4662274,0.0004661551,0.4465859,0.00026004048,0.00011868518,0.012849661,0.054340295,0.0018795925,0.017272256],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9969506,0.0012544346,0.00039395085,0.0007591165,0.00052176375,0.00012008201],"domain_scores_gemma":[0.9877215,0.0057724565,0.00035281907,0.0026345353,0.0031894664,0.00032930574],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003284489,0.000616948,0.00062970095,0.0048903874,0.002506697,0.0021984566,0.0011743875,0.0007903087,0.017109621],"category_scores_gemma":[0.010844105,0.00052859425,0.0003726947,0.006084061,0.0017317241,0.0026765321,0.0046338597,0.0014899145,0.005738737],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006411718,0.0011134881,0.033040885,0.003077986,0.000054122076,0.0034872463,0.07761822,0.0038924455,0.07589496,0.11994884,0.058697157,0.6225334],"study_design_scores_gemma":[0.00020752996,0.00044414224,0.08443544,0.00078366784,0.00008514284,0.0048246114,0.025978785,0.023793498,0.051400118,0.02972574,0.7780632,0.00025817132],"about_ca_topic_score_codex":0.0024359121,"about_ca_topic_score_gemma":0.0037038394,"teacher_disagreement_score":0.017109621,"about_ca_system_score_codex":0.0013900984,"about_ca_system_score_gemma":0.0027468812,"threshold_uncertainty_score":0.057237446},"labels":[],"label_agreement":null},{"id":"W4401543475","doi":"10.3828/coma.2022.3","title":"Linking Archives, Linked Open Data, and the Development of the World-Wide Directory of Repositories Holding Archives of Literature and Art","year":2022,"lang":"en","type":"article","venue":"Comma","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"","keywords":"Directory; World Wide Web; Library science; Order (exchange); Computer science; Museology; History; Business; Archaeology","score_opus":0.022903494812461626,"score_gpt":0.28543257107223496,"score_spread":0.26252907625977334,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401543475","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020671027,0.0032092954,0.89019847,0.016680153,0.001486719,0.0007133839,0.004225344,0.0099062575,0.05290941],"genre_scores_gemma":[0.1004823,0.0028672286,0.8594845,0.0012807923,0.00053007226,0.0007440019,0.006496712,0.0022333022,0.025881154],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.98057437,0.009488442,0.0017125712,0.0022282866,0.0053964816,0.0005997998],"domain_scores_gemma":[0.9083582,0.039791875,0.0071907565,0.026937628,0.014627972,0.0030935965],"candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.03064344,0.0005272125,0.00060800987,0.01778602,0.00692219,0.015931275,0.0027419496,0.0018818957,0.005100752],"category_scores_gemma":[0.0812275,0.0010541787,0.0007741827,0.027368391,0.0060237614,0.03806394,0.012847706,0.0042883386,0.0022974482],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008477645,0.00008431759,0.004746808,0.00059836067,0.00004458816,0.00031644333,0.01564732,0.0020563924,0.0011571341,0.5713191,0.037452094,0.3664927],"study_design_scores_gemma":[0.000015617112,0.000034846275,0.0018614436,0.0006143326,0.000028826573,0.0005019569,0.008906913,0.0077605466,0.0028550613,0.19226928,0.7850312,0.00011994932],"about_ca_topic_score_codex":0.01666862,"about_ca_topic_score_gemma":0.02457771,"teacher_disagreement_score":0.98406875,"about_ca_system_score_codex":0.00404448,"about_ca_system_score_gemma":0.013334644,"threshold_uncertainty_score":0.16205996},"labels":[],"label_agreement":null},{"id":"W4401631128","doi":"10.1016/b978-0-323-95504-1.00091-0","title":"Generative Grammar","year":2024,"lang":"en","type":"book-chapter","venue":"Elsevier eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Generative grammar; Linguistics; Computer science; Grammar; Natural language processing; Artificial intelligence; Philosophy","score_opus":0.013915929446035997,"score_gpt":0.25601743113234604,"score_spread":0.24210150168631003,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401631128","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.001828483,0.004079563,0.10206018,0.0014116573,0.00035349216,0.00003762314,0.0009825861,0.0020442652,0.88720214],"genre_scores_gemma":[0.067084804,0.007032231,0.04169335,0.0006537629,0.0004914966,0.00013091191,0.0047940514,0.003168975,0.87495047],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9997595,0.000054099255,0.00001348436,0.00006236236,0.000094115676,0.000016421745],"domain_scores_gemma":[0.99974066,0.000111942776,0.000007653555,0.00008922447,0.00003942966,0.00001106281],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002627782,0.000795471,0.0007283113,0.0012389442,0.0008924876,0.0031076134,0.000941883,0.0008727749,0.11826801],"category_scores_gemma":[0.0009923237,0.0006739695,0.00065866794,0.0019208174,0.0017357426,0.0029096096,0.0013655612,0.0018075866,0.056994077],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000072745024,0.00001741619,0.00012464817,0.00014797099,0.000009480159,0.00009762105,0.00040980583,0.0010034218,0.00070825464,0.7006134,0.091158554,0.20570211],"study_design_scores_gemma":[0.0000060547013,0.000005180184,0.00020885054,0.00009726375,0.000008062429,0.00023464863,0.00007838374,0.0014072005,0.0005145713,0.5215661,0.4758651,0.000008634056],"about_ca_topic_score_codex":0.0023433752,"about_ca_topic_score_gemma":0.0031689103,"teacher_disagreement_score":0.11826801,"about_ca_system_score_codex":0.0010993131,"about_ca_system_score_gemma":0.0007543949,"threshold_uncertainty_score":0.39564592},"labels":[],"label_agreement":null},{"id":"W4401631969","doi":"10.22215/etd/2024-16145","title":"Collaborative Interface Design and Language Archiving: The Case of Tsuut’ina","year":2024,"lang":"en","type":"dissertation","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Documentation; Interface (matter); Computer science; Field (mathematics); World Wide Web; Work (physics); Linguistics; Engineering","score_opus":0.008300208709231498,"score_gpt":0.3111344058850265,"score_spread":0.302834197175795,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401631969","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.65470105,0.0025405723,0.17364383,0.008977833,0.00043075823,0.0008965687,0.00014922417,0.0010368786,0.1576233],"genre_scores_gemma":[0.8567724,0.000831086,0.1085073,0.00057490723,0.000060860257,0.00068680296,0.00011452816,0.00042042986,0.03203161],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.97142047,0.02197577,0.0010661087,0.00151323,0.0030151743,0.0010093122],"domain_scores_gemma":[0.9645568,0.026222406,0.0010303009,0.0034357577,0.0030874335,0.0016672673],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.024272572,0.00055443816,0.00060247094,0.002072047,0.016046198,0.014783977,0.0031948618,0.0040980643,0.0061875777],"category_scores_gemma":[0.03425004,0.00062027626,0.0009651925,0.003616288,0.011211204,0.008689513,0.009507229,0.0029930694,0.0011242105],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018412259,0.00032230132,0.0031129075,0.0005052733,0.0000267959,0.0041498346,0.83446515,0.0012888118,0.0030540787,0.06830311,0.003892933,0.08069476],"study_design_scores_gemma":[0.0001484612,0.0005611295,0.003797321,0.0008320795,0.00010194213,0.0041573667,0.43692118,0.008479783,0.008396623,0.035537563,0.50091934,0.0001471924],"about_ca_topic_score_codex":0.011201583,"about_ca_topic_score_gemma":0.013099303,"teacher_disagreement_score":0.98879844,"about_ca_system_score_codex":0.0070799575,"about_ca_system_score_gemma":0.0076267226,"threshold_uncertainty_score":0.12836719},"labels":[],"label_agreement":null},{"id":"W4401662241","doi":"10.48550/arxiv.2407.11722","title":"Exploring Quantization for Efficient Pre-Training of Transformer Language Models","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Fonds de recherche du Québec – Nature et technologies; Alliance de recherche numérique du Canada; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Transformer; Computer science; Quantization (signal processing); Language model; Artificial intelligence; Engineering; Algorithm; Electrical engineering; Voltage","score_opus":0.1879526093765473,"score_gpt":0.24154932460240108,"score_spread":0.05359671522585377,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401662241","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.040811393,0.0006066905,0.9505255,0.0005189205,0.000093256465,0.000101147394,0.00017848307,0.0052237455,0.0019408744],"genre_scores_gemma":[0.65976685,0.00036375455,0.33469412,0.00058562687,0.00004679767,0.00025976938,0.0006051309,0.0010415836,0.002636291],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99916446,0.0002964873,0.00006772641,0.00022076584,0.0001361846,0.00011440136],"domain_scores_gemma":[0.99684465,0.0021463723,0.00013093505,0.00042746653,0.00034527498,0.00010525704],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021040346,0.0011554137,0.00074706087,0.00044947883,0.0004575455,0.0014166132,0.0018559836,0.0010991347,0.0044842786],"category_scores_gemma":[0.014577296,0.0007385218,0.00058640447,0.00046294942,0.00095553295,0.0029680098,0.0022888125,0.0035879007,0.0017171588],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005795837,0.00029574783,0.00280039,0.00040941336,0.00010657293,0.00022020144,0.0005948357,0.51285464,0.034948822,0.025425365,0.007717052,0.4140474],"study_design_scores_gemma":[0.000034323588,0.00006448949,0.00017914928,0.000030252122,0.000014666896,0.000036167396,0.00004792699,0.97507745,0.010212877,0.01323048,0.0010582814,0.0000138691985],"about_ca_topic_score_codex":0.004840416,"about_ca_topic_score_gemma":0.008667803,"teacher_disagreement_score":0.004840416,"about_ca_system_score_codex":0.0008927721,"about_ca_system_score_gemma":0.001859338,"threshold_uncertainty_score":0.015001357},"labels":[],"label_agreement":null},{"id":"W4401705640","doi":"10.1093/acrefore/9780199384655.013.370","title":"Bilingual Language Processing","year":2024,"lang":"en","type":"reference-entry","venue":"Oxford Research Encyclopedia of Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Wilfrid Laurier University","funders":"","keywords":"Computer science; Programming language; Linguistics; Natural language processing; Philosophy","score_opus":0.04317874665805476,"score_gpt":0.3849855820199475,"score_spread":0.3418068353618927,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401705640","genre_codex":"empirical","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5491369,0.0033734876,0.026676904,0.003833876,0.0005095301,0.00012353975,0.002443779,0.0012724156,0.4126296],"genre_scores_gemma":[0.9727252,0.00067718927,0.0062237754,0.0003570487,0.00009071231,0.00003451194,0.00090632326,0.00019618016,0.018789018],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9996835,0.00006464901,0.000020167106,0.00011220292,0.00007358265,0.00004599602],"domain_scores_gemma":[0.9992316,0.00026441782,0.00010115057,0.000107593536,0.00020826266,0.00008707126],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00041046864,0.00031101765,0.0002483288,0.00074917654,0.00046899368,0.0020923095,0.00026156238,0.0004295535,0.03405963],"category_scores_gemma":[0.0027854657,0.00014807406,0.00020641401,0.00048424935,0.0004858873,0.0018793877,0.0014726684,0.0004939919,0.00445569],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018390765,0.00026431296,0.028853668,0.0016609294,0.00013077441,0.0034387156,0.012115517,0.0013570216,0.29209855,0.10608026,0.03227005,0.51989114],"study_design_scores_gemma":[0.00034031225,0.0007586596,0.2573731,0.0008588692,0.00028692704,0.011722213,0.01637583,0.020703891,0.111683905,0.3435689,0.23611209,0.00021543501],"about_ca_topic_score_codex":0.0026024252,"about_ca_topic_score_gemma":0.003044921,"teacher_disagreement_score":0.03405963,"about_ca_system_score_codex":0.00076117995,"about_ca_system_score_gemma":0.0007394499,"threshold_uncertainty_score":0.113940775},"labels":[],"label_agreement":null},{"id":"W4401952885","doi":"10.1038/d41586-024-02527-x","title":"LLMs produce racist output when prompted in African American English","year":2024,"lang":"en","type":"article","venue":"Nature","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ubisoft (Canada)","funders":"","keywords":"Covert; Harm; Racism; Race (biology); Political science; Criminology; Sociology; Psychology; Gender studies; Social psychology; Linguistics","score_opus":0.0066576415112439965,"score_gpt":0.2659464306549486,"score_spread":0.2592887891437046,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401952885","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.84985983,0.00015981118,0.017792117,0.0016919129,0.000564659,0.00014953387,0.005071176,0.015675986,0.10903485],"genre_scores_gemma":[0.95796937,0.00008695557,0.013933194,0.00054144184,0.00007760786,0.00007732602,0.0033270752,0.0020539616,0.021933034],"study_design_codex":"bench_or_experimental","study_design_gemma":"observational","domain_scores_codex":[0.9992773,0.00030772417,0.0000378293,0.00012746014,0.0001656279,0.00008410633],"domain_scores_gemma":[0.9890352,0.008546139,0.0004081342,0.0005540515,0.0011326367,0.00032372292],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007389579,0.0006283784,0.00037822247,0.00038345662,0.0007423129,0.0011197174,0.0003255873,0.00094268605,0.02572038],"category_scores_gemma":[0.010230644,0.00020804213,0.00018813527,0.00033028304,0.0003330012,0.00083329144,0.0009067787,0.000705247,0.009046433],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0072286665,0.00077939825,0.06412737,0.0023406858,0.0001442074,0.0073447116,0.034046326,0.0038828868,0.4172983,0.010078903,0.15166843,0.30106008],"study_design_scores_gemma":[0.0005105483,0.0015128693,0.23796672,0.0008054159,0.00043105788,0.004407702,0.035395086,0.12332738,0.3412853,0.00952426,0.24449646,0.00033723915],"about_ca_topic_score_codex":0.00560901,"about_ca_topic_score_gemma":0.0077216937,"teacher_disagreement_score":0.02572038,"about_ca_system_score_codex":0.00046084286,"about_ca_system_score_gemma":0.00044024122,"threshold_uncertainty_score":0.0860433},"labels":[],"label_agreement":null},{"id":"W4402028425","doi":"10.17605/osf.io/9qa2z","title":"Probing cross-linguistic differences in time representations using arithmetic word problems","year":2024,"lang":"en","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Arithmetic; Linguistics; Word (group theory); Computer science; Mathematics; Natural language processing; Philosophy","score_opus":0.025574415906224942,"score_gpt":0.28914224439921965,"score_spread":0.2635678284929947,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402028425","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98679334,0.00012327648,0.0051039136,0.00008361006,0.000028812123,0.00007000749,0.00023572183,0.00004141699,0.007519824],"genre_scores_gemma":[0.9905822,0.00015319626,0.005827437,0.00010417558,0.000020658845,0.0002502646,0.0005446217,0.00006452938,0.0024528601],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9986223,0.0003736437,0.00014963812,0.00043889284,0.00032415026,0.00009142179],"domain_scores_gemma":[0.99403363,0.0030559094,0.0014831218,0.00081775343,0.0003641949,0.00024549454],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013735618,0.00060639664,0.00044702453,0.0011917391,0.00046146155,0.0027744623,0.0006930455,0.0007318206,0.0077262544],"category_scores_gemma":[0.017947936,0.0002911407,0.00044381118,0.00085002737,0.0010616351,0.0032002784,0.0020688067,0.0012280873,0.0011742908],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.005020829,0.0047342884,0.25130382,0.001805251,0.0006504925,0.0015110332,0.109222844,0.007054984,0.27096632,0.051692147,0.004898582,0.29113948],"study_design_scores_gemma":[0.00047657432,0.004648743,0.79579914,0.0003095439,0.00035432482,0.0022358436,0.03752878,0.01889281,0.03059274,0.08249043,0.026246415,0.00042463685],"about_ca_topic_score_codex":0.0014136375,"about_ca_topic_score_gemma":0.00091875676,"teacher_disagreement_score":0.0077262544,"about_ca_system_score_codex":0.00042947225,"about_ca_system_score_gemma":0.00020833116,"threshold_uncertainty_score":0.025846899},"labels":[],"label_agreement":null},{"id":"W4402302580","doi":"10.55016/ojs/cpai.v3i1.68399","title":"Text-matching software in post-secondary contexts: A systematic review protocol","year":2020,"lang":"en","type":"review","venue":"Canadian Perspectives on Academic Integrity","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Protocol (science); Computer science; Matching (statistics); Systematic review; Software; Information retrieval; Software engineering; Programming language; MEDLINE; Medicine; Political science; Mathematics; Statistics","score_opus":0.022848456449461606,"score_gpt":0.3532106363324247,"score_spread":0.3303621798829631,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402302580","genre_codex":"protocol","genre_gemma":"protocol","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"protocol","genre_consensus":"protocol","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00038434853,0.0013044404,0.0030598275,0.0004775954,0.000293601,0.990104,0.003751132,0.00012261576,0.00050236663],"genre_scores_gemma":[0.0002802808,0.00041806413,0.0038078371,0.00014320873,0.000019523184,0.994863,0.0002584199,0.000007811713,0.00020193319],"study_design_codex":"systematic_review","study_design_gemma":"systematic_review","domain_scores_codex":[0.8971258,0.051794946,0.03166871,0.0060171606,0.010314485,0.003078909],"domain_scores_gemma":[0.85851794,0.06867533,0.019963393,0.017311627,0.031327862,0.004203826],"candidate_categories":["metaresearch","research_integrity"],"consensus_categories":[],"category_scores_codex":[0.17812362,0.005321801,0.01235442,0.02204933,0.0047660256,0.0070404354,0.0046602213,0.009324013,0.091969],"category_scores_gemma":[0.2078816,0.006475179,0.012495629,0.016467588,0.006527464,0.0093826,0.006804157,0.0088908635,0.019848075],"study_design_candidate":"systematic_review","study_design_consensus":"systematic_review","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0077580833,0.0004799737,0.0010936706,0.8020624,0.0022440974,0.0009916513,0.0075974567,0.0017241514,0.0026851736,0.014061135,0.05625682,0.103045285],"study_design_scores_gemma":[0.035716288,0.0023541271,0.0065319375,0.6163513,0.007896563,0.00076308934,0.0061211376,0.0021101232,0.0037567327,0.03233331,0.285046,0.001019387],"about_ca_topic_score_codex":0.007168101,"about_ca_topic_score_gemma":0.0143898,"teacher_disagreement_score":0.990676,"about_ca_system_score_codex":0.013456826,"about_ca_system_score_gemma":0.08503295,"threshold_uncertainty_score":0.942019},"labels":[],"label_agreement":null},{"id":"W4402324443","doi":"10.21203/rs.3.rs-5038817/v1","title":"LLMs for Closed-Library Multi-Document Query, Test Generation, and Evaluation","year":2024,"lang":"en","type":"preprint","venue":"Research Square","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Computer science; Leverage (statistics); Knowledge base; World Wide Web; Test (biology); Information retrieval; Knowledge management; Data science; Artificial intelligence","score_opus":0.10931125545381631,"score_gpt":0.45314457102939015,"score_spread":0.34383331557557384,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402324443","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01383284,0.000593941,0.87259245,0.0003845357,0.000088900524,0.00033705158,0.0018913229,0.106915735,0.0033632447],"genre_scores_gemma":[0.26419678,0.00018323764,0.7187841,0.0003715603,0.00016389014,0.00052535103,0.0055918577,0.0051076957,0.0050754636],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9887221,0.0053674662,0.0009855211,0.0012878657,0.0030561264,0.00058092567],"domain_scores_gemma":[0.97386825,0.01572269,0.0011166128,0.005613847,0.00308009,0.0005985025],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006455031,0.0018516443,0.0017613415,0.003281872,0.0010402467,0.003358833,0.004057306,0.002346548,0.023498066],"category_scores_gemma":[0.034624796,0.00092436397,0.0016727957,0.0022642736,0.0013094073,0.004902925,0.003865416,0.0018237457,0.009175009],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0023460987,0.00054066506,0.0029413754,0.0008141412,0.00020719289,0.00032826432,0.00030948853,0.043540683,0.021144815,0.027004866,0.06869471,0.83212775],"study_design_scores_gemma":[0.000297829,0.0003207347,0.00087889814,0.000073509415,0.00007788125,0.00021005068,0.0001152359,0.9216456,0.027429668,0.036979727,0.011915041,0.00005585479],"about_ca_topic_score_codex":0.009283123,"about_ca_topic_score_gemma":0.013949925,"teacher_disagreement_score":0.023498066,"about_ca_system_score_codex":0.0029109602,"about_ca_system_score_gemma":0.0034056893,"threshold_uncertainty_score":0.07860881},"labels":[{"model":"gemma","categories":[],"domain":null,"study_design":"simulation_or_modeling","genre":"methods","about_ca_system":false,"about_ca_topic":false,"confidence":"low"},{"model":"gpt","categories":[],"domain":null,"study_design":"simulation_or_modeling","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"medium"}],"label_agreement":"agree"},{"id":"W4402423652","doi":"10.1007/978-3-031-70645-5_7","title":"Οpen Parliamentary Data as a Tool for Linguistic Research: Exploring the ‘Greek Language Question’ in the Journal of Parliamentary Debates","year":2024,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Library of Parliament","funders":"","keywords":"Computer science; Linguistics; Artificial intelligence; Natural language processing; Philosophy","score_opus":0.09835994509403702,"score_gpt":0.36892924929544874,"score_spread":0.27056930420141173,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402423652","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.22149535,0.030270765,0.058201596,0.20121235,0.0025754871,0.00010460443,0.0036469488,0.00027305097,0.48221982],"genre_scores_gemma":[0.94888526,0.004636316,0.01818761,0.0040455596,0.0012322768,0.000119310265,0.0014657547,0.0003879258,0.021039994],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.9935516,0.0049330625,0.00024417727,0.00037691058,0.00072394754,0.00017029178],"domain_scores_gemma":[0.9536521,0.041298617,0.0017077535,0.0013557338,0.0014684515,0.00051735033],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007842937,0.00028682617,0.00044348746,0.0040641194,0.0031340355,0.016251627,0.0011806851,0.0025142913,0.009673523],"category_scores_gemma":[0.038600188,0.0004608422,0.00034613494,0.0122846505,0.009303427,0.026605317,0.0035363736,0.0036456524,0.0015750043],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000078898986,0.00004028352,0.0038516726,0.00025534426,0.000019117608,0.00014962119,0.04475434,0.0002106529,0.0004632106,0.87585634,0.024561435,0.04975903],"study_design_scores_gemma":[0.000021140979,0.00002383536,0.007421858,0.0007584657,0.000024264315,0.00024664376,0.05391978,0.0023780665,0.0009942446,0.64050937,0.29365876,0.00004363501],"about_ca_topic_score_codex":0.0043344684,"about_ca_topic_score_gemma":0.0062267426,"teacher_disagreement_score":0.016251627,"about_ca_system_score_codex":0.002888666,"about_ca_system_score_gemma":0.0018581013,"threshold_uncertainty_score":0.04147792},"labels":[],"label_agreement":null},{"id":"W4402423727","doi":"10.24908/iqurcp18054","title":"Leveraging Large Language Models for Automating Inductive Qualitative Coding: A Comparative Study of Prompt Engineering Techniques","year":2024,"lang":"en","type":"article","venue":"Inquiry Queen s Undergraduate Research Conference Proceedings","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Coding (social sciences); Inductive reasoning; Programming language; Inductive method; Software engineering; Natural language processing; Artificial intelligence; Mathematics education; Teaching method; Psychology; Sociology","score_opus":0.17615556714151862,"score_gpt":0.4539327931313133,"score_spread":0.2777772259897947,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402423727","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07478977,0.0003534402,0.91704506,0.0007740102,0.00005625899,0.0017864369,0.00026071633,0.0017111182,0.0032232252],"genre_scores_gemma":[0.20800251,0.00034113275,0.7877709,0.00026714843,0.000021492588,0.0021775488,0.0003138047,0.0003764098,0.0007290196],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.8881613,0.09864073,0.0030071486,0.0040168855,0.0053501325,0.0008237702],"domain_scores_gemma":[0.46669886,0.4817317,0.008063257,0.027409947,0.01498815,0.0011080356],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0878983,0.0012786735,0.00077566074,0.0030244973,0.0013663195,0.0039927275,0.0032441113,0.0012599132,0.0030210286],"category_scores_gemma":[0.28718144,0.0009113225,0.00108223,0.0021494923,0.0034564836,0.0081991125,0.0067630247,0.002562233,0.0011974112],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00086208654,0.0006046005,0.008941511,0.0039558527,0.00012969524,0.0004263759,0.17460777,0.014320415,0.017135764,0.029855838,0.002528147,0.7466319],"study_design_scores_gemma":[0.0008746142,0.0031136016,0.014214503,0.0049565514,0.00049527356,0.0031591821,0.15029895,0.41950473,0.08451139,0.19559437,0.12247554,0.00080129487],"about_ca_topic_score_codex":0.0017990385,"about_ca_topic_score_gemma":0.0036264495,"teacher_disagreement_score":0.0878983,"about_ca_system_score_codex":0.00371907,"about_ca_system_score_gemma":0.005510842,"threshold_uncertainty_score":0.4648562},"labels":[],"label_agreement":null},{"id":"W4402458066","doi":"10.1515/lingty-2024-2001","title":"Grammar Highlights 2023","year":2024,"lang":"en","type":"article","venue":"Linguistic Typology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Grammar; Linguistics; Philosophy","score_opus":0.010685327121638949,"score_gpt":0.28523761531493924,"score_spread":0.2745522881933003,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402458066","genre_codex":"editorial","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00041353304,0.004317684,0.0043695876,0.08577525,0.6311126,0.00026169288,0.0022539038,0.002172984,0.2693228],"genre_scores_gemma":[0.008781819,0.005155293,0.003974412,0.052826114,0.20310345,0.00046166702,0.0025334265,0.0048400634,0.7183237],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9955649,0.00067127374,0.00031780507,0.0005951025,0.0024606963,0.00039012145],"domain_scores_gemma":[0.98599905,0.0045927917,0.0008320247,0.000940317,0.0064501986,0.0011854413],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035016194,0.0014225737,0.00067949214,0.0023824333,0.003197198,0.009342416,0.0017775614,0.0035251935,0.20733738],"category_scores_gemma":[0.027108977,0.00054053456,0.0011662057,0.0011342808,0.002085354,0.0042176624,0.0028955815,0.0056265397,0.12559597],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000005655343,0.0000025999832,0.000015817706,0.000054166216,9.206184e-7,0.000032543634,0.000061898354,0.00001817784,0.00006392952,0.004114591,0.9887804,0.0068494333],"study_design_scores_gemma":[0.000001554173,0.0000017388473,0.000025190813,0.000056496272,7.2521686e-7,0.00002735407,0.000038118116,0.000014048853,0.000034284767,0.0007459162,0.9990507,0.0000037919767],"about_ca_topic_score_codex":0.0032977203,"about_ca_topic_score_gemma":0.0054112277,"teacher_disagreement_score":0.20733738,"about_ca_system_score_codex":0.003936688,"about_ca_system_score_gemma":0.0058086645,"threshold_uncertainty_score":0.69361264},"labels":[],"label_agreement":null},{"id":"W4402465635","doi":"10.29173/cais1849","title":"Imagining a Fuller Potential for Plain Language Summaries: A Case study of Canadian Science Publishing","year":2024,"lang":"en","type":"article","venue":"Proceedings of the Annual Conference of CAIS / Actes du congrès annuel de l ACSI","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":true,"ca_institutions":"University of Ottawa","funders":"","keywords":"Publishing; Plain language; History; Library science; Sociology; Political science; Computer science; Law","score_opus":0.021329706745875608,"score_gpt":0.27521166757372156,"score_spread":0.253881960827846,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402465635","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"reporting","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"reporting","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8829689,0.0019778835,0.0051450925,0.029441634,0.00021690525,0.0007270743,0.0009358644,0.0002501216,0.078336485],"genre_scores_gemma":[0.96653414,0.0019339714,0.013696037,0.0024091892,0.00008167023,0.0002048882,0.0004745104,0.000113326045,0.014552324],"study_design_codex":"qualitative","study_design_gemma":"case_report","domain_scores_codex":[0.96738976,0.019213706,0.0015524473,0.001155823,0.008520517,0.00216766],"domain_scores_gemma":[0.8380614,0.11562703,0.0083892085,0.005455991,0.02452552,0.007940907],"candidate_categories":["metaresearch","scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.021388061,0.00058375066,0.00049640797,0.005250905,0.024390528,0.0134295,0.0025040326,0.003431117,0.0046820687],"category_scores_gemma":[0.100901574,0.0004899799,0.0005564332,0.013496505,0.008644329,0.0069540055,0.00484615,0.0026961723,0.0007010905],"study_design_candidate":"case_report","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":true,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020508002,0.00024305198,0.015959995,0.0007518325,0.000039208913,0.009229547,0.88523304,0.000706765,0.0018982065,0.018156478,0.017407564,0.05016927],"study_design_scores_gemma":[0.000058950423,0.00016819505,0.012970627,0.0004910526,0.00006919471,0.0015787405,0.77059764,0.00081960205,0.0016021949,0.003469882,0.20804283,0.00013110555],"about_ca_topic_score_codex":0.7285962,"about_ca_topic_score_gemma":0.88265264,"teacher_disagreement_score":0.9865705,"about_ca_system_score_codex":0.047323268,"about_ca_system_score_gemma":0.061472714,"threshold_uncertainty_score":0.5460043},"labels":[],"label_agreement":null},{"id":"W4402474940","doi":"10.1109/ccece59415.2024.10667295","title":"Large Language Model Translation of Indigenous Languages","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Translation (biology); Indigenous; Linguistics; Natural language processing; Machine translation; Artificial intelligence; Philosophy","score_opus":0.011846132930103741,"score_gpt":0.2999957549146045,"score_spread":0.2881496219845008,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402474940","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.033146463,0.0027133387,0.84637904,0.0031366819,0.001101838,0.0010202772,0.038182475,0.037274774,0.037045214],"genre_scores_gemma":[0.38139048,0.0020992737,0.50024354,0.00094658684,0.00037767165,0.0009634035,0.08776832,0.0044324817,0.021778246],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9986192,0.00050386,0.00012344195,0.00033729535,0.00030261962,0.00011361544],"domain_scores_gemma":[0.99821234,0.0008137964,0.000093340175,0.00038730697,0.00042535283,0.000067825065],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013798013,0.0012432833,0.00072656403,0.0013811943,0.0013852095,0.0024862941,0.0018123327,0.0010928931,0.017847175],"category_scores_gemma":[0.004936835,0.0005509789,0.0015626707,0.0015581279,0.0009394828,0.0029277604,0.0028657373,0.0022376454,0.0093455985],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001061735,0.0005266812,0.0037719067,0.0032270707,0.00058176054,0.0039939554,0.003640983,0.10717692,0.021853467,0.13597743,0.2161763,0.5020117],"study_design_scores_gemma":[0.00015707577,0.00017321711,0.0017742474,0.0003666954,0.00014857347,0.0010268531,0.0011671417,0.7167272,0.009399737,0.12605199,0.14289528,0.00011203176],"about_ca_topic_score_codex":0.02596092,"about_ca_topic_score_gemma":0.03940441,"teacher_disagreement_score":0.02596092,"about_ca_system_score_codex":0.0017251176,"about_ca_system_score_gemma":0.0029295983,"threshold_uncertainty_score":0.05970478},"labels":[],"label_agreement":null},{"id":"W4402487814","doi":"10.3384/nejlt.2000-1533.2024.5217","title":"Documenting Geographically and Contextually Diverse Language Data Sources","year":2024,"lang":"en","type":"article","venue":"Northern European Journal of Language Technology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Sorbonne Université; Università Bocconi; Centre National de la Recherche Scientifique; Grand Équipement National De Calcul Intensif; Simon Fraser University","keywords":"Spatial contextual awareness; Context (archaeology); Geography; Remote sensing","score_opus":0.011623514470331659,"score_gpt":0.2669204921008031,"score_spread":0.25529697763047143,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402487814","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14721452,0.0010864297,0.77889717,0.003677502,0.00016250415,0.0032951368,0.0376208,0.0058153113,0.022230634],"genre_scores_gemma":[0.19747004,0.00058379455,0.75924337,0.00044405274,0.00007318078,0.0025639385,0.03466426,0.0012607733,0.003696623],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9779856,0.009104858,0.0040272176,0.003510428,0.0047534243,0.00061848375],"domain_scores_gemma":[0.9177834,0.04165714,0.0075557386,0.01622535,0.015351724,0.0014266885],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02702418,0.0009206197,0.0011962085,0.022892518,0.0038330078,0.00893464,0.0023445282,0.0011770884,0.0031381024],"category_scores_gemma":[0.06588727,0.001117426,0.0009540666,0.02172525,0.0019914925,0.010880614,0.009095831,0.002061008,0.0016997805],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003087536,0.00059580605,0.11406042,0.0032673778,0.00030591091,0.002311643,0.091193624,0.0071841436,0.024984267,0.09389028,0.030766545,0.6311313],"study_design_scores_gemma":[0.00017060731,0.00027731928,0.082400955,0.0022581299,0.000379736,0.003076494,0.09938381,0.048744783,0.04705425,0.12249271,0.59322083,0.000540409],"about_ca_topic_score_codex":0.013581217,"about_ca_topic_score_gemma":0.02901388,"teacher_disagreement_score":0.02702418,"about_ca_system_score_codex":0.0029675993,"about_ca_system_score_gemma":0.009662352,"threshold_uncertainty_score":0.14291924},"labels":[],"label_agreement":null},{"id":"W4402502868","doi":"10.48550/arxiv.2408.08896","title":"LLMJudge: LLMs for Relevance Judgments","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Engineering and Physical Sciences Research Council; University of Waterloo; Università degli Studi di Padova; Universiteit van Amsterdam; University College London","keywords":"Relevance (law); Psychology; Political science; Cognitive psychology; Social psychology; Law","score_opus":0.059173457496228536,"score_gpt":0.22006947681848313,"score_spread":0.16089601932225459,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402502868","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00566837,0.0017537634,0.8693114,0.0049084057,0.0013463519,0.00085206766,0.008135528,0.09641902,0.011605129],"genre_scores_gemma":[0.09489875,0.0007481835,0.8416609,0.0024841323,0.0009897274,0.0018585784,0.025117802,0.019846499,0.0123955095],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.95732236,0.024564026,0.0031288206,0.0043972423,0.009615577,0.0009719652],"domain_scores_gemma":[0.89047635,0.064456284,0.0026962557,0.02767572,0.012360215,0.00233506],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.027260061,0.0020840664,0.0017239173,0.004279318,0.002198576,0.0075974413,0.0049366476,0.005079692,0.040143523],"category_scores_gemma":[0.19917604,0.0013998009,0.0024373678,0.003582794,0.0025512304,0.012288041,0.012338791,0.006930022,0.03875458],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013253599,0.00034177306,0.00137505,0.0017095794,0.00013982026,0.00019542161,0.0011130633,0.010616731,0.007876952,0.06742908,0.36918792,0.5386893],"study_design_scores_gemma":[0.00055554934,0.00028197776,0.001438415,0.0005423168,0.000055023414,0.00029823033,0.00044395908,0.25400046,0.01367151,0.37702712,0.3514697,0.00021576793],"about_ca_topic_score_codex":0.0030606335,"about_ca_topic_score_gemma":0.0049980427,"teacher_disagreement_score":0.040143523,"about_ca_system_score_codex":0.0030530626,"about_ca_system_score_gemma":0.004372131,"threshold_uncertainty_score":0.14416671},"labels":[],"label_agreement":null},{"id":"W4402519936","doi":"10.1075/ijlcr.24010.paq","title":"The Core Metadata Schema for Learner Corpora (LC-meta)","year":2024,"lang":"en","type":"article","venue":"International Journal of Learner Corpus Research","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canarie","funders":"","keywords":"Metadata; Computer science; Schema (genetic algorithms); Information retrieval; Annotation; World Wide Web; Artificial intelligence","score_opus":0.22576869424191484,"score_gpt":0.4712745965182499,"score_spread":0.24550590227633504,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402519936","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029344397,0.0005759986,0.8791844,0.0039029268,0.0004995584,0.004398335,0.042248927,0.012410263,0.027435105],"genre_scores_gemma":[0.089417785,0.00059175317,0.83940345,0.0014080887,0.00017826418,0.004912674,0.051919244,0.003852716,0.008315968],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9845399,0.0062060254,0.004727648,0.0017177296,0.0024263973,0.0003822814],"domain_scores_gemma":[0.95492953,0.010011135,0.0037383502,0.017676279,0.012338774,0.0013059409],"candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.032135747,0.00057927927,0.00093291106,0.00533175,0.0020082386,0.0076931347,0.002999089,0.0018580217,0.006353126],"category_scores_gemma":[0.035820257,0.0011745442,0.0009474005,0.004815846,0.00232282,0.008995158,0.0057595996,0.002878771,0.0042914813],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00058971986,0.00045794016,0.026556233,0.0025072445,0.00016124443,0.0007716767,0.021295857,0.0069431732,0.025353605,0.49171346,0.114452034,0.30919793],"study_design_scores_gemma":[0.00007422345,0.00016601507,0.006641854,0.0012996611,0.00007840232,0.00086278876,0.0032341578,0.010812779,0.02195141,0.051696137,0.9030045,0.00017795367],"about_ca_topic_score_codex":0.0055580684,"about_ca_topic_score_gemma":0.0053197364,"teacher_disagreement_score":0.9923069,"about_ca_system_score_codex":0.0036299748,"about_ca_system_score_gemma":0.011569524,"threshold_uncertainty_score":0.1699521},"labels":[],"label_agreement":null},{"id":"W4402635148","doi":"10.1007/978-3-031-71908-0_8","title":"Overview of the CLEF 2024 JOKER Track","year":2024,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba","funders":"Agence Nationale de la Recherche","keywords":"Clef; Computer science; Track (disk drive); Information retrieval; Artificial intelligence; Computer graphics (images); Operating system; Engineering; Systems engineering","score_opus":0.02276968012044491,"score_gpt":0.2862518246063643,"score_spread":0.2634821444859194,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402635148","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010977546,0.07313871,0.08250938,0.027558036,0.021619162,0.0022655125,0.1564515,0.08845566,0.5370245],"genre_scores_gemma":[0.020631975,0.014725987,0.0752553,0.009472917,0.008041978,0.0017465259,0.42020193,0.02206037,0.42786303],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9959174,0.00095206476,0.00020898735,0.00066389696,0.001849606,0.00040812418],"domain_scores_gemma":[0.9952892,0.0010544799,0.00014701733,0.0009366917,0.0016588625,0.00091375236],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0066117244,0.0022214297,0.002352943,0.0094085885,0.0029275154,0.00610333,0.0037434816,0.003144834,0.24146652],"category_scores_gemma":[0.008819288,0.0013580266,0.001423315,0.0076402463,0.0006196408,0.0068264436,0.004034687,0.0029100897,0.21402052],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013119189,0.00006473031,0.00010394199,0.00023756395,0.000018148967,0.000028823293,0.000011887297,0.00024212015,0.0009564235,0.0013849228,0.9228735,0.073946804],"study_design_scores_gemma":[0.00011624843,0.00009389007,0.00088507554,0.000098469354,0.000017300998,0.00016992705,0.000020628453,0.0019312075,0.0014498525,0.0028598667,0.99231464,0.000042971253],"about_ca_topic_score_codex":0.008885126,"about_ca_topic_score_gemma":0.016812574,"teacher_disagreement_score":0.24146652,"about_ca_system_score_codex":0.0030642587,"about_ca_system_score_gemma":0.0032784408,"threshold_uncertainty_score":0.807786},"labels":[],"label_agreement":null},{"id":"W4402667043","doi":"10.18653/v1/2024.acl-long.37","title":"OPEx: A Component-Wise Analysis of LLM-Centric Agents in Embodied Instruction Following","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Component (thermodynamics); Embodied cognition; Computer science; Artificial intelligence; Physics","score_opus":0.02009801514821536,"score_gpt":0.3025195860109078,"score_spread":0.28242157086269243,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402667043","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.050695576,0.000095724965,0.9403335,0.00010779751,0.000033359098,0.00009412682,0.00032677283,0.0018573934,0.006455723],"genre_scores_gemma":[0.7177081,0.00013274713,0.2658494,0.000066819826,0.000026209174,0.00017367184,0.0006708554,0.00056489697,0.01480731],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9998031,0.000036327478,0.0000097930015,0.00005523037,0.00005349437,0.000042019696],"domain_scores_gemma":[0.99958414,0.00015212515,0.000046082812,0.00006796421,0.00011612485,0.000033426735],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00037043408,0.00059972843,0.00039793548,0.0007453667,0.0004928645,0.0012174891,0.0010242971,0.0005896605,0.0113376],"category_scores_gemma":[0.0017648847,0.00027401545,0.00055211276,0.00041732256,0.00048150463,0.001526699,0.0013625778,0.0007527934,0.0011557295],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008652488,0.0002319231,0.007962347,0.0003830028,0.00013169,0.0006144562,0.0010345771,0.23285735,0.04026877,0.27446145,0.0055623604,0.43562692],"study_design_scores_gemma":[0.000010847345,0.00008654277,0.0021023136,0.000022926959,0.00004036875,0.00006112538,0.00018546014,0.9113113,0.009356497,0.07325483,0.0035516338,0.000016037347],"about_ca_topic_score_codex":0.005736961,"about_ca_topic_score_gemma":0.004619121,"teacher_disagreement_score":0.0113376,"about_ca_system_score_codex":0.000531121,"about_ca_system_score_gemma":0.0011171955,"threshold_uncertainty_score":0.037928104},"labels":[],"label_agreement":null},{"id":"W4402669689","doi":"10.18653/v1/2024.iwslt-1.14","title":"UM IWSLT 2024 Low-Resource Speech Translation: Combining Maltese and North Levantine Arabic","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Maltese; Arabic; Computer science; Translation (biology); Natural language processing; Resource (disambiguation); Speech recognition; Artificial intelligence; Linguistics; Computer network; Philosophy","score_opus":0.01643098223594226,"score_gpt":0.2627284942965571,"score_spread":0.24629751206061484,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402669689","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2682938,0.0066128555,0.4259032,0.008969283,0.006033754,0.0019228217,0.036570404,0.09742362,0.14827034],"genre_scores_gemma":[0.3957459,0.0017436314,0.37476167,0.0011619914,0.0006217794,0.0009652375,0.09345382,0.011009538,0.12053636],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9990169,0.00040310435,0.00008621266,0.0001781417,0.00017630607,0.00013940215],"domain_scores_gemma":[0.9986953,0.00020668928,0.000036969628,0.00027926388,0.00066830416,0.000113537986],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002158689,0.0012082465,0.00082063524,0.0013849668,0.0014194051,0.0025011555,0.001117454,0.0011411204,0.023469603],"category_scores_gemma":[0.0027360702,0.0005057645,0.0005184381,0.0011273377,0.0006064169,0.0020780277,0.0027502482,0.00086459995,0.02023212],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0026364878,0.00040041603,0.0018903036,0.00072096335,0.00013801816,0.0015861217,0.0022101563,0.0038050718,0.13992107,0.0046684155,0.17159826,0.6704247],"study_design_scores_gemma":[0.00071535574,0.000946599,0.0058003054,0.0002443046,0.0003975311,0.0010708704,0.004483373,0.080034725,0.29707366,0.009114447,0.59988934,0.00022952721],"about_ca_topic_score_codex":0.016124949,"about_ca_topic_score_gemma":0.023060732,"teacher_disagreement_score":0.023469603,"about_ca_system_score_codex":0.0007906555,"about_ca_system_score_gemma":0.0020759127,"threshold_uncertainty_score":0.07851362},"labels":[],"label_agreement":null},{"id":"W4402669690","doi":"10.18653/v1/2024.findings-acl.906","title":"TAXI: Evaluating Categorical Knowledge Editing for Language Models","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"Arizona State University","keywords":"Computer science; Categorical variable; Natural language processing; Artificial intelligence; Machine learning","score_opus":0.05127591607564936,"score_gpt":0.38142122886814167,"score_spread":0.3301453127924923,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402669690","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.59857166,0.016547058,0.13778405,0.0038683596,0.0022214341,0.0017437526,0.092976965,0.11848599,0.027800813],"genre_scores_gemma":[0.6500206,0.0013576607,0.20166971,0.0012916159,0.0003511468,0.0005950296,0.13707915,0.0030792414,0.004555874],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9840845,0.006393216,0.0016548372,0.0034492349,0.003915612,0.00050256524],"domain_scores_gemma":[0.9298581,0.04947979,0.0027391873,0.012338898,0.0042181807,0.0013658312],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012462294,0.0022582768,0.0011906725,0.0041358825,0.0012315953,0.0030609667,0.004471181,0.0031015421,0.0029646126],"category_scores_gemma":[0.07909601,0.0005371639,0.0015488606,0.0032641417,0.0014896898,0.0060588545,0.0028036798,0.0032730184,0.0018364901],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0038029172,0.0023798503,0.07302657,0.0059362473,0.0022128045,0.0010219095,0.0018584807,0.22251919,0.019157853,0.010300645,0.23492925,0.42285421],"study_design_scores_gemma":[0.00079168583,0.0016452107,0.016835257,0.00032078123,0.0003553445,0.0012483363,0.00087666535,0.88633466,0.021842944,0.015573385,0.053940035,0.00023572534],"about_ca_topic_score_codex":0.016650653,"about_ca_topic_score_gemma":0.028233489,"teacher_disagreement_score":0.016650653,"about_ca_system_score_codex":0.0023798463,"about_ca_system_score_gemma":0.002427877,"threshold_uncertainty_score":0.06590766},"labels":[],"label_agreement":null},{"id":"W4402670124","doi":"10.18653/v1/2024.wassa-1.9","title":"MBIAS: Mitigating Bias in Large Language Models While Retaining Context","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Government of Canada","keywords":"Computer science; Context (archaeology); Geology","score_opus":0.04101104206431441,"score_gpt":0.30098565219099893,"score_spread":0.25997461012668455,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402670124","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05138556,0.0028209279,0.8783517,0.0045512193,0.0009655973,0.0010073871,0.009594146,0.04367715,0.007646367],"genre_scores_gemma":[0.4310714,0.0008591164,0.5288471,0.0035744575,0.0005369697,0.0016327996,0.020792056,0.004492538,0.008193494],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9881174,0.0076887785,0.00051733456,0.001965928,0.0013815431,0.00032890568],"domain_scores_gemma":[0.9655411,0.023649057,0.0012862294,0.0065894397,0.002357322,0.00057686213],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.014705395,0.0025799295,0.001165324,0.0018273027,0.0012305945,0.0037831583,0.0030693437,0.0026896477,0.0073496313],"category_scores_gemma":[0.058249634,0.0009177528,0.0018041373,0.0010202082,0.00201432,0.0059689684,0.00654868,0.0055554304,0.0062614824],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018561158,0.0005574859,0.017821448,0.0018499279,0.000765883,0.00058130606,0.0025135255,0.101438515,0.022202548,0.02972235,0.117911525,0.70277935],"study_design_scores_gemma":[0.00026942333,0.00053156697,0.0032543242,0.00045816335,0.00021890826,0.00047665538,0.0006140412,0.831399,0.023323175,0.089730516,0.049536075,0.0001881845],"about_ca_topic_score_codex":0.0052716634,"about_ca_topic_score_gemma":0.015085624,"teacher_disagreement_score":0.014705395,"about_ca_system_score_codex":0.0017566703,"about_ca_system_score_gemma":0.0024402728,"threshold_uncertainty_score":0.07777047},"labels":[],"label_agreement":null},{"id":"W4402670845","doi":"10.18653/v1/2024.findings-acl.115","title":"A Graph per Persona: Reasoning about Subjective Natural Language Descriptions","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Defense Advanced Research Projects Agency; Canadian Institute for Advanced Research","keywords":"Persona; Computer science; Natural language processing; Natural language; Artificial intelligence; Graph; Natural (archaeology); Programming language; Theoretical computer science; Human–computer interaction; History","score_opus":0.007859001521344656,"score_gpt":0.26895631810638704,"score_spread":0.26109731658504237,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402670845","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017481178,0.00026571637,0.9745394,0.0024742312,0.00006808255,0.000108777436,0.0010209742,0.0011747751,0.0028669166],"genre_scores_gemma":[0.46779555,0.0005455353,0.52274567,0.0009451181,0.00015371073,0.00020243236,0.002761304,0.0002903753,0.0045603355],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9981712,0.0008729003,0.00008746473,0.00049420755,0.00030985643,0.00006439049],"domain_scores_gemma":[0.99333173,0.004253893,0.00061309506,0.0009485964,0.0005790558,0.0002735558],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025608123,0.000956972,0.00056203635,0.0021455644,0.0009374316,0.0024621778,0.0018551544,0.0016526214,0.0040627704],"category_scores_gemma":[0.0147579275,0.0006199087,0.0019191306,0.001351164,0.0015690328,0.008154852,0.0016527033,0.0027615014,0.0009876753],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00034525726,0.00022558273,0.008240019,0.00057817966,0.00043219826,0.0008104919,0.0023371135,0.18322791,0.0062662703,0.49547577,0.02688558,0.27517557],"study_design_scores_gemma":[0.000013009918,0.00002677054,0.0005679684,0.000043257554,0.000054165357,0.0001312272,0.0001708756,0.6824816,0.0011058125,0.30855933,0.006816573,0.000029299295],"about_ca_topic_score_codex":0.010948574,"about_ca_topic_score_gemma":0.016416516,"teacher_disagreement_score":0.010948574,"about_ca_system_score_codex":0.0013376985,"about_ca_system_score_gemma":0.0011341376,"threshold_uncertainty_score":0.021769702},"labels":[],"label_agreement":null},{"id":"W4402671031","doi":"10.18653/v1/2024.acl-long.691","title":"Cheetah: Natural Language Generation for 517 African Languages","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; State Government of Victoria; Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Natural language generation; Computer science; Natural language; Linguistics; Natural language processing; Philosophy","score_opus":0.013876248973698201,"score_gpt":0.30758121256106136,"score_spread":0.29370496358736314,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402671031","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20842628,0.0012293189,0.61810154,0.0030148434,0.0005633551,0.0019402867,0.021472018,0.11233662,0.032915775],"genre_scores_gemma":[0.5241164,0.00039653163,0.43086317,0.00067132834,0.00005798485,0.00081203395,0.027510742,0.0026812933,0.012890519],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99961734,0.00015268761,0.000022537797,0.00010500563,0.00006980961,0.00003258204],"domain_scores_gemma":[0.9990263,0.00058184355,0.000041068666,0.00017239283,0.00013117031,0.00004728079],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011348694,0.00064373465,0.00034747962,0.0007350981,0.0006663688,0.0008731143,0.0011963373,0.00076589425,0.0129476115],"category_scores_gemma":[0.0031972986,0.00031670672,0.00066634885,0.00045783038,0.0005068103,0.0022647833,0.001994615,0.0012415798,0.002977813],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010646858,0.0005226499,0.01011571,0.001456665,0.00024875024,0.0015115014,0.002753179,0.08571617,0.032386422,0.04166129,0.17380361,0.64875937],"study_design_scores_gemma":[0.00029907917,0.00028132153,0.0031910285,0.000105360705,0.000079792226,0.00086448475,0.00080018817,0.80030316,0.026258318,0.030208977,0.1375152,0.00009308873],"about_ca_topic_score_codex":0.0043250085,"about_ca_topic_score_gemma":0.00914443,"teacher_disagreement_score":0.0129476115,"about_ca_system_score_codex":0.0006559195,"about_ca_system_score_gemma":0.0012192676,"threshold_uncertainty_score":0.0433141},"labels":[],"label_agreement":null},{"id":"W4402671170","doi":"10.18653/v1/2024.acl-long.781","title":"Emergent Word Order Universals from Cognitively-Motivated Language Models","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"Precursory Research for Embryonic Science and Technology","keywords":"Word order; Computer science; Problem of universals; Natural language processing; Word (group theory); Linguistics; Artificial intelligence; Linguistic universal; Cognitive science; Psychology; Theoretical linguistics; Philosophy","score_opus":0.015868285975450338,"score_gpt":0.2703472470005801,"score_spread":0.2544789610251298,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402671170","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6528035,0.0009137545,0.2846114,0.0031883135,0.00021997326,0.00008497767,0.0011857363,0.0009376319,0.0560546],"genre_scores_gemma":[0.9796373,0.00023246651,0.017803075,0.0001555467,0.000072454066,0.00006516536,0.00044668722,0.00018112523,0.0014061328],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99963284,0.000107004234,0.00002107816,0.000118925505,0.000054287128,0.00006582225],"domain_scores_gemma":[0.9980252,0.0010817503,0.00014422952,0.0003694755,0.00021864557,0.00016074705],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008821851,0.00034893543,0.00067656656,0.0010332926,0.0010861491,0.0018598841,0.0006933487,0.0005767665,0.004381055],"category_scores_gemma":[0.0053095906,0.0004238486,0.00079732237,0.0005645044,0.0016817999,0.004352442,0.0016912806,0.0013486987,0.0004450786],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000051514326,0.000048288162,0.0028928698,0.00008945127,0.000043808162,0.00022506734,0.0014984828,0.005185257,0.0035949817,0.97210103,0.0023857972,0.011883448],"study_design_scores_gemma":[0.000015056423,0.000010389532,0.0014015158,0.000010440016,0.000018099872,0.00009071132,0.00022162125,0.03923859,0.00037589867,0.9574505,0.0011517329,0.000015388448],"about_ca_topic_score_codex":0.0012865977,"about_ca_topic_score_gemma":0.0023212105,"teacher_disagreement_score":0.004381055,"about_ca_system_score_codex":0.0006868843,"about_ca_system_score_gemma":0.0006522214,"threshold_uncertainty_score":0.0146561265},"labels":[],"label_agreement":null},{"id":"W4402671201","doi":"10.18653/v1/2024.arabicnlp-1.11","title":"Towards Zero-Shot Text-To-Speech for Arabic Dialects","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Zero (linguistics); Arabic; Computer science; Natural language processing; Speech recognition; Artificial intelligence; Linguistics; Philosophy","score_opus":0.024523793653798878,"score_gpt":0.3145577619983406,"score_spread":0.29003396834454176,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402671201","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1646427,0.0030616384,0.79702896,0.00092780817,0.0009263033,0.00030653752,0.0041226265,0.023138734,0.005844717],"genre_scores_gemma":[0.6031367,0.00091623876,0.3666516,0.00068433647,0.00036678778,0.0003267984,0.015751328,0.0012063998,0.010959899],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9989151,0.00037906654,0.000055662338,0.0004314615,0.00014166524,0.00007714608],"domain_scores_gemma":[0.9976871,0.0012439853,0.00007811266,0.0003303052,0.00052108755,0.00013937747],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016758458,0.0015333936,0.000944417,0.0010597404,0.0006021386,0.0014773605,0.0014704431,0.0015009972,0.0034823255],"category_scores_gemma":[0.0050114635,0.00041065493,0.0011393325,0.0004466567,0.00056269945,0.00217294,0.001762388,0.0018634408,0.0049913814],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0024462256,0.00053293846,0.004861469,0.00083838863,0.00032105146,0.0006931141,0.0011096835,0.07903639,0.10552903,0.0052063963,0.023938097,0.7754872],"study_design_scores_gemma":[0.00006362856,0.0003310046,0.002459951,0.00006918767,0.00009378712,0.00048998307,0.00042708725,0.9358719,0.043813206,0.007553259,0.008761904,0.0000651333],"about_ca_topic_score_codex":0.005844029,"about_ca_topic_score_gemma":0.0076017417,"teacher_disagreement_score":0.005844029,"about_ca_system_score_codex":0.00069933984,"about_ca_system_score_gemma":0.0008587664,"threshold_uncertainty_score":0.011649489},"labels":[],"label_agreement":null},{"id":"W4402671205","doi":"10.18653/v1/2024.arabicnlp-1.4","title":"Exploiting Dialect Identification in Automatic Dialectal Text Normalization","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"York University; New York University Abu Dhabi","keywords":"Computer science; Normalization (sociology); Natural language processing; Identification (biology); Artificial intelligence; Speech recognition; Biology","score_opus":0.01044162544620815,"score_gpt":0.2726622949550494,"score_spread":0.26222066950884126,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402671205","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.31181583,0.002775527,0.6272297,0.0009140573,0.0010933327,0.0005229094,0.0058185277,0.03824314,0.011587054],"genre_scores_gemma":[0.64950913,0.00066513714,0.32104388,0.0005228559,0.00017704717,0.0003097582,0.014445428,0.0016588126,0.011667911],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9988186,0.00033701395,0.00007980014,0.0005573483,0.000114895556,0.000092342125],"domain_scores_gemma":[0.9981173,0.00085053575,0.00008555451,0.00040024717,0.00047248337,0.00007388903],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014158174,0.0013906572,0.0008071055,0.0014716422,0.00075014745,0.0018488581,0.0011116762,0.0011204354,0.004479496],"category_scores_gemma":[0.0042497613,0.0004595937,0.0011848097,0.0010386355,0.0007062312,0.002353629,0.0014176131,0.001993794,0.006547916],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008579729,0.0004022521,0.010910992,0.00043052543,0.00026698236,0.00065118214,0.0012493532,0.050794736,0.08316084,0.0040834886,0.02742554,0.81976604],"study_design_scores_gemma":[0.00007102273,0.00023628905,0.008562129,0.000060590657,0.00010403121,0.00093135436,0.0006758432,0.8822383,0.06959412,0.0103523405,0.027061472,0.000112430076],"about_ca_topic_score_codex":0.0085809985,"about_ca_topic_score_gemma":0.013243725,"teacher_disagreement_score":0.0085809985,"about_ca_system_score_codex":0.0009148158,"about_ca_system_score_gemma":0.0010640434,"threshold_uncertainty_score":0.017062128},"labels":[],"label_agreement":null},{"id":"W4402671356","doi":"10.18653/v1/2024.arabicnlp-1.18","title":"John vs. Ahmed: Debate-Induced Bias in Multilingual LLMs","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"Alliance de recherche numérique du Canada; Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Political science; Computer science","score_opus":0.038650880480335485,"score_gpt":0.3257001436754829,"score_spread":0.2870492631951474,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402671356","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7641789,0.0009927454,0.08005169,0.01551949,0.0005397628,0.0002800442,0.00042763219,0.00073165,0.13727811],"genre_scores_gemma":[0.9884625,0.0001252423,0.007816914,0.0009219367,0.000048982765,0.000056666664,0.000085428685,0.00019457008,0.0022878007],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9630549,0.02825893,0.0014195794,0.002170415,0.0041853082,0.0009108671],"domain_scores_gemma":[0.84206504,0.12805676,0.008949414,0.010361734,0.009374582,0.0011924458],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.025337286,0.00048486874,0.00049016613,0.0010521414,0.0026519443,0.0049326834,0.0008793749,0.0019878896,0.008426353],"category_scores_gemma":[0.15314865,0.00039350675,0.00038352824,0.0007837684,0.0038689699,0.009852735,0.0048363796,0.00294601,0.0012985151],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0025338838,0.0003711369,0.072397776,0.0015982803,0.00018131903,0.0024637585,0.26054868,0.0039943634,0.018280394,0.44089052,0.021016043,0.17572385],"study_design_scores_gemma":[0.00036312058,0.00066543766,0.03265296,0.0018851741,0.0003842435,0.0031642325,0.15585619,0.057459723,0.042571988,0.40309012,0.3013961,0.0005106766],"about_ca_topic_score_codex":0.0022665553,"about_ca_topic_score_gemma":0.0026406744,"teacher_disagreement_score":0.025337286,"about_ca_system_score_codex":0.0028090067,"about_ca_system_score_gemma":0.0018795647,"threshold_uncertainty_score":0.13399804},"labels":[],"label_agreement":null},{"id":"W4402671366","doi":"10.18653/v1/2024.arabicnlp-1.13","title":"Arabic Automatic Story Generation with Large Language Models","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Computer science; Arabic; Natural language processing; Artificial intelligence; Linguistics; Philosophy","score_opus":0.012673150787336291,"score_gpt":0.26252941410536923,"score_spread":0.24985626331803293,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402671366","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07040655,0.0010263548,0.87118256,0.00129862,0.00043942226,0.0004899936,0.008602708,0.030297289,0.01625658],"genre_scores_gemma":[0.5211055,0.000458895,0.45281976,0.00022147264,0.00014623073,0.0003931243,0.014233024,0.0013691518,0.009252812],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99943036,0.00022836162,0.00004335755,0.00015010938,0.00010607118,0.00004172527],"domain_scores_gemma":[0.99869955,0.00075324555,0.00005137472,0.00015583204,0.00028322628,0.000056731536],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00051494874,0.0011590454,0.0005761571,0.0010815149,0.00077647646,0.0013159028,0.0010294017,0.0008174439,0.015131823],"category_scores_gemma":[0.0034171299,0.00058934354,0.00095580786,0.00068295246,0.00036264106,0.002159324,0.001212033,0.0013955621,0.0061902055],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014423721,0.0003361316,0.0024560927,0.0010216762,0.00023107472,0.002151438,0.0009218947,0.13078049,0.040652994,0.037087407,0.07652761,0.70639086],"study_design_scores_gemma":[0.000078882156,0.000074544645,0.00043090936,0.000042234737,0.00005626743,0.00028537217,0.00020203173,0.94704473,0.015750408,0.020468537,0.015538442,0.000027547114],"about_ca_topic_score_codex":0.0031426388,"about_ca_topic_score_gemma":0.0046943934,"teacher_disagreement_score":0.015131823,"about_ca_system_score_codex":0.0006076301,"about_ca_system_score_gemma":0.00074439135,"threshold_uncertainty_score":0.050621033},"labels":[],"label_agreement":null},{"id":"W4402671393","doi":"10.18653/v1/2024.arabicnlp-1.80","title":"Arabic Train at NADI 2024 shared task: LLMs’ Ability to Translate Arabic Dialects into Modern Standard Arabic","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Arabic; Task (project management); Modern Standard Arabic; Computer science; Arabic languages; Natural language processing; Linguistics; History; Engineering; Philosophy","score_opus":0.011274656232142174,"score_gpt":0.2841371138451353,"score_spread":0.2728624576129931,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402671393","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8713946,0.0019354579,0.030916356,0.002141059,0.00091314124,0.00051172124,0.013565768,0.017899245,0.060722705],"genre_scores_gemma":[0.93066454,0.00016397049,0.030694611,0.00072172924,0.000075555254,0.00031386534,0.022673106,0.00050875364,0.0141838],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9979482,0.00092181674,0.00014056449,0.0005451566,0.00026508953,0.00017908345],"domain_scores_gemma":[0.99339104,0.0032970472,0.00023783663,0.0016062788,0.00080522196,0.0006626273],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035349592,0.0015142276,0.0009196501,0.0007524961,0.0010749825,0.001940253,0.0014925736,0.0019874347,0.009372615],"category_scores_gemma":[0.016327923,0.00030393564,0.00062517414,0.0006117859,0.00065045065,0.0037276575,0.0035891815,0.0024670968,0.007892421],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004077401,0.0020728435,0.028225293,0.0015102664,0.00048217142,0.0011478649,0.0044624475,0.034719154,0.050562203,0.0045133103,0.19188608,0.676341],"study_design_scores_gemma":[0.0011797389,0.004284585,0.05928546,0.00042319007,0.0004617332,0.0021851843,0.008565768,0.5724886,0.14237265,0.025427694,0.18263559,0.0006898503],"about_ca_topic_score_codex":0.012142484,"about_ca_topic_score_gemma":0.017625984,"teacher_disagreement_score":0.012142484,"about_ca_system_score_codex":0.0011479072,"about_ca_system_score_gemma":0.0018492786,"threshold_uncertainty_score":0.031354547},"labels":[],"label_agreement":null},{"id":"W4402671442","doi":"10.18653/v1/2024.arabicnlp-1.25","title":"From Nile Sands to Digital Hands: Machine Translation of Coptic Texts","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Machine translation; Computer science; Translation (biology); Natural language processing; Artificial intelligence; Chemistry","score_opus":0.00950545921014078,"score_gpt":0.2676956908769667,"score_spread":0.2581902316668259,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402671442","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6199531,0.004134713,0.20663935,0.0036014854,0.0029007634,0.0014957603,0.02767532,0.04533444,0.08826506],"genre_scores_gemma":[0.67425126,0.0010766516,0.2314261,0.00079251453,0.00037462023,0.0007317112,0.06665604,0.0026365204,0.022054609],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9987478,0.00053334574,0.000100525314,0.00031921847,0.00019875656,0.00010042729],"domain_scores_gemma":[0.9977671,0.00096410967,0.00009047782,0.0005136649,0.0005687808,0.00009579996],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010926473,0.0011340707,0.0006895458,0.0015767337,0.001398992,0.00185147,0.0010338874,0.0012557063,0.012628474],"category_scores_gemma":[0.0066971574,0.00044423068,0.0006399027,0.0016019287,0.0009302908,0.0028095334,0.0024149423,0.0015585679,0.010228187],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013817167,0.0008744159,0.0035908385,0.0021789062,0.00016614702,0.00229952,0.0039468217,0.022298852,0.040761065,0.010246629,0.14626133,0.7659938],"study_design_scores_gemma":[0.000668146,0.00086427334,0.011738658,0.0006252085,0.0001927697,0.0021801763,0.008438235,0.56837064,0.12926485,0.028177097,0.2492295,0.00025054518],"about_ca_topic_score_codex":0.0068276403,"about_ca_topic_score_gemma":0.010836655,"teacher_disagreement_score":0.012628474,"about_ca_system_score_codex":0.0008077779,"about_ca_system_score_gemma":0.0013351318,"threshold_uncertainty_score":0.0422464},"labels":[],"label_agreement":null},{"id":"W4402671446","doi":"10.18653/v1/2024.arabicnlp-1.27","title":"Dallah: A Dialect-Aware Multimodal Large Language Model for Arabic","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Arabic; Computer science; Natural language processing; Linguistics; Artificial intelligence; Speech recognition; Philosophy","score_opus":0.013746684461389673,"score_gpt":0.30489965522193296,"score_spread":0.2911529707605433,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402671446","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.043564998,0.00073464034,0.91032624,0.0007563105,0.0003032048,0.000317655,0.003150846,0.03156676,0.009279386],"genre_scores_gemma":[0.38200045,0.00054064236,0.5816019,0.0006258894,0.000090281676,0.0007509326,0.008182444,0.0019021548,0.02430523],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9998072,0.00006758707,0.000012232898,0.00005755108,0.00003842106,0.00001706704],"domain_scores_gemma":[0.999749,0.00010959818,0.000015066236,0.000034008382,0.000065663735,0.000026652073],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00038541178,0.0007300754,0.00034368396,0.0003203462,0.00042491377,0.00092970417,0.0010427147,0.0005428378,0.007353027],"category_scores_gemma":[0.0014013074,0.00023550243,0.00070104445,0.0002081703,0.00025295248,0.0010492967,0.0010918769,0.0010601272,0.0029455465],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011418046,0.00037656902,0.0031234291,0.00049150124,0.00027867104,0.0007203669,0.0010655448,0.30121318,0.041564662,0.033167582,0.06687428,0.54998237],"study_design_scores_gemma":[0.000038180915,0.00006564674,0.00029808425,0.00001604991,0.000026049589,0.000096050724,0.0000889158,0.96822363,0.004959862,0.0060991812,0.020057365,0.00003090186],"about_ca_topic_score_codex":0.00926529,"about_ca_topic_score_gemma":0.017183706,"teacher_disagreement_score":0.00926529,"about_ca_system_score_codex":0.00063695206,"about_ca_system_score_gemma":0.0008277261,"threshold_uncertainty_score":0.02459836},"labels":[],"label_agreement":null},{"id":"W4402671457","doi":"10.18653/v1/2024.arabicnlp-1.79","title":"NADI 2024: The Fifth Nuanced Arabic Dialect Identification Shared Task","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs; UK Research and Innovation","keywords":"Arabic; Identification (biology); Task (project management); Computer science; Linguistics; Natural language processing; Artificial intelligence; Philosophy; Engineering","score_opus":0.011262454846642168,"score_gpt":0.2741339424653891,"score_spread":0.2628714876187469,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402671457","genre_codex":"empirical","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.49300805,0.0032474448,0.22662415,0.008705873,0.00733707,0.016174862,0.09102345,0.02424653,0.12963253],"genre_scores_gemma":[0.43186128,0.0002979129,0.3557493,0.002429537,0.00074293005,0.013161029,0.14289984,0.002809092,0.050049074],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.98246366,0.008169289,0.0009172036,0.0030595593,0.0035097224,0.0018806016],"domain_scores_gemma":[0.9707781,0.006127778,0.00090935343,0.006826161,0.008301509,0.0070571164],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.020938966,0.0021868069,0.0022091758,0.0025444648,0.003950083,0.004521771,0.0039164433,0.002944418,0.010967896],"category_scores_gemma":[0.028693747,0.00072432787,0.001698981,0.0014930655,0.0018969967,0.0038425147,0.015599646,0.0045652078,0.011009987],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.006681101,0.004643108,0.029373463,0.002722153,0.00074898347,0.0014800613,0.016310168,0.013772804,0.074361525,0.013640635,0.394577,0.44168904],"study_design_scores_gemma":[0.0024198096,0.0052625486,0.055760656,0.000548571,0.0003879777,0.0018405722,0.015024612,0.060039412,0.0729591,0.028729238,0.756252,0.00077560253],"about_ca_topic_score_codex":0.014078141,"about_ca_topic_score_gemma":0.018947145,"teacher_disagreement_score":0.020938966,"about_ca_system_score_codex":0.0035637112,"about_ca_system_score_gemma":0.009310312,"threshold_uncertainty_score":0.110737145},"labels":[],"label_agreement":null},{"id":"W4402671797","doi":"10.18653/v1/2024.acl-long.372","title":"Translation-based Lexicalization Generation and Lexical Gap Detection: Application to Kinship Terms","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Machine Intelligence Institute","keywords":"Lexicalization; Kinship; Computer science; Translation (biology); Natural language processing; Artificial intelligence; Machine translation; Linguistics; Sociology; Philosophy","score_opus":0.033474632355703214,"score_gpt":0.3022512037481904,"score_spread":0.26877657139248723,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402671797","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1118903,0.00046173905,0.87086064,0.0003492088,0.00011695898,0.00037017657,0.0006610695,0.011140505,0.004149357],"genre_scores_gemma":[0.33693507,0.00015442101,0.65778697,0.00012827163,0.000041614243,0.00021382123,0.0016128842,0.00093655096,0.002190304],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9986499,0.00047754403,0.00015140878,0.00032497512,0.00031344246,0.00008274621],"domain_scores_gemma":[0.9969373,0.0014485408,0.00019618157,0.0005758722,0.00074916246,0.00009287088],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013787484,0.00070585747,0.0008799893,0.0024990493,0.0010080131,0.0013842032,0.0010908055,0.00089904026,0.005154212],"category_scores_gemma":[0.0076332106,0.0003859578,0.00055759115,0.002373228,0.0008622622,0.0020372393,0.0026325015,0.00085036026,0.0023417226],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00034815248,0.00023268332,0.0050363378,0.0006531539,0.000092284616,0.0011887752,0.0024340672,0.012858703,0.040495165,0.017337233,0.009122792,0.9102006],"study_design_scores_gemma":[0.00022440206,0.00023409993,0.0052496544,0.00012870727,0.00012307773,0.0025212774,0.0018080224,0.8133904,0.0780091,0.06787852,0.03030684,0.0001259054],"about_ca_topic_score_codex":0.0021758687,"about_ca_topic_score_gemma":0.0034420616,"teacher_disagreement_score":0.005154212,"about_ca_system_score_codex":0.0005694985,"about_ca_system_score_gemma":0.0010725188,"threshold_uncertainty_score":0.01724261},"labels":[],"label_agreement":null},{"id":"W4402683050","doi":"10.18653/v1/2024.kallm-1.12","title":"Fine-tuning Language Models for Triple Extraction with Data Augmentation","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Extraction (chemistry); Chemistry","score_opus":0.04681412255596676,"score_gpt":0.3526810946123056,"score_spread":0.30586697205633884,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402683050","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17737101,0.001836965,0.6903947,0.0017682665,0.00066704914,0.0012258515,0.015404988,0.10422583,0.0071052774],"genre_scores_gemma":[0.33227918,0.00053651,0.6120121,0.0011766189,0.000105231105,0.0012962752,0.04535361,0.0027198924,0.0045205397],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9974477,0.00089587347,0.0002445837,0.0008833967,0.0003747224,0.00015375805],"domain_scores_gemma":[0.99186474,0.0049124956,0.00027239896,0.0016238167,0.0011742071,0.00015226257],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004635173,0.0023232433,0.001073695,0.0018916028,0.0006759343,0.0019538929,0.0029517806,0.0013500301,0.0032824418],"category_scores_gemma":[0.016132629,0.0007856105,0.002205773,0.0016167971,0.0008549747,0.0048519783,0.002566896,0.0038377317,0.0050216527],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012207633,0.0014391255,0.015523143,0.0014708177,0.00070609525,0.00089733425,0.0012040128,0.20313297,0.05229438,0.0062709516,0.06502756,0.6508128],"study_design_scores_gemma":[0.00012885942,0.00018716136,0.0013597301,0.00009532339,0.00016155042,0.00027170416,0.0004327327,0.92073435,0.044612356,0.010326042,0.021608654,0.00008154406],"about_ca_topic_score_codex":0.008120813,"about_ca_topic_score_gemma":0.014624772,"teacher_disagreement_score":0.008120813,"about_ca_system_score_codex":0.001145325,"about_ca_system_score_gemma":0.002307267,"threshold_uncertainty_score":0.024513423},"labels":[],"label_agreement":null},{"id":"W4402684015","doi":"10.18653/v1/2024.acl-long.188","title":"Collaboration or Corporate Capture? Quantifying NLP’s Reliance on Industry Artifacts and Contributions","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Natural language processing; Artificial intelligence","score_opus":0.05251448698163644,"score_gpt":0.3398627932147888,"score_spread":0.28734830623315233,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402684015","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"incentives","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"incentives","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.566673,0.030932505,0.120644145,0.03705717,0.0011828417,0.00048574957,0.0073659373,0.0015498094,0.23410885],"genre_scores_gemma":[0.9503938,0.0085292645,0.028184367,0.0015032701,0.0008667775,0.00032424123,0.0040881787,0.0006651926,0.005444873],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.92529136,0.037082814,0.007774902,0.0069912886,0.021084214,0.0017753472],"domain_scores_gemma":[0.45659754,0.41277248,0.041149806,0.056476492,0.029110324,0.0038932986],"candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.074187994,0.00064837665,0.0008571914,0.018356994,0.002392566,0.018670445,0.002112574,0.0026802553,0.005257343],"category_scores_gemma":[0.3187877,0.00075381965,0.000880947,0.03432789,0.003777667,0.027772548,0.01055148,0.002256023,0.0026569623],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00030710324,0.00021410789,0.3322673,0.0049247188,0.00074117165,0.0017934755,0.042096406,0.0053533483,0.0063452264,0.09681822,0.028529042,0.48060986],"study_design_scores_gemma":[0.00009184668,0.0004307653,0.24351312,0.0054379026,0.0009571946,0.0048904195,0.05938551,0.026355984,0.017406909,0.114486,0.526717,0.0003273663],"about_ca_topic_score_codex":0.0022088205,"about_ca_topic_score_gemma":0.0032943145,"teacher_disagreement_score":0.981643,"about_ca_system_score_codex":0.0022862223,"about_ca_system_score_gemma":0.00375317,"threshold_uncertainty_score":0.3923483},"labels":[],"label_agreement":null},{"id":"W4402684242","doi":"10.18653/v1/2024.arabicnlp-1.23","title":"On the Utility of Pretraining Language Models on Synthetic Data","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Computer science; Natural language processing; Data modeling; Language model; Artificial intelligence; Data science; Database","score_opus":0.06481140776400941,"score_gpt":0.3232609385198948,"score_spread":0.2584495307558854,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402684242","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.67066705,0.00408513,0.28950992,0.0023066504,0.00070039957,0.0005661411,0.006390669,0.012912936,0.0128611205],"genre_scores_gemma":[0.851197,0.0010999402,0.1233053,0.00079647574,0.000098075645,0.0004928532,0.019104984,0.00061742344,0.003287884],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99747616,0.001480992,0.00013873712,0.0005163612,0.0002543503,0.000133361],"domain_scores_gemma":[0.9877751,0.008946494,0.0002813289,0.0016153379,0.0012154711,0.00016626922],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005278997,0.0025538004,0.0008298142,0.00093766744,0.0007314947,0.0015927757,0.0018937179,0.0019359151,0.0028109604],"category_scores_gemma":[0.022421824,0.00083186623,0.0009508604,0.00092691724,0.00111992,0.004008265,0.0016726492,0.0030775664,0.0018072237],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005198837,0.00063481345,0.008794424,0.0005059558,0.00047141017,0.00025549665,0.00023678936,0.7963132,0.012372964,0.0024523234,0.010320882,0.1671219],"study_design_scores_gemma":[0.0000543224,0.0002960148,0.0019475978,0.000073650466,0.000060509086,0.00009765163,0.00014694687,0.9748584,0.016851133,0.0020529332,0.0035117695,0.00004909296],"about_ca_topic_score_codex":0.016399233,"about_ca_topic_score_gemma":0.026192069,"teacher_disagreement_score":0.016399233,"about_ca_system_score_codex":0.00114987,"about_ca_system_score_gemma":0.0013666474,"threshold_uncertainty_score":0.032607555},"labels":[],"label_agreement":null},{"id":"W4402704556","doi":"10.1109/cvpr52733.2024.02578","title":"DIEM: Decomposition-Integration Enhancing Multimodal Insights","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of New Brunswick","funders":"","keywords":"Decomposition; Computer science; Artificial intelligence; Human–computer interaction; Chemistry","score_opus":0.007621317954046461,"score_gpt":0.2900166391867566,"score_spread":0.2823953212327101,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402704556","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022202438,0.001094728,0.9610865,0.0005820273,0.00006278708,0.0002928802,0.0007272418,0.011357136,0.0025943008],"genre_scores_gemma":[0.16309427,0.00039981797,0.83126634,0.00045727927,0.000056351404,0.000287941,0.0018078882,0.0003301247,0.0023000673],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99862385,0.00049269805,0.000072657516,0.00040864787,0.00029494258,0.000107208325],"domain_scores_gemma":[0.99788326,0.0012337614,0.00013986697,0.0002919003,0.00033796186,0.00011315468],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028418216,0.001641733,0.0007712065,0.0027894103,0.0003767519,0.0016205282,0.0014258104,0.0015964166,0.0047759744],"category_scores_gemma":[0.008164346,0.00047982318,0.0018266106,0.0010227774,0.0007849296,0.0035093315,0.0040623033,0.0021511733,0.001456848],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005025473,0.00040416833,0.004525713,0.00078491797,0.00019166995,0.00024652205,0.0013415788,0.020236213,0.045275558,0.014337658,0.012118045,0.9000354],"study_design_scores_gemma":[0.0000951498,0.00045597745,0.0061303526,0.00015201358,0.00019700799,0.00067517237,0.0007183627,0.85581815,0.041956607,0.057381894,0.03631094,0.000108387314],"about_ca_topic_score_codex":0.0022122273,"about_ca_topic_score_gemma":0.0037659165,"teacher_disagreement_score":0.0047759744,"about_ca_system_score_codex":0.00088792975,"about_ca_system_score_gemma":0.0007869832,"threshold_uncertainty_score":0.015977204},"labels":[],"label_agreement":null},{"id":"W4402705719","doi":"10.48550/arxiv.2408.15390","title":"Avoiding abelian and additive powers in rich words","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Abelian group; Mathematics; Pure mathematics; Arithmetic; Linguistics; Philosophy","score_opus":0.03087361768892987,"score_gpt":0.20250510672238037,"score_spread":0.1716314890334505,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402705719","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.624922,0.0005639873,0.31448886,0.0007448926,0.00015878331,0.000064864755,0.000206046,0.0006403333,0.058210284],"genre_scores_gemma":[0.9505185,0.00022204744,0.031411905,0.00023067316,0.00016003213,0.000090381676,0.00024121579,0.00030842083,0.016816845],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9981863,0.0004787274,0.00017984938,0.00033821346,0.00046321764,0.00035370988],"domain_scores_gemma":[0.99624264,0.0022039786,0.00035919787,0.0005512921,0.00036704892,0.00027581683],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010173838,0.0006810825,0.00073619344,0.0012142649,0.0017787743,0.0025292893,0.00081435655,0.0007806568,0.0059930934],"category_scores_gemma":[0.0041587315,0.0006550945,0.0013465675,0.0006778819,0.003717694,0.0067188516,0.003819259,0.0020873186,0.0013047719],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015367291,0.000027042614,0.00059149106,0.00010350333,0.000019031791,0.00062363513,0.0009789901,0.0021276637,0.006133495,0.97947156,0.00048254136,0.009287251],"study_design_scores_gemma":[0.000022971892,0.00006190652,0.00023574443,0.00002813161,0.000034027107,0.0005001736,0.00026661882,0.0054199225,0.006579826,0.980793,0.0060271574,0.00003057935],"about_ca_topic_score_codex":0.0004270994,"about_ca_topic_score_gemma":0.00055840675,"teacher_disagreement_score":0.0059930934,"about_ca_system_score_codex":0.000665155,"about_ca_system_score_gemma":0.00037086694,"threshold_uncertainty_score":0.020048916},"labels":[],"label_agreement":null},{"id":"W4402705760","doi":"10.48550/arxiv.2408.15417","title":"Implicit Geometry of Next-token Prediction: From Language Sparsity Patterns to Model Representations","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Security token; Computer science; Geometry; Artificial intelligence; Mathematics; Computer network","score_opus":0.0584834549110922,"score_gpt":0.2321547135932132,"score_spread":0.17367125868212102,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402705760","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09582778,0.00030164217,0.9002243,0.0011771009,0.000039121056,0.000047898713,0.0003527331,0.0009150501,0.0011143001],"genre_scores_gemma":[0.8250816,0.0003170602,0.17066342,0.00032373852,0.000099140256,0.00014054422,0.0014181738,0.00026454122,0.0016917702],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988356,0.0005952823,0.000049427268,0.0002830823,0.00013327252,0.00010334807],"domain_scores_gemma":[0.9920322,0.0053406856,0.0007179772,0.0011487859,0.00046758872,0.00029279405],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019205697,0.000776969,0.00093448337,0.0007831659,0.00062268437,0.0015929139,0.0015830329,0.0010149808,0.0018342328],"category_scores_gemma":[0.018894957,0.0007773483,0.0006919443,0.0009648376,0.0020145732,0.004685735,0.0027976034,0.0033161938,0.00071170396],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003824253,0.00022361877,0.016018474,0.00029039328,0.00013603127,0.00036412658,0.00085150514,0.6765315,0.010010478,0.067819074,0.0071975742,0.22017483],"study_design_scores_gemma":[0.000006529819,0.000021941212,0.00044110575,0.000009776167,0.0000041465264,0.000023920875,0.000039766946,0.9536313,0.000867659,0.04458082,0.00036460056,0.000008547295],"about_ca_topic_score_codex":0.0042738966,"about_ca_topic_score_gemma":0.006546946,"teacher_disagreement_score":0.0042738966,"about_ca_system_score_codex":0.0010605783,"about_ca_system_score_gemma":0.0010735119,"threshold_uncertainty_score":0.010157049},"labels":[],"label_agreement":null},{"id":"W4402715008","doi":"10.18653/v1/2024.sigdial-1.1","title":"Dialogue Discourse Parsing as Generation: A Sequence-to-Sequence LLM-based Approach","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Alliance de recherche numérique du Canada","keywords":"Computer science; Sequence (biology); Parsing; Natural language processing; Artificial intelligence; Programming language","score_opus":0.06300484498857047,"score_gpt":0.3450055187298891,"score_spread":0.28200067374131865,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402715008","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.002670765,0.00013978794,0.98630875,0.00025902633,0.00006980494,0.00015169918,0.0002243857,0.007594959,0.0025808702],"genre_scores_gemma":[0.14235425,0.00018493707,0.84858894,0.00043375103,0.00013677521,0.0002811321,0.0011323914,0.0016772993,0.005210476],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9968104,0.0016222137,0.00019084795,0.00056737306,0.00062709436,0.00018210769],"domain_scores_gemma":[0.9955058,0.0027034222,0.00020627573,0.0007669315,0.0007057594,0.00011181312],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024978942,0.0011751793,0.0015198686,0.0023022406,0.0010533299,0.0027675615,0.0030147429,0.002059162,0.012288013],"category_scores_gemma":[0.007997568,0.0008365585,0.0016024656,0.0013145555,0.0012611834,0.0030113526,0.0029964151,0.0019751894,0.0046273763],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00062438595,0.00046979001,0.0016070295,0.0008221797,0.00017106837,0.0008566631,0.0013107659,0.078518726,0.03767459,0.105477765,0.022161573,0.7503055],"study_design_scores_gemma":[0.00004150379,0.00011288897,0.00032565382,0.00006591773,0.00008739887,0.00023617306,0.00017260542,0.89223284,0.023740752,0.06822491,0.014710198,0.00004911538],"about_ca_topic_score_codex":0.00243882,"about_ca_topic_score_gemma":0.0032346176,"teacher_disagreement_score":0.012288013,"about_ca_system_score_codex":0.0010402581,"about_ca_system_score_gemma":0.0017837281,"threshold_uncertainty_score":0.041107476},"labels":[],"label_agreement":null},{"id":"W4402755500","doi":"","title":"McCATMuS : retours sur la production d'un méta-dataset multilingue et multiséculaire","year":2024,"lang":"fr","type":"article","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Political science","score_opus":0.01629570940677649,"score_gpt":0.27016299519610254,"score_spread":0.25386728578932605,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402755500","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.074763566,0.005103599,0.1826821,0.006362784,0.0044854144,0.0013776576,0.52494806,0.18544729,0.014829541],"genre_scores_gemma":[0.05156534,0.0010300054,0.24384566,0.0010368343,0.00024421205,0.0010984938,0.6793941,0.013701579,0.008083807],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9935861,0.0016673767,0.0005448071,0.0019152281,0.0018631043,0.00042325997],"domain_scores_gemma":[0.98637855,0.0054283263,0.00026457393,0.00380963,0.0032562271,0.0008626995],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0052435803,0.004674617,0.0023620888,0.005987784,0.0024885978,0.004807215,0.0051019164,0.003664224,0.016196776],"category_scores_gemma":[0.02464406,0.0018137133,0.004359287,0.004840345,0.0010562807,0.0061231093,0.00483677,0.0042220782,0.014357483],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012677398,0.0006745278,0.004764152,0.003149554,0.0014187787,0.0012481859,0.00078515994,0.011236282,0.0293225,0.0056086336,0.754946,0.18557848],"study_design_scores_gemma":[0.0011033698,0.0006297768,0.018932927,0.0009594337,0.0010498108,0.0017415574,0.002274112,0.2401387,0.05655596,0.018491808,0.6577501,0.00037238814],"about_ca_topic_score_codex":0.063119456,"about_ca_topic_score_gemma":0.106693596,"teacher_disagreement_score":0.063119456,"about_ca_system_score_codex":0.0022990324,"about_ca_system_score_gemma":0.004442494,"threshold_uncertainty_score":0.12550414},"labels":[],"label_agreement":null},{"id":"W4402775761","doi":"10.1109/cvpr52733.2024.01497","title":"Transferable and Principled Efficiency for Open-Vocabulary Segmentation","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Fundamental Research Funds for the Central Universities","keywords":"Computer science; Segmentation; Vocabulary; Artificial intelligence; Natural language processing; Linguistics","score_opus":0.018343700402388512,"score_gpt":0.3150522730493097,"score_spread":0.2967085726469212,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402775761","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02057528,0.00082210096,0.95010775,0.0004057898,0.00018738056,0.00016119667,0.00065249513,0.018800233,0.008287673],"genre_scores_gemma":[0.3201851,0.00056025555,0.66350496,0.00057401106,0.00014611013,0.00032571823,0.004056237,0.004887388,0.005760305],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9977976,0.0004242457,0.0001527449,0.0007507012,0.0005829319,0.0002917709],"domain_scores_gemma":[0.9970161,0.0012242214,0.00013766444,0.00109701,0.00038536038,0.00013976888],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002059314,0.0018579395,0.0015270655,0.0014470216,0.0010464834,0.003310778,0.004091047,0.0023661093,0.008664633],"category_scores_gemma":[0.0094183,0.0010194747,0.0015985203,0.0015434149,0.0019680106,0.0059004673,0.004565234,0.0027818615,0.007130724],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00050007633,0.0002916025,0.002173648,0.0007202902,0.00017267265,0.0004668887,0.0008667069,0.1967637,0.059125785,0.058147997,0.028434614,0.6523361],"study_design_scores_gemma":[0.000059173846,0.00010642571,0.00039508552,0.00006025222,0.000040004066,0.00027711235,0.00021823021,0.874593,0.027715506,0.083226465,0.013263109,0.000045600937],"about_ca_topic_score_codex":0.00648696,"about_ca_topic_score_gemma":0.010150567,"teacher_disagreement_score":0.008664633,"about_ca_system_score_codex":0.0015355488,"about_ca_system_score_gemma":0.002273795,"threshold_uncertainty_score":0.028986096},"labels":[],"label_agreement":null},{"id":"W4402807503","doi":"10.1109/tdsc.2024.3449641","title":"NeuroYara: Learning to Rank for Yara Rules Generation Through Deep Language Modeling and Discriminative N-Gram Encoding","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Dependable and Secure Computing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Defence Research and Development Canada; Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Discriminative model; n-gram; Computer science; Encoding (memory); Gram; Rank (graph theory); Artificial intelligence; Language model; Natural language processing; Mathematics","score_opus":0.024654335472194734,"score_gpt":0.2976123308965028,"score_spread":0.27295799542430804,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402807503","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.027084006,0.00072839786,0.9451281,0.0003253325,0.00013398047,0.00017725423,0.0008469588,0.023835067,0.0017409885],"genre_scores_gemma":[0.29172048,0.0005652414,0.694317,0.00060487416,0.00014211582,0.00042427419,0.0045039603,0.0009940434,0.006728032],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986958,0.00027552078,0.000119934135,0.00040067837,0.00036412198,0.00014407957],"domain_scores_gemma":[0.9974853,0.0010292036,0.00028943922,0.0004399216,0.00063498865,0.00012121363],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014424799,0.0016924653,0.0013744949,0.002163565,0.0006468235,0.0019347878,0.0026133547,0.0013605859,0.0030798668],"category_scores_gemma":[0.006230668,0.0006448138,0.0011203415,0.0011441164,0.0006557492,0.0025334118,0.0012510737,0.0023513932,0.003929606],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035319716,0.00037716172,0.003253749,0.00020166767,0.000129039,0.00024571468,0.0001617123,0.09618541,0.014022971,0.006400487,0.014025599,0.8646433],"study_design_scores_gemma":[0.000016714506,0.0000713959,0.00023827294,0.000017040953,0.000016703307,0.000080231395,0.000024745235,0.98587555,0.0068672663,0.0050466806,0.0017237125,0.000021713804],"about_ca_topic_score_codex":0.0074434495,"about_ca_topic_score_gemma":0.014235789,"teacher_disagreement_score":0.0074434495,"about_ca_system_score_codex":0.0009652904,"about_ca_system_score_gemma":0.0019902892,"threshold_uncertainty_score":0.0148002505},"labels":[],"label_agreement":null},{"id":"W4402817547","doi":"10.3917/lfa.226.0069","title":"Écrire « à la manière de… » avec ChatGPT au secondaire québécois","year":2024,"lang":"fr","type":"article","venue":"Le Français aujourd hui","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Psychology","score_opus":0.009672439392037475,"score_gpt":0.25322046143504906,"score_spread":0.2435480220430116,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402817547","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8591695,0.0030048059,0.032376595,0.002546578,0.0005941235,0.00037736955,0.0042085713,0.0005570298,0.09716538],"genre_scores_gemma":[0.86832255,0.0010895769,0.017661443,0.00043150823,0.0001276322,0.00026021726,0.0035894953,0.0002660058,0.10825158],"study_design_codex":"qualitative","study_design_gemma":"observational","domain_scores_codex":[0.9986437,0.0004334167,0.00005119998,0.00031080918,0.00041440554,0.00014644348],"domain_scores_gemma":[0.9954465,0.002181756,0.00025900506,0.00029501272,0.0014670873,0.00035059295],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014743451,0.0006016361,0.00037250732,0.0025016323,0.0037163552,0.003087004,0.00071340945,0.0007924885,0.009546951],"category_scores_gemma":[0.0055541084,0.00019932391,0.00025150648,0.0029314638,0.002323681,0.0016684813,0.0014676542,0.0010603903,0.0013465463],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00057158363,0.00015652544,0.05890003,0.0016341791,0.00011634218,0.004351598,0.42757967,0.0017869094,0.04825293,0.030942919,0.04598849,0.37971872],"study_design_scores_gemma":[0.000025143994,0.00017138402,0.17690907,0.00057749153,0.0000908357,0.0020087485,0.13445744,0.004645641,0.014693162,0.00231547,0.66396654,0.00013906353],"about_ca_topic_score_codex":0.43120965,"about_ca_topic_score_gemma":0.6388655,"teacher_disagreement_score":0.5687903,"about_ca_system_score_codex":0.0067326827,"about_ca_system_score_gemma":0.0052020336,"threshold_uncertainty_score":0.85739946},"labels":[],"label_agreement":null},{"id":"W4402896984","doi":"10.1109/iwqos61813.2024.10682846","title":"Relic: Federated Conditional Textual Inversion with Prototype Alignment","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Inversion (geology); Natural language processing; Artificial intelligence; Information retrieval; Programming language; Geology","score_opus":0.01049221137940297,"score_gpt":0.2593024431672381,"score_spread":0.24881023178783512,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402896984","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022786267,0.00030525244,0.9632983,0.00026276664,0.00006784989,0.00016259044,0.000283701,0.011142614,0.0016904934],"genre_scores_gemma":[0.52736706,0.00017889465,0.4633975,0.0007253775,0.00009878461,0.00032343258,0.0016837876,0.0007435233,0.005481647],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9984623,0.00039327715,0.00007253907,0.00044609082,0.00047359106,0.00015221882],"domain_scores_gemma":[0.99664646,0.0009780552,0.00027264637,0.001484501,0.00046408264,0.00015423409],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018387721,0.0011778174,0.0011338892,0.0007515427,0.00050742977,0.0011662529,0.0035573295,0.0016616052,0.003719966],"category_scores_gemma":[0.008311336,0.0005146028,0.00090087793,0.0008277711,0.0013068551,0.0035371464,0.0030535606,0.0026702539,0.0022501112],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00097645813,0.0007591603,0.0025070228,0.00026205095,0.00015092341,0.00042028452,0.00031294225,0.24092422,0.032958485,0.0237635,0.017088806,0.67987615],"study_design_scores_gemma":[0.00003733439,0.00009865087,0.00016597567,0.000007691907,0.000011297888,0.00014549511,0.000027260685,0.97625875,0.00985371,0.011756178,0.0016127089,0.000024945177],"about_ca_topic_score_codex":0.0037828404,"about_ca_topic_score_gemma":0.0051394915,"teacher_disagreement_score":0.0037828404,"about_ca_system_score_codex":0.0008769066,"about_ca_system_score_gemma":0.0015626202,"threshold_uncertainty_score":0.012444496},"labels":[],"label_agreement":null},{"id":"W4402920036","doi":"10.1007/978-3-031-71918-9_3","title":"Towards AI-Assisted Protocol Analysis in Design Research: Automating Question Labelling with GPT-4 According to Eris’ (2004) Taxonomy","year":2024,"lang":"en","type":"book-chapter","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Labelling; Taxonomy (biology); Computer science; Protocol (science); Information retrieval; Computational biology; Biology; Biochemistry; Medicine; Ecology","score_opus":0.11983504282014194,"score_gpt":0.384692741145786,"score_spread":0.26485769832564404,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402920036","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0005798189,0.0001157028,0.9935748,0.00061573955,0.000032615902,0.00023990509,0.00015828715,0.0018792675,0.0028040018],"genre_scores_gemma":[0.011788825,0.00015463446,0.98411256,0.00025677355,0.000017483611,0.00034975362,0.00049633277,0.0005163786,0.0023071615],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9712832,0.020341486,0.001996553,0.0025491093,0.0032683022,0.0005613911],"domain_scores_gemma":[0.9413163,0.036900695,0.0020746572,0.013953167,0.0052300575,0.0005251264],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.023957863,0.001989707,0.001220958,0.0046470263,0.002099636,0.010123576,0.0036002062,0.0035308176,0.012783813],"category_scores_gemma":[0.049967937,0.0020641456,0.0033312757,0.004291424,0.0063710585,0.011718462,0.007394004,0.0062539657,0.008377189],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011107706,0.00016730152,0.0015093015,0.001611569,0.00012159981,0.00018721093,0.009871469,0.0050697406,0.009844971,0.3997107,0.019138318,0.5526567],"study_design_scores_gemma":[0.000046683253,0.000063576226,0.00074115256,0.0008645414,0.0001355442,0.00050270994,0.0019080791,0.08260775,0.020942055,0.74980944,0.14228046,0.00009807638],"about_ca_topic_score_codex":0.004725767,"about_ca_topic_score_gemma":0.0064072837,"teacher_disagreement_score":0.023957863,"about_ca_system_score_codex":0.0030506703,"about_ca_system_score_gemma":0.0060600834,"threshold_uncertainty_score":0.12670279},"labels":[],"label_agreement":null},{"id":"W4402927714","doi":"10.23977/acss.2024.080605","title":"Optimizing Multilingual Communication with Computer-Assisted Translation Tools","year":2024,"lang":"en","type":"article","venue":"Advances in Computer Signals and Systems","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Translation (biology); Computer science; Natural language processing; Artificial intelligence; Biology","score_opus":0.03114946480861948,"score_gpt":0.30945659846913176,"score_spread":0.2783071336605123,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402927714","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.027975908,0.0007724856,0.95106983,0.0006603904,0.00017616365,0.00018396552,0.0002275537,0.008601541,0.010332151],"genre_scores_gemma":[0.23590653,0.00084504404,0.7542676,0.0002355943,0.00011302785,0.00024249633,0.0008255422,0.00153762,0.00602664],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9967489,0.0015859964,0.0003182231,0.00044245875,0.00073389424,0.00017060469],"domain_scores_gemma":[0.9930801,0.0037612503,0.00053838093,0.0009679742,0.0015349435,0.000117305244],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002874341,0.001350922,0.00080414995,0.0015066133,0.0012164923,0.004009497,0.0013833328,0.00134837,0.008440725],"category_scores_gemma":[0.015184498,0.00048804434,0.0006677372,0.0022224255,0.0009154634,0.003799069,0.0030724707,0.0014521449,0.006526264],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006953855,0.0002835047,0.0018245748,0.0011229586,0.000119633834,0.0012577955,0.0039825686,0.0749557,0.06794524,0.05082385,0.016294995,0.7806938],"study_design_scores_gemma":[0.00023958793,0.00059521815,0.0012142885,0.00038603423,0.00022688857,0.0014208958,0.0037565974,0.6127553,0.14146627,0.09147483,0.14621592,0.00024818335],"about_ca_topic_score_codex":0.0019111034,"about_ca_topic_score_gemma":0.0025348137,"teacher_disagreement_score":0.008440725,"about_ca_system_score_codex":0.00074670545,"about_ca_system_score_gemma":0.0018991503,"threshold_uncertainty_score":0.028237045},"labels":[],"label_agreement":null},{"id":"W4402981248","doi":"10.1109/nvmsa63038.2024.10693654","title":"Key-Space Partitioned LSM Tree for CMM-H","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"National Research Foundation of Korea","keywords":"Key (lock); Computer science; Tree (set theory); Space (punctuation); Theoretical computer science; Mathematics; Combinatorics; Computer security; Operating system","score_opus":0.015161804343710996,"score_gpt":0.2907845246799314,"score_spread":0.2756227203362204,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402981248","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13719234,0.0034102565,0.80564773,0.0014408852,0.00044228588,0.0005640024,0.0034338026,0.017043296,0.030825322],"genre_scores_gemma":[0.5860047,0.0006588036,0.38896576,0.00070227927,0.00015556515,0.0003808509,0.004658608,0.00048557468,0.017987832],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996295,0.00004218013,0.000029079943,0.00007052171,0.0001501917,0.00007850379],"domain_scores_gemma":[0.9993967,0.00010793415,0.000055837012,0.00023979785,0.0001610521,0.000038674032],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00032891083,0.00031359104,0.00036400455,0.000711472,0.0007301717,0.0009981389,0.0014156038,0.0004618805,0.007953872],"category_scores_gemma":[0.0012509801,0.00020954157,0.00031554757,0.001260894,0.0003789978,0.0016978267,0.0013845124,0.00051625335,0.0023333693],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014760006,0.00026134643,0.0043545035,0.000551624,0.00007719369,0.00046741666,0.00042802503,0.03299536,0.090637736,0.09374521,0.06886344,0.7061421],"study_design_scores_gemma":[0.00030072258,0.00094075076,0.0047113723,0.00013189431,0.0000988905,0.0011409076,0.0006499921,0.5424984,0.14597929,0.0759799,0.2274321,0.00013582915],"about_ca_topic_score_codex":0.003919125,"about_ca_topic_score_gemma":0.006459752,"teacher_disagreement_score":0.007953872,"about_ca_system_score_codex":0.0012695907,"about_ca_system_score_gemma":0.001621532,"threshold_uncertainty_score":0.026608407},"labels":[],"label_agreement":null},{"id":"W4402982628","doi":"10.1109/otcon60325.2024.10687541","title":"Utilizing Deep Learning Algorithms for Real-Time Language Translation and Breaking Down Communication Barriers","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Horizon College and Seminary","funders":"","keywords":"Computer science; Translation (biology); Artificial intelligence; Algorithm","score_opus":0.016219421475211503,"score_gpt":0.30110071020263035,"score_spread":0.28488128872741886,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402982628","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017664978,0.0012279204,0.973766,0.0007410906,0.0003074316,0.000042726406,0.0001011538,0.00220647,0.003942242],"genre_scores_gemma":[0.52503043,0.002028617,0.46067482,0.0010740696,0.00023752371,0.00020745696,0.000923579,0.0007402692,0.009083186],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99851364,0.00053741306,0.0001231904,0.00032822526,0.00036662893,0.00013087838],"domain_scores_gemma":[0.99820864,0.0007591388,0.00019712198,0.00040155588,0.0003772673,0.00005627129],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001972776,0.0011902484,0.0007196907,0.0007436724,0.0007321387,0.0016462394,0.0012434786,0.0014698755,0.003832507],"category_scores_gemma":[0.0074448516,0.00042359534,0.00080838864,0.0012031221,0.0010980666,0.003400645,0.0020646635,0.0026278119,0.00243885],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028214254,0.00017148147,0.0020491285,0.00052333495,0.0001466479,0.00039732998,0.0005623835,0.19815706,0.023346202,0.036439072,0.011450124,0.7264752],"study_design_scores_gemma":[0.000029795612,0.000085821965,0.00034952784,0.00007036407,0.000055526987,0.00014511496,0.00014665103,0.93377596,0.014297252,0.039190702,0.011820299,0.000032976965],"about_ca_topic_score_codex":0.0037538141,"about_ca_topic_score_gemma":0.003559296,"teacher_disagreement_score":0.003832507,"about_ca_system_score_codex":0.0009872604,"about_ca_system_score_gemma":0.0016452539,"threshold_uncertainty_score":0.012821078},"labels":[],"label_agreement":null},{"id":"W4403002917","doi":"10.1007/978-3-031-72952-2_4","title":"TC4D: Trajectory-Conditioned Text-to-4D Generation","year":2024,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University; Vector Institute; University of Toronto","funders":"","keywords":"Computer science; Trajectory; Text generation; Artificial intelligence; Natural language processing; Speech recognition; Physics","score_opus":0.016197048865719384,"score_gpt":0.26595608770256446,"score_spread":0.24975903883684508,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403002917","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0045284913,0.00019622155,0.90510654,0.000193292,0.00080107455,0.00035034673,0.00292074,0.0731063,0.012796986],"genre_scores_gemma":[0.12972733,0.00023721877,0.8143183,0.00038073058,0.00016421199,0.00069282285,0.012172067,0.015536415,0.026770959],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99933594,0.0000824421,0.000037419737,0.00015728881,0.00032294382,0.00006392151],"domain_scores_gemma":[0.99911493,0.00030817368,0.00003439423,0.00026804177,0.00022349301,0.000050931285],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00049880013,0.0020485774,0.0009880451,0.0011405422,0.0006770261,0.001384556,0.0032191055,0.0020344462,0.084455624],"category_scores_gemma":[0.0029603252,0.0008602041,0.0013286017,0.0009741556,0.0008021971,0.0012454529,0.003190317,0.0015318737,0.02385565],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006893439,0.00014038435,0.00055547396,0.00064971595,0.00010070836,0.0008185467,0.0003098417,0.062192887,0.055585064,0.03207685,0.17906556,0.6678157],"study_design_scores_gemma":[0.00016906655,0.00015089699,0.00035816594,0.00006112788,0.000038395392,0.00063542783,0.00010708008,0.81198686,0.07018746,0.034840807,0.08137971,0.00008495892],"about_ca_topic_score_codex":0.0031106863,"about_ca_topic_score_gemma":0.004476507,"teacher_disagreement_score":0.084455624,"about_ca_system_score_codex":0.0006844811,"about_ca_system_score_gemma":0.0008275562,"threshold_uncertainty_score":0.28253222},"labels":[],"label_agreement":null},{"id":"W4403036283","doi":"10.1007/s10664-024-10548-3","title":"An empirical study on the effectiveness of large language models for SATD identification and classification","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba; Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Identification (biology); Computer science; Empirical research; Natural language processing; Artificial intelligence; Data science; Statistics; Mathematics; Biology","score_opus":0.027547982238886545,"score_gpt":0.3451586090035,"score_spread":0.31761062676461344,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403036283","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.94217074,0.0013393159,0.04735081,0.001391003,0.000091315516,0.00020172917,0.0018096904,0.0012010378,0.004444287],"genre_scores_gemma":[0.9724717,0.00023063301,0.023082172,0.00019752934,0.00004232367,0.000085571046,0.0028338467,0.00012588617,0.00093048875],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9847825,0.011712421,0.0007426565,0.0014367873,0.001056684,0.0002690011],"domain_scores_gemma":[0.6920501,0.2898922,0.0029988163,0.010154541,0.0038451988,0.0010591106],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.015189369,0.0011013073,0.00081594894,0.0022314962,0.00113818,0.0024639713,0.0020086744,0.0018480892,0.0033711274],"category_scores_gemma":[0.11443673,0.0004722135,0.0009694789,0.002179855,0.001260876,0.0063402974,0.00175773,0.0026596729,0.0016387127],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0060349857,0.0069002258,0.2963994,0.0011027524,0.0009099252,0.0005962156,0.0037512963,0.09666128,0.0071284347,0.011383045,0.017515473,0.5516169],"study_design_scores_gemma":[0.00034343515,0.0011339091,0.03934,0.00014393391,0.0003854061,0.0007106051,0.002456486,0.9291992,0.007151946,0.014302467,0.0047350796,0.00009753853],"about_ca_topic_score_codex":0.0062788273,"about_ca_topic_score_gemma":0.006135588,"teacher_disagreement_score":0.015189369,"about_ca_system_score_codex":0.0014939806,"about_ca_system_score_gemma":0.0013677562,"threshold_uncertainty_score":0.080330014},"labels":[],"label_agreement":null},{"id":"W4403119743","doi":"10.1016/b978-0-323-95504-1.00141-1","title":"Technology for Translators","year":2024,"lang":"en","type":"book-chapter","venue":"Elsevier eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Computer science","score_opus":0.011693746127601953,"score_gpt":0.2598763800091486,"score_spread":0.24818263388154663,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403119743","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0022774746,0.019883817,0.18939862,0.005433439,0.0031051827,0.0001363509,0.0013642524,0.00709838,0.7713024],"genre_scores_gemma":[0.015911777,0.014654798,0.081771895,0.001295142,0.00090110546,0.00019535937,0.0022792032,0.0029933916,0.8799973],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99907035,0.00024799656,0.00007212797,0.00024431408,0.0002966186,0.00006854693],"domain_scores_gemma":[0.99857664,0.00065098726,0.000041284995,0.00039228285,0.00027781946,0.000060989325],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010939096,0.0010197797,0.0007617155,0.0021354398,0.0014763984,0.004947333,0.0011894375,0.0019092935,0.1802708],"category_scores_gemma":[0.0033992203,0.0010429656,0.0006258762,0.00245489,0.0014189342,0.009609062,0.0028931347,0.0024404223,0.11168309],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000036913047,0.0000345365,0.00013326903,0.0006275036,0.000011533413,0.00027475198,0.0011056986,0.00024009291,0.005151718,0.17281967,0.26866433,0.55090004],"study_design_scores_gemma":[0.0000052986056,0.000014416682,0.00011086186,0.00021589984,0.000007965171,0.00036625992,0.00018972623,0.00048745458,0.0016165818,0.038945068,0.9580316,0.000008873291],"about_ca_topic_score_codex":0.00092240336,"about_ca_topic_score_gemma":0.0009903819,"teacher_disagreement_score":0.1802708,"about_ca_system_score_codex":0.00071373925,"about_ca_system_score_gemma":0.0011327324,"threshold_uncertainty_score":0.60306597},"labels":[],"label_agreement":null},{"id":"W4403289793","doi":"10.18357/kula.291","title":"Large Language Publishing","year":2024,"lang":"en","type":"article","venue":"KULA knowledge creation dissemination and preservation studies","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Publishing; Computer science; Linguistics; Art; Philosophy; Literature","score_opus":0.029346763359652542,"score_gpt":0.38171319064367726,"score_spread":0.35236642728402473,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403289793","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0011455741,0.011639723,0.0033062554,0.04640255,0.035231847,0.00008830022,0.0023013016,0.0023641218,0.8975203],"genre_scores_gemma":[0.024643144,0.01184767,0.0027971035,0.012973132,0.02100691,0.00013943714,0.0028856727,0.0022716362,0.9214353],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.993389,0.0010823732,0.0006645491,0.0009961162,0.0034495878,0.00041835476],"domain_scores_gemma":[0.9694482,0.010052887,0.0021292698,0.008586483,0.0065143537,0.0032687357],"candidate_categories":["scholarly_communication","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0046691564,0.00073709764,0.0008831094,0.005437341,0.0046269326,0.031264953,0.0019074298,0.0031137634,0.30275968],"category_scores_gemma":[0.038171552,0.00043685257,0.00061824924,0.008174588,0.0039510457,0.01456283,0.0053937766,0.0040094513,0.19654113],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000024411358,0.000015692092,0.000195547,0.0003482688,0.0000119837,0.00012672898,0.00049091276,0.000065720924,0.00028503904,0.16424625,0.75059104,0.08359847],"study_design_scores_gemma":[0.0000018964089,0.0000037347227,0.00010762506,0.0000819115,0.0000014749162,0.000053404478,0.00010487438,0.000022873826,0.00007790206,0.006383243,0.99315625,0.0000048689762],"about_ca_topic_score_codex":0.001506957,"about_ca_topic_score_gemma":0.0025239128,"teacher_disagreement_score":0.96873504,"about_ca_system_score_codex":0.004033393,"about_ca_system_score_gemma":0.004902513,"threshold_uncertainty_score":0.9945287},"labels":[],"label_agreement":null},{"id":"W4403305855","doi":"10.1017/9781009302180.019","title":"Parsing with Context-Free Grammars","year":2024,"lang":"en","type":"book-chapter","venue":"Cambridge University Press eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Parsing; Context-free grammar; Rule-based machine translation; Computer science; Tree-adjoining grammar; L-attributed grammar; Extended Affix Grammar; Parsing expression grammar; Context (archaeology); Natural language processing; Artificial intelligence; History","score_opus":0.014856873551741832,"score_gpt":0.19804661672524068,"score_spread":0.18318974317349884,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403305855","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0013579556,0.0077708676,0.80030817,0.0030307153,0.0011132235,0.00016675016,0.0010186775,0.007121829,0.17811188],"genre_scores_gemma":[0.022294413,0.01158843,0.6613249,0.002077126,0.00055754837,0.00044847722,0.004234215,0.004481775,0.29299322],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9994215,0.00011342069,0.000045070137,0.00013151484,0.00024990694,0.00003865372],"domain_scores_gemma":[0.9995757,0.00022734996,0.000015457379,0.000089109075,0.000076230295,0.000016070937],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000658806,0.0013072684,0.0006594484,0.0010804619,0.0007529484,0.0035623382,0.0014557295,0.0013472987,0.04127211],"category_scores_gemma":[0.002625868,0.00085324206,0.0012261758,0.0017894802,0.0013070574,0.0056512845,0.0015357475,0.0033960312,0.03073346],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000019707868,0.000036806046,0.00013125598,0.00044514667,0.000026110374,0.00013167474,0.0005249323,0.0047116256,0.0021614053,0.38915697,0.19132936,0.41132513],"study_design_scores_gemma":[0.000008796751,0.000014275132,0.00019895217,0.00028851966,0.000010730324,0.0002871607,0.00008572279,0.006518474,0.0019116856,0.3167624,0.6738896,0.000023672987],"about_ca_topic_score_codex":0.0013351961,"about_ca_topic_score_gemma":0.0016707886,"teacher_disagreement_score":0.04127211,"about_ca_system_score_codex":0.0009777634,"about_ca_system_score_gemma":0.001238204,"threshold_uncertainty_score":0.13806891},"labels":[],"label_agreement":null},{"id":"W4403336593","doi":"10.2139/ssrn.4949879","title":"Does the LLMperor Have New Clothes? Some Thoughts on the Use of LLMs in eDiscovery","year":2024,"lang":"en","type":"article","venue":"SSRN Electronic Journal","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University; University of Waterloo","funders":"","keywords":"Clothing; Aesthetics; History; Art; Archaeology","score_opus":0.0186128884877075,"score_gpt":0.26440901256555305,"score_spread":0.24579612407784554,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403336593","genre_codex":"commentary","genre_gemma":"commentary","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"commentary","genre_consensus":"commentary","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006962632,0.049291145,0.18870603,0.6119257,0.00717876,0.000091766866,0.00023969899,0.0013122521,0.13429192],"genre_scores_gemma":[0.41731027,0.04611022,0.2827133,0.09052411,0.0214741,0.00059375266,0.00035654745,0.0022155421,0.13870211],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9925882,0.0046344893,0.00039997578,0.00075123133,0.0011265868,0.0004995613],"domain_scores_gemma":[0.978784,0.014229282,0.0005627027,0.0033330421,0.0021849307,0.00090591324],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.016425889,0.0008631065,0.001303803,0.0027975899,0.0039999075,0.0116509795,0.0049616704,0.0072185006,0.01948459],"category_scores_gemma":[0.030567592,0.0008579252,0.0012222689,0.0026228176,0.025528207,0.05098824,0.0056450684,0.010531237,0.004231441],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006756986,0.000032345633,0.0002971481,0.00012434127,0.000013038044,0.00014897563,0.0014143286,0.00046307847,0.00026452777,0.93656826,0.027782524,0.032823842],"study_design_scores_gemma":[0.000032892807,0.000049561666,0.00023523699,0.0003360249,0.000011955256,0.00033789966,0.0019549616,0.0034536077,0.000757936,0.7846363,0.20809214,0.00010132916],"about_ca_topic_score_codex":0.0060824556,"about_ca_topic_score_gemma":0.0063156304,"teacher_disagreement_score":0.01948459,"about_ca_system_score_codex":0.0033294556,"about_ca_system_score_gemma":0.0016132586,"threshold_uncertainty_score":0.08686948},"labels":[],"label_agreement":null},{"id":"W4403486620","doi":"10.3233/faia240955","title":"SETTP: Style Extraction and Tunable Inference via Dual-Level Transferable Prompt Learning","year":2024,"lang":"en","type":"book-chapter","venue":"Frontiers in artificial intelligence and applications","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Inference; Style (visual arts); Extraction (chemistry); Dual (grammatical number); Computer science; Artificial intelligence; Geography; Art; Chemistry; Chromatography; Archaeology","score_opus":0.03530895307235937,"score_gpt":0.29701150762031864,"score_spread":0.26170255454795927,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403486620","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029724931,0.0006296009,0.9297764,0.00026482667,0.00044800088,0.0002699255,0.001454115,0.03297415,0.0044580936],"genre_scores_gemma":[0.38827014,0.00050511584,0.5859992,0.0007792251,0.0003083559,0.000659968,0.0071298336,0.0013992271,0.014948934],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99865025,0.00033896687,0.000071309616,0.0006321243,0.0002254316,0.00008184557],"domain_scores_gemma":[0.9977884,0.0010107574,0.0001095856,0.00057742844,0.00037154905,0.00014226661],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014891429,0.0019825113,0.0008163891,0.00089330005,0.000438399,0.0014228068,0.0023161662,0.0013984012,0.008858125],"category_scores_gemma":[0.0078090797,0.0005555366,0.0012665608,0.0010701444,0.0005969808,0.0037324703,0.0022434175,0.0028143495,0.004733075],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00057496486,0.00034735902,0.0028425122,0.00035803334,0.00009295876,0.00028847347,0.0003554401,0.020980986,0.029132191,0.00449388,0.021804433,0.9187288],"study_design_scores_gemma":[0.00021454254,0.0004635953,0.0024889186,0.000052389016,0.00008373625,0.00033989793,0.0002281525,0.91016304,0.03079037,0.039007623,0.016093215,0.000074517775],"about_ca_topic_score_codex":0.001224501,"about_ca_topic_score_gemma":0.0021184878,"teacher_disagreement_score":0.008858125,"about_ca_system_score_codex":0.0006175802,"about_ca_system_score_gemma":0.0010663043,"threshold_uncertainty_score":0.029633343},"labels":[],"label_agreement":null},{"id":"W4403486628","doi":"10.3233/faia240968","title":"CoTran: An LLM-Based Code Translator Using Reinforcement Learning with Feedback from Compiler and Symbolic Execution","year":2024,"lang":"en","type":"book-chapter","venue":"Frontiers in artificial intelligence and applications","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Compiler; Computer science; Programming language; Reinforcement learning; Code (set theory); Parallel computing; Artificial intelligence","score_opus":0.034853866022306855,"score_gpt":0.28210929988269373,"score_spread":0.24725543386038687,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403486628","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.047149383,0.00043577905,0.8575024,0.0003128848,0.00021421564,0.0002686641,0.0002511979,0.089800656,0.0040648044],"genre_scores_gemma":[0.36755705,0.0001578959,0.62170094,0.00046217165,0.000046480713,0.00047248605,0.0008289992,0.003949563,0.004824379],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99883896,0.00031951623,0.000070611844,0.0003435998,0.0003348511,0.00009252879],"domain_scores_gemma":[0.9974636,0.0012500379,0.000252478,0.00047096508,0.0004336628,0.00012934975],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013466451,0.0017039429,0.0006596448,0.0006400708,0.00031417335,0.0007080717,0.0024466235,0.0010731054,0.005684972],"category_scores_gemma":[0.007009638,0.0005896793,0.00069050526,0.00031034616,0.0008890303,0.0011942829,0.0013586786,0.0017712573,0.0023485224],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00083808135,0.00068406056,0.0056886384,0.0008073112,0.0001630304,0.0005939646,0.00039613163,0.30083838,0.058852475,0.0050851842,0.014423757,0.61162895],"study_design_scores_gemma":[0.00010721813,0.0002642643,0.00053194514,0.00003782912,0.000027750537,0.000111750654,0.000030107645,0.9707901,0.020220399,0.0024473434,0.0053983843,0.00003285363],"about_ca_topic_score_codex":0.0025007795,"about_ca_topic_score_gemma":0.0035241067,"teacher_disagreement_score":0.005684972,"about_ca_system_score_codex":0.00061988237,"about_ca_system_score_gemma":0.0014206661,"threshold_uncertainty_score":0.019018114},"labels":[],"label_agreement":null},{"id":"W4403648112","doi":"10.1145/3691621.3694939","title":"A Preliminary Study of Multilingual Code Language Models for Code Generation Task Using Translated Benchmarks","year":2024,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia, Okanagan Campus; Kelowna General Hospital; University of British Columbia","funders":"","keywords":"Computer science; Code (set theory); Task (project management); Programming language; Code generation; Natural language processing; Operating system; Engineering","score_opus":0.05547638290687799,"score_gpt":0.35269475624778535,"score_spread":0.29721837334090734,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403648112","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.84975356,0.0014847725,0.11171361,0.0011249537,0.00060951913,0.0015216718,0.0067077773,0.015439067,0.011645053],"genre_scores_gemma":[0.80733573,0.00033373066,0.16033626,0.0006140468,0.000113688504,0.0016241228,0.02353656,0.0024138002,0.0036921108],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98990554,0.0059786253,0.0008092917,0.0017569767,0.001159343,0.00039031266],"domain_scores_gemma":[0.9598457,0.022268722,0.0013050617,0.0062442026,0.0093514025,0.0009848878],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008542735,0.0021178175,0.00096780993,0.001670981,0.001068437,0.0022281173,0.0023984462,0.0012871792,0.004852888],"category_scores_gemma":[0.059388004,0.00070287276,0.0009002361,0.001800435,0.0010511802,0.0038281712,0.0025567317,0.0029105546,0.0026507112],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0054331594,0.006142241,0.050783288,0.004888415,0.0007268645,0.0017845184,0.008593065,0.14419994,0.055309556,0.009824319,0.054995924,0.6573188],"study_design_scores_gemma":[0.0014452615,0.005813355,0.030723365,0.0006138467,0.0003741908,0.0010435985,0.0047833566,0.8026832,0.09706699,0.010633955,0.04443023,0.0003887977],"about_ca_topic_score_codex":0.010785229,"about_ca_topic_score_gemma":0.014111715,"teacher_disagreement_score":0.010785229,"about_ca_system_score_codex":0.0014267531,"about_ca_system_score_gemma":0.002266196,"threshold_uncertainty_score":0.04517883},"labels":[],"label_agreement":null},{"id":"W4403740969","doi":"10.1016/j.eswa.2024.125589","title":"FRGEM: Feature integration pre-training based Gaussian embedding model for Chinese word representation","year":2024,"lang":"en","type":"article","venue":"Expert Systems with Applications","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"Sichuan Province Science and Technology Support Program; National University's Basic Research Foundation of China; China Postdoctoral Science Foundation; National Natural Science Foundation of China","keywords":"Computer science; Word embedding; Feature (linguistics); Word (group theory); Artificial intelligence; Representation (politics); Embedding; Training (meteorology); Natural language processing; Pattern recognition (psychology); Gaussian; Training set; Machine learning; Mathematics; Linguistics","score_opus":0.023426597846608593,"score_gpt":0.34731843608318186,"score_spread":0.32389183823657325,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403740969","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02792366,0.00068734755,0.9593274,0.00016828786,0.00016567278,0.0001019097,0.0009356288,0.009295538,0.0013945848],"genre_scores_gemma":[0.3781573,0.0009791713,0.5913173,0.00036997983,0.00014099131,0.000486101,0.009229979,0.0011174683,0.018201714],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996643,0.000059791808,0.00002007674,0.00012017851,0.00007484569,0.00006088504],"domain_scores_gemma":[0.999718,0.000074562864,0.000013531141,0.000056594592,0.000116132316,0.000021136844],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00056750875,0.0011012475,0.00089508883,0.0006723179,0.00042902748,0.0005144797,0.0013467696,0.0007539209,0.004394558],"category_scores_gemma":[0.001135721,0.00037231972,0.0008547193,0.0010396859,0.00029535842,0.0014178178,0.0011242105,0.0016251612,0.0024532473],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003469872,0.00017578236,0.001375814,0.0001926466,0.000118328884,0.00018060148,0.00014460608,0.06817334,0.022549687,0.00692983,0.021355828,0.87845653],"study_design_scores_gemma":[0.000031430107,0.00010005266,0.00076491025,0.000011974341,0.00003624282,0.000080036174,0.000027873428,0.9821035,0.009434029,0.003076725,0.0043109707,0.000022198488],"about_ca_topic_score_codex":0.021827802,"about_ca_topic_score_gemma":0.022864446,"teacher_disagreement_score":0.021827802,"about_ca_system_score_codex":0.0004804424,"about_ca_system_score_gemma":0.0013952425,"threshold_uncertainty_score":0.04340154},"labels":[],"label_agreement":null},{"id":"W4403779633","doi":"10.1111/lnc3.70001","title":"The Roles of Neural Networks in Language Acquisition","year":2024,"lang":"en","type":"article","venue":"Language and Linguistics Compass","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Mila - Quebec Artificial Intelligence Institute; HEC Montréal","funders":"","keywords":"Linguistics; Computer science; Artificial neural network; Cognitive science; Natural language processing; Artificial intelligence; Psychology; Communication; Philosophy","score_opus":0.006272683252111985,"score_gpt":0.268238702716715,"score_spread":0.261966019464603,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403779633","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.43404102,0.0039407434,0.45079237,0.024798732,0.00012887911,0.000072777744,0.00029340188,0.00046214648,0.085469894],"genre_scores_gemma":[0.96892357,0.0007681591,0.02715802,0.00016772066,0.000029822446,0.000043064574,0.00005904524,0.000030943615,0.0028197493],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9985196,0.0009800307,0.000042824726,0.00017908213,0.00020177496,0.000076673416],"domain_scores_gemma":[0.991682,0.006315403,0.0006101571,0.00047707843,0.0006526152,0.000262822],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002972998,0.0003587031,0.0003233739,0.00096302613,0.00047474002,0.0035527912,0.0011368596,0.0011151624,0.0034007747],"category_scores_gemma":[0.014532369,0.0004252436,0.00039911742,0.00071401853,0.003496437,0.008671738,0.0017358771,0.0017616451,0.0004552026],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010015259,0.00007477898,0.0099631315,0.00016306913,0.00006234268,0.00017707334,0.0017003636,0.06139385,0.0034539746,0.86990535,0.0011357254,0.051870134],"study_design_scores_gemma":[0.000008180156,0.000034762863,0.002282368,0.0000534916,0.000011801138,0.00007594182,0.00032253133,0.21440364,0.0011263872,0.7791286,0.0025285857,0.000023743296],"about_ca_topic_score_codex":0.00297754,"about_ca_topic_score_gemma":0.0026529448,"teacher_disagreement_score":0.0035527912,"about_ca_system_score_codex":0.0015467339,"about_ca_system_score_gemma":0.00092310895,"threshold_uncertainty_score":0.01572293},"labels":[],"label_agreement":null},{"id":"W4403804633","doi":"10.1163/9789004702660_014","title":"Structuring the Lexicon of Old English with Syntactic Principles: The Role of Deverbal Nominalisations with Aspectual and Control Verbs","year":2024,"lang":"en","type":"book-chapter","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Lexicon; Linguistics; Structuring; Computer science; Natural language processing; Control (management); Artificial intelligence; Philosophy; Political science","score_opus":0.00761726786217476,"score_gpt":0.20167329717771054,"score_spread":0.19405602931553578,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403804633","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10499315,0.025121242,0.14629364,0.003619298,0.00047491657,0.00010245629,0.00046190122,0.000664754,0.7182687],"genre_scores_gemma":[0.7204184,0.012846531,0.081717946,0.00084053824,0.00047316187,0.0001398477,0.0014630728,0.0006933459,0.18140712],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9998673,0.000046029316,0.000010497795,0.000024344597,0.00003817377,0.000013642791],"domain_scores_gemma":[0.99978906,0.00011926021,0.00001653221,0.000027131953,0.000035648944,0.000012388693],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00027744338,0.00032515917,0.00027888158,0.001129131,0.0006176626,0.0032634528,0.0004997677,0.00026230558,0.003952808],"category_scores_gemma":[0.0005831139,0.00020809232,0.00023905581,0.0010566557,0.0027838834,0.004215886,0.000821182,0.0010063042,0.0013492621],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000018137685,0.000017862823,0.0009259903,0.00022475462,0.000006717572,0.00030581802,0.008891638,0.0005253477,0.002917122,0.8773897,0.009794519,0.098982416],"study_design_scores_gemma":[0.000008810683,0.000030678282,0.004528814,0.00024144782,0.000021212687,0.00091001875,0.003996085,0.0031112165,0.0031166181,0.39079633,0.5932102,0.000028495504],"about_ca_topic_score_codex":0.0013501226,"about_ca_topic_score_gemma":0.0040627047,"teacher_disagreement_score":0.003952808,"about_ca_system_score_codex":0.0013828501,"about_ca_system_score_gemma":0.0005857638,"threshold_uncertainty_score":0.01322341},"labels":[],"label_agreement":null},{"id":"W4403887388","doi":"10.1007/s10664-024-10549-2","title":"Towards effectively testing machine translation systems from white-box perspectives","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University; University of Waterloo","funders":"","keywords":"Computer science; Translation (biology); Machine translation; White box; Systems engineering; Artificial intelligence; Engineering; Software engineering","score_opus":0.02217214222265518,"score_gpt":0.27888827832438084,"score_spread":0.25671613610172567,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403887388","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09494362,0.00028431445,0.8935737,0.001760985,0.000084495725,0.00018547174,0.00021766385,0.0052251434,0.0037246454],"genre_scores_gemma":[0.5383083,0.00015737486,0.457419,0.0006555539,0.00009743655,0.00019550534,0.00054820866,0.0010893457,0.0015293489],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.96979386,0.021093413,0.0013506467,0.002358413,0.00448338,0.0009203235],"domain_scores_gemma":[0.83437306,0.1376906,0.003751718,0.014566704,0.008431872,0.0011860877],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.017273396,0.0017123667,0.0012218999,0.0023808877,0.0010312756,0.005320158,0.0030245604,0.0037891674,0.005006432],"category_scores_gemma":[0.1130148,0.0012648002,0.0009909312,0.0014019621,0.0037120376,0.013016649,0.003983439,0.0034023423,0.0012394241],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013670146,0.002335174,0.022116413,0.0018097475,0.00048739035,0.0021664202,0.004273767,0.12451614,0.098821335,0.2421831,0.010367657,0.48955595],"study_design_scores_gemma":[0.00017233477,0.00041575776,0.001874954,0.00020380654,0.000119483404,0.00044722244,0.0009892301,0.68732035,0.055132903,0.24790622,0.0053597414,0.000058061098],"about_ca_topic_score_codex":0.001810842,"about_ca_topic_score_gemma":0.0028131332,"teacher_disagreement_score":0.017273396,"about_ca_system_score_codex":0.001103138,"about_ca_system_score_gemma":0.0033094187,"threshold_uncertainty_score":0.09135157},"labels":[],"label_agreement":null},{"id":"W4403920266","doi":"10.48550/arxiv.2410.03441","title":"CLoSD: Closing the Loop between Simulation and Diffusion for multi-task character control","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Tel Aviv University","keywords":"Closing (real estate); Character (mathematics); Loop (graph theory); Task (project management); Control (management); Diffusion; Computer science; Control theory (sociology); Mathematics; Physics; Engineering; Artificial intelligence; Business; Geometry; Finance; Systems engineering","score_opus":0.07930640759990543,"score_gpt":0.24915546575469363,"score_spread":0.1698490581547882,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403920266","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0044414336,0.00007642473,0.9905059,0.00013876727,0.00005063072,0.000037042402,0.00004938556,0.0033227676,0.0013776757],"genre_scores_gemma":[0.40455586,0.00018391262,0.58677393,0.00027969183,0.00005913928,0.0004196532,0.00037009653,0.0013819564,0.0059757596],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994373,0.00014412563,0.000034933266,0.00015267628,0.00018499923,0.00004592614],"domain_scores_gemma":[0.998978,0.00057299074,0.00007039393,0.0001677355,0.00011924349,0.00009172855],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009899886,0.00071199704,0.00060556177,0.00041588768,0.00047535764,0.000966251,0.0019059968,0.0010085832,0.0052390182],"category_scores_gemma":[0.0032834264,0.0004953067,0.00059499126,0.00022810505,0.0014489016,0.0016116089,0.0025533002,0.002175465,0.000898464],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00048780718,0.0001777167,0.000841229,0.0003337945,0.000082943545,0.00023639145,0.00053548114,0.62886626,0.03087861,0.09698386,0.009671478,0.23090433],"study_design_scores_gemma":[0.000028388975,0.000020870386,0.000040646486,0.000006729578,0.000003642094,0.000015423106,0.000011365238,0.97968936,0.0026821347,0.014108985,0.0033855718,0.0000069249],"about_ca_topic_score_codex":0.004562955,"about_ca_topic_score_gemma":0.005142901,"teacher_disagreement_score":0.0052390182,"about_ca_system_score_codex":0.0009857715,"about_ca_system_score_gemma":0.0010271961,"threshold_uncertainty_score":0.01752621},"labels":[],"label_agreement":null},{"id":"W4403988122","doi":"10.15837/ijccc.2024.6.6853","title":"Evaluating and Mitigating Gender Bias in Generative Large Language Models","year":2024,"lang":"en","type":"article","venue":"International Journal of Computers Communications & Control","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Generative grammar; Natural language processing; Artificial intelligence","score_opus":0.08070041347036223,"score_gpt":0.3999135031773028,"score_spread":0.31921308970694057,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403988122","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.27123213,0.0018349538,0.7121328,0.0013350801,0.00042112605,0.00041904766,0.00088934577,0.0057618557,0.0059737214],"genre_scores_gemma":[0.7970182,0.00043083346,0.19560014,0.0006818778,0.00009420248,0.00026046083,0.001815822,0.0008249176,0.003273647],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.995069,0.003150886,0.00021106731,0.00052950595,0.0008530599,0.00018649128],"domain_scores_gemma":[0.97455,0.021331007,0.00067204575,0.0017593179,0.0014445942,0.00024287225],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009040958,0.00087706157,0.00059043994,0.00089571887,0.0005861011,0.0017316261,0.0011829068,0.0011660024,0.0025981118],"category_scores_gemma":[0.041317586,0.000385561,0.0007103149,0.00048205917,0.0008402494,0.0018914586,0.002359738,0.0012149104,0.0013332213],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013774399,0.00031893913,0.023211606,0.0007689582,0.0003062636,0.0005625158,0.00262668,0.2654544,0.027328927,0.017464014,0.009574687,0.6510055],"study_design_scores_gemma":[0.00007249916,0.00036762573,0.0023920713,0.0001092725,0.00008962831,0.00030003363,0.00064783107,0.9381152,0.027389102,0.024071682,0.006390215,0.000054849123],"about_ca_topic_score_codex":0.0027164838,"about_ca_topic_score_gemma":0.005610457,"teacher_disagreement_score":0.009040958,"about_ca_system_score_codex":0.0008855437,"about_ca_system_score_gemma":0.0012512495,"threshold_uncertainty_score":0.047813714},"labels":[],"label_agreement":null},{"id":"W4404060074","doi":"10.1145/3702979","title":"ZS4C: Zero-Shot Synthesis of Compilable Code for Incomplete Code Snippets Using LLMs","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University; Queen's University; University of Windsor; University of Manitoba","funders":"","keywords":"Computer science; Code (set theory); Zero (linguistics); Programming language; Linguistics","score_opus":0.1333233069938429,"score_gpt":0.3604581061756769,"score_spread":0.22713479918183402,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404060074","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.050171938,0.0015525544,0.48680526,0.0006575026,0.00068291195,0.0005871658,0.014560451,0.4387998,0.006182438],"genre_scores_gemma":[0.17394607,0.0006859449,0.7104835,0.000978777,0.00018924227,0.0010858624,0.057887238,0.04563713,0.009106202],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9970407,0.00048426926,0.00022387633,0.0008967197,0.0011435624,0.00021086905],"domain_scores_gemma":[0.9936745,0.0031947824,0.00046267008,0.0013733234,0.0011370634,0.00015768348],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017801846,0.0038404106,0.0011744575,0.0034440102,0.00089903304,0.0019720767,0.0029975907,0.001692748,0.009031501],"category_scores_gemma":[0.012709071,0.0011835141,0.0024689506,0.0016421503,0.0015063358,0.0024345892,0.0028600988,0.0019847758,0.0072137685],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018220335,0.00040820395,0.013673028,0.0034681996,0.0005027654,0.0017662862,0.0013692424,0.042521622,0.084329665,0.0078363605,0.17671032,0.66559225],"study_design_scores_gemma":[0.00038183815,0.0006379863,0.0064501534,0.0003716647,0.0002652653,0.0010971802,0.0007690455,0.6819559,0.17051227,0.026429106,0.11086968,0.0002599916],"about_ca_topic_score_codex":0.004637171,"about_ca_topic_score_gemma":0.012033711,"teacher_disagreement_score":0.009031501,"about_ca_system_score_codex":0.0009578751,"about_ca_system_score_gemma":0.0026531997,"threshold_uncertainty_score":0.030213416},"labels":[],"label_agreement":null},{"id":"W4404081729","doi":"10.2139/ssrn.5010083","title":"Infusing Clinical Knowledge into Language Models by Subword Optimisation and Embedding Initialisation","year":2024,"lang":"en","type":"preprint","venue":"SSRN Electronic Journal","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Institute of Population and Public Health","funders":"","keywords":"Embedding; Computer science; Natural language processing; Artificial intelligence","score_opus":0.022804271457585665,"score_gpt":0.35964512334235893,"score_spread":0.33684085188477325,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404081729","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025736509,0.00030867933,0.96293,0.0012998505,0.00020495441,0.00015508349,0.0007840257,0.0049122726,0.0036686829],"genre_scores_gemma":[0.5875461,0.00035865395,0.3979658,0.00044090458,0.00025747332,0.00033913789,0.0027749105,0.0018772564,0.00843971],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9974868,0.0012846168,0.00019568842,0.00056572724,0.0002674662,0.00019962835],"domain_scores_gemma":[0.98395985,0.013037306,0.00039420853,0.0012035783,0.0011843995,0.00022071],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036130424,0.0015638154,0.0011044182,0.0014418598,0.00058268785,0.002568304,0.0018240598,0.0024238382,0.01045262],"category_scores_gemma":[0.020491684,0.0012578873,0.0022006189,0.0010524327,0.0012022422,0.004606794,0.0036117139,0.0045889076,0.0047179237],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015578945,0.00037096906,0.0025551503,0.0007526111,0.00027651852,0.0010104066,0.0015480567,0.45901394,0.011222682,0.0504472,0.012942001,0.45830256],"study_design_scores_gemma":[0.000043012464,0.00008576398,0.00014633703,0.00005765246,0.00007551773,0.000100320096,0.00012034118,0.9041016,0.0050936746,0.086879954,0.003265979,0.000029814953],"about_ca_topic_score_codex":0.0038098437,"about_ca_topic_score_gemma":0.005177816,"teacher_disagreement_score":0.01045262,"about_ca_system_score_codex":0.0012018763,"about_ca_system_score_gemma":0.0018925979,"threshold_uncertainty_score":0.03496754},"labels":[],"label_agreement":null},{"id":"W4404305672","doi":"10.48550/arxiv.2407.18213","title":"Scaling Trends in Language Model Robustness","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Robustness (evolution); Computer science; Scale (ratio); Language model; Natural language processing; Geography; Cartography; Chemistry","score_opus":0.05449121822374094,"score_gpt":0.22403434284388354,"score_spread":0.1695431246201426,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404305672","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4990205,0.013258934,0.4079153,0.028956661,0.001202954,0.0002346179,0.002192435,0.008237969,0.038980685],"genre_scores_gemma":[0.9587035,0.0019706963,0.033890743,0.0011322927,0.0003975919,0.0000819533,0.00086731766,0.00053377764,0.0024220822],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9961671,0.0011535187,0.00018056255,0.0010663575,0.0010477076,0.00038476687],"domain_scores_gemma":[0.97152203,0.016799701,0.0014955677,0.008031308,0.0015581684,0.0005932536],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0062706214,0.0011319461,0.00080132886,0.0014699257,0.0006787362,0.0023739054,0.0014031209,0.0017809412,0.0036132345],"category_scores_gemma":[0.051412202,0.0006999159,0.0012059141,0.0008866624,0.0028074153,0.0098910285,0.0027070376,0.0065502278,0.0012348378],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007546115,0.00044725178,0.025067125,0.00076813035,0.00049229374,0.00061272737,0.0016632094,0.42860678,0.02897525,0.23015693,0.03207216,0.2503836],"study_design_scores_gemma":[0.000048233585,0.00042751528,0.0060338215,0.00017052641,0.00009574317,0.00053661456,0.00047339342,0.7829982,0.019011818,0.17001814,0.020082913,0.00010310431],"about_ca_topic_score_codex":0.002107731,"about_ca_topic_score_gemma":0.0015402231,"teacher_disagreement_score":0.0062706214,"about_ca_system_score_codex":0.0013743672,"about_ca_system_score_gemma":0.0007549278,"threshold_uncertainty_score":0.033162594},"labels":[],"label_agreement":null},{"id":"W4404315596","doi":"10.1115/detc2024-139024","title":"DesignQA: Benchmarking Multimodal Large Language Models on Questions Grounded in Engineering Documentation","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Autodesk (Canada)","funders":"","keywords":"Documentation; Benchmarking; Computer science; Software engineering; Natural language processing; Artificial intelligence; Programming language","score_opus":0.010496371983679852,"score_gpt":0.28531260999824004,"score_spread":0.2748162380145602,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404315596","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.49241623,0.003600102,0.3355539,0.0026458313,0.00091881637,0.0027860678,0.03569705,0.08319153,0.04319035],"genre_scores_gemma":[0.6730596,0.000554099,0.22852187,0.0010572622,0.00011557999,0.0021939483,0.08337111,0.002911471,0.008215001],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98540527,0.0096000675,0.001058984,0.0016190244,0.001992593,0.00032405017],"domain_scores_gemma":[0.9610438,0.028214201,0.0010843701,0.004864017,0.0039587966,0.0008348047],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008252601,0.002047981,0.0007596289,0.0027040257,0.00072354794,0.0031751222,0.0028463304,0.0026606154,0.010996874],"category_scores_gemma":[0.049996275,0.0005178452,0.0013526111,0.0015390026,0.0011720994,0.0033167943,0.0040737344,0.00238334,0.0056569385],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0025343979,0.004422948,0.018041762,0.005279575,0.0006456717,0.00077408366,0.0049986155,0.20393758,0.026211236,0.020448642,0.13655293,0.57615256],"study_design_scores_gemma":[0.00077801425,0.0020393045,0.009350726,0.0005819014,0.00014603391,0.0005415877,0.0024646423,0.8344962,0.032067668,0.025828343,0.09145207,0.0002535535],"about_ca_topic_score_codex":0.007070434,"about_ca_topic_score_gemma":0.0071007474,"teacher_disagreement_score":0.010996874,"about_ca_system_score_codex":0.0018051844,"about_ca_system_score_gemma":0.0019555432,"threshold_uncertainty_score":0.043644488},"labels":[],"label_agreement":null},{"id":"W4404390328","doi":"10.48550/arxiv.2411.07180","title":"Gumbel Counterfactual Generation From Language Models","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Institute for Catastrophic Loss Reduction","keywords":"Counterfactual thinking; Computer science; Linguistics; Natural language processing; Psychology; Philosophy; Social psychology","score_opus":0.07081071935022155,"score_gpt":0.21031131842669465,"score_spread":0.1395005990764731,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404390328","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012566863,0.00011218569,0.984674,0.00043959855,0.000030171628,0.000111945556,0.0001187895,0.00049109897,0.0014553278],"genre_scores_gemma":[0.4128053,0.000248083,0.58113784,0.00050489017,0.000087814726,0.0008716122,0.0005699416,0.00039086572,0.0033836146],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99006945,0.0066131954,0.00032346798,0.0012904739,0.0013681862,0.00033517703],"domain_scores_gemma":[0.94665444,0.046670105,0.001655127,0.0036526613,0.0010314543,0.00033614258],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01495408,0.0014051514,0.0013420199,0.0015840175,0.0010262242,0.003092938,0.0024114903,0.0017857125,0.0074802376],"category_scores_gemma":[0.062334143,0.0011063499,0.0025810797,0.0012596,0.003529761,0.005055586,0.0036347413,0.0045225997,0.00090563117],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036677133,0.00014131315,0.0016787011,0.00020682847,0.00013755368,0.00024706672,0.00070131803,0.35858187,0.0016341143,0.5656229,0.0024782447,0.06820323],"study_design_scores_gemma":[0.000042612148,0.000024366956,0.000080569516,0.0000197637,0.000014801255,0.000033150376,0.000021835422,0.73418605,0.00090669893,0.26378456,0.0008688186,0.000016735878],"about_ca_topic_score_codex":0.0023437813,"about_ca_topic_score_gemma":0.0032036388,"teacher_disagreement_score":0.01495408,"about_ca_system_score_codex":0.002539748,"about_ca_system_score_gemma":0.0019048968,"threshold_uncertainty_score":0.07908565},"labels":[],"label_agreement":null},{"id":"W4404445012","doi":"10.1609/aiide.v20i1.31877","title":"Procedural Content Generation in Games: A Survey with Insights on Emerging LLM Integration","year":2024,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence and Interactive Digital Entertainment","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Content (measure theory); Computer science; Psychology; Mathematics","score_opus":0.07120850862497619,"score_gpt":0.3041764801660764,"score_spread":0.23296797154110022,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404445012","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04249863,0.36216858,0.48452112,0.0037435708,0.0007260587,0.00081927306,0.001105705,0.0049193106,0.09949778],"genre_scores_gemma":[0.2340033,0.2803833,0.44793433,0.001792365,0.00096039457,0.0009284809,0.0035607386,0.0021398226,0.028297236],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9980775,0.00056857424,0.00018846485,0.0003285522,0.00071982906,0.00011710656],"domain_scores_gemma":[0.99455136,0.0041599423,0.000202931,0.00038446602,0.000547862,0.00015345757],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023901786,0.0010175442,0.0007245271,0.004776424,0.0004930195,0.004212911,0.0020305947,0.0014640472,0.0058596414],"category_scores_gemma":[0.009320173,0.0007368115,0.0009322913,0.004203633,0.0011900238,0.005246847,0.0017673098,0.0013101422,0.0031436773],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009026153,0.0002048494,0.0030674667,0.0043561333,0.00005462273,0.00010088192,0.00095162634,0.0052572014,0.0026645048,0.038505837,0.011306852,0.93343973],"study_design_scores_gemma":[0.000081666934,0.00042995904,0.008829708,0.00570476,0.00015456304,0.0021141488,0.002305625,0.09729222,0.015690781,0.06542968,0.8017831,0.00018390261],"about_ca_topic_score_codex":0.0017324763,"about_ca_topic_score_gemma":0.001702182,"teacher_disagreement_score":0.0058596414,"about_ca_system_score_codex":0.001047398,"about_ca_system_score_gemma":0.0010926276,"threshold_uncertainty_score":0.019602478},"labels":[],"label_agreement":null},{"id":"W4404573785","doi":"10.48550/arxiv.2411.12372","title":"RedPajama: an Open Dataset for Training Large Language Models","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Oak Ridge National Laboratory; Office of Naval Research; Canadian Institute for Advanced Research; U.S. Army Combat Capabilities Development Command; National Institutes of Health; National Science Foundation; VMware; Office of Science; Accenture; Canada Excellence Research Chairs, Government of Canada; U.S. Department of Energy","keywords":"Training (meteorology); Computer science; Natural language processing; Artificial intelligence; Geography","score_opus":0.145504564827458,"score_gpt":0.27491864055155424,"score_spread":0.12941407572409624,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404573785","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.050653666,0.002084165,0.059600137,0.0030375437,0.0013378486,0.00097633817,0.7735432,0.09344415,0.0153229805],"genre_scores_gemma":[0.03353515,0.0002674272,0.04875783,0.00065895,0.00009144544,0.0013178969,0.908457,0.0026425577,0.0042717815],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99642426,0.0012950543,0.00034610962,0.0008892807,0.0008054103,0.00023981584],"domain_scores_gemma":[0.98794633,0.0052575916,0.00053798064,0.004072009,0.0016609019,0.0005251275],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035705483,0.0019075748,0.0008241982,0.0022414757,0.0015700042,0.0021790266,0.004934833,0.0026272773,0.012918131],"category_scores_gemma":[0.02229944,0.00088616664,0.0021637557,0.0023376278,0.001255513,0.0045191916,0.0036863345,0.004599643,0.01930777],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008991411,0.0006749732,0.009797354,0.0016138287,0.00033260134,0.0004976931,0.00082237937,0.011699253,0.005779133,0.006435221,0.8686353,0.0928131],"study_design_scores_gemma":[0.0008287736,0.00048056463,0.017554523,0.00046889996,0.00018290334,0.00082306075,0.00085181306,0.099314794,0.014893134,0.015997883,0.8482597,0.00034394767],"about_ca_topic_score_codex":0.021950087,"about_ca_topic_score_gemma":0.04430265,"teacher_disagreement_score":0.021950087,"about_ca_system_score_codex":0.0013523259,"about_ca_system_score_gemma":0.00313902,"threshold_uncertainty_score":0.043644667},"labels":[],"label_agreement":null},{"id":"W4404752327","doi":"10.18653/v1/2022.aacl-main.39","title":"Director: Generator-Classifiers For Supervised Language Modeling","year":2022,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Generator (circuit theory); Artificial intelligence; Language model; Natural language processing; Physics; Power (physics)","score_opus":0.02489321284759761,"score_gpt":0.2742018675418575,"score_spread":0.2493086546942599,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404752327","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.002082816,0.00076470175,0.8979804,0.00053436385,0.0004278253,0.00016101197,0.0049116556,0.090255536,0.0028816238],"genre_scores_gemma":[0.08522369,0.00085818226,0.8390605,0.0009311634,0.0006496214,0.0011691324,0.037339527,0.009749952,0.025018279],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9976572,0.0009222451,0.00013095257,0.0006249695,0.0005229895,0.00014173075],"domain_scores_gemma":[0.9962675,0.0018855325,0.000110393856,0.0009859388,0.00059415423,0.00015641873],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035128852,0.0018441386,0.0015206893,0.0022952352,0.001114339,0.0022376508,0.0036626991,0.0024581135,0.020705163],"category_scores_gemma":[0.011624269,0.0015884928,0.0016271112,0.0014245503,0.0006892508,0.004883482,0.00292248,0.0039084693,0.026507279],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00070806633,0.00022506046,0.001659037,0.00031719127,0.0002389144,0.00016838593,0.00013289007,0.030914476,0.004207788,0.020118674,0.3771462,0.56416327],"study_design_scores_gemma":[0.00023849597,0.00009006234,0.00043225515,0.00005685028,0.00006368247,0.00014761387,0.000048839196,0.86668646,0.011198478,0.059568204,0.061424978,0.00004395454],"about_ca_topic_score_codex":0.0037219704,"about_ca_topic_score_gemma":0.008749693,"teacher_disagreement_score":0.020705163,"about_ca_system_score_codex":0.0008035719,"about_ca_system_score_gemma":0.0017256744,"threshold_uncertainty_score":0.06926566},"labels":[],"label_agreement":null},{"id":"W4404780584","doi":"10.18653/v1/2024.wmt-1.7","title":"MSLC24 Submissions to the General Machine Translation Task","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Computer science; Machine translation; Task (project management); Translation (biology); Natural language processing; Artificial intelligence; Information retrieval; Engineering; Systems engineering","score_opus":0.016635187454286823,"score_gpt":0.30386025433527164,"score_spread":0.2872250668809848,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404780584","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12307472,0.006367753,0.08163395,0.009800438,0.04232526,0.006001576,0.4376493,0.14417134,0.14897564],"genre_scores_gemma":[0.072880305,0.00066858943,0.055124857,0.001840213,0.0025660077,0.003275598,0.7596931,0.015935956,0.0880154],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98622096,0.0052251765,0.0010173657,0.0017582267,0.00464537,0.0011328011],"domain_scores_gemma":[0.97499555,0.0039803227,0.00062999054,0.0059884293,0.010963882,0.0034419612],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009488415,0.004519048,0.003140148,0.0034636145,0.002697795,0.0046262103,0.0026328845,0.0036189093,0.047806967],"category_scores_gemma":[0.035955314,0.00080846174,0.0019622003,0.0035960434,0.0012774136,0.0025044088,0.0070522595,0.0033342773,0.04723853],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005841692,0.00018597418,0.00037734857,0.0006223031,0.000131517,0.00028348452,0.00014444356,0.0016841142,0.004088953,0.00085717655,0.9483174,0.04272311],"study_design_scores_gemma":[0.0016105531,0.0014068006,0.007799956,0.00022849908,0.00014947541,0.0014373079,0.0008540175,0.029459165,0.02050296,0.010651986,0.9255773,0.00032190132],"about_ca_topic_score_codex":0.006494041,"about_ca_topic_score_gemma":0.016052945,"teacher_disagreement_score":0.047806967,"about_ca_system_score_codex":0.0021239684,"about_ca_system_score_gemma":0.003727631,"threshold_uncertainty_score":0.15993029},"labels":[],"label_agreement":null},{"id":"W4404780614","doi":"10.18653/v1/2024.wmt-1.94","title":"TRIBBLE - TRanslating IBerian languages Based on Limited E-resources","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Agencia Estatal de Investigación; HORIZON EUROPE Framework Programme; Generalitat de Catalunya; European Commission; McGill University","keywords":"Computer science; Natural language processing; Programming language; Artificial intelligence; Linguistics","score_opus":0.011317060625135088,"score_gpt":0.2748438908309091,"score_spread":0.263526830205774,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404780614","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16432837,0.0008484444,0.6076811,0.0012521659,0.00066601485,0.00052015594,0.009823397,0.030549878,0.18433046],"genre_scores_gemma":[0.5135256,0.00077178614,0.41522953,0.00042554757,0.00008540963,0.0002916501,0.010799775,0.00430974,0.054561023],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996859,0.00007856656,0.000031672877,0.00010358253,0.000054823202,0.000045505763],"domain_scores_gemma":[0.99961275,0.000110478846,0.000022210847,0.000106386484,0.00013322594,0.00001506526],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00043504522,0.0004822345,0.00033128273,0.0009469863,0.00073149434,0.0017125272,0.00041977395,0.00037927154,0.014240742],"category_scores_gemma":[0.0010597979,0.00030758,0.000525773,0.00072020805,0.00045664335,0.0019726867,0.0014621922,0.0006447232,0.003782454],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009263952,0.0002637986,0.0040699085,0.0016062749,0.00009598116,0.0016101875,0.0052981754,0.017420484,0.07778529,0.32949343,0.04244818,0.51898193],"study_design_scores_gemma":[0.00012332504,0.00021277866,0.0041472665,0.0008397153,0.00021015358,0.001117074,0.0033248714,0.090449855,0.11415933,0.10506965,0.6802464,0.00009962327],"about_ca_topic_score_codex":0.0069588544,"about_ca_topic_score_gemma":0.008679595,"teacher_disagreement_score":0.014240742,"about_ca_system_score_codex":0.00077084976,"about_ca_system_score_gemma":0.001648629,"threshold_uncertainty_score":0.047640026},"labels":[],"label_agreement":null},{"id":"W4404780647","doi":"10.18653/v1/2024.wmt-1.129","title":"Efficient Technical Term Translation: A Knowledge Distillation Approach for Parenthetical Terminology Translation","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Prompt (Canada)","funders":"","keywords":"Terminology; Term (time); Translation (biology); Computer science; Natural language processing; Distillation; Artificial intelligence; Information retrieval; Linguistics; Chemistry","score_opus":0.040001263061061054,"score_gpt":0.3264166708705529,"score_spread":0.2864154078094919,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404780647","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024397941,0.0015367779,0.95001096,0.0010309463,0.0005724935,0.00018862415,0.0019872803,0.014036574,0.0062383614],"genre_scores_gemma":[0.24956328,0.0006800754,0.7335729,0.0007083889,0.00031919862,0.00034860292,0.006933538,0.0014998165,0.006374175],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99745077,0.0010624749,0.00017970333,0.0008277744,0.00033836704,0.00014088702],"domain_scores_gemma":[0.99619955,0.001627946,0.0002183164,0.0012595921,0.0005793869,0.00011523496],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002755073,0.0017465824,0.0010914458,0.0019266622,0.0011039759,0.0022509238,0.0026616012,0.001961426,0.0063065067],"category_scores_gemma":[0.012632702,0.00055802235,0.0013704677,0.002011994,0.0013261707,0.0048339586,0.0044266544,0.004023057,0.006264174],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00044692995,0.00027903638,0.0011259593,0.0010105016,0.00015633054,0.00042054732,0.00090637506,0.068355285,0.03589723,0.033599228,0.03273399,0.8250685],"study_design_scores_gemma":[0.0001694697,0.00028960806,0.00060581014,0.0001583106,0.00014404395,0.0005402615,0.0004150832,0.7967118,0.04042843,0.101652674,0.05875455,0.00012995403],"about_ca_topic_score_codex":0.0028745965,"about_ca_topic_score_gemma":0.0062091574,"teacher_disagreement_score":0.0063065067,"about_ca_system_score_codex":0.0009791389,"about_ca_system_score_gemma":0.0026105014,"threshold_uncertainty_score":0.021097362},"labels":[],"label_agreement":null},{"id":"W4404780849","doi":"10.18653/v1/2024.findings-emnlp.730","title":"“Knowing When You Don’t Know”: A Multilingual Relevance Assessment Dataset for Robust Retrieval-Augmented Generation","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Relevance (law); Computer science; Information retrieval; Relevance feedback; Artificial intelligence; Natural language processing; Image retrieval; Image (mathematics); Political science","score_opus":0.03810460359996788,"score_gpt":0.34331725208100444,"score_spread":0.3052126484810366,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404780849","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17754558,0.006327914,0.031710155,0.0031815371,0.0013862391,0.0018916749,0.7306902,0.022599604,0.024667047],"genre_scores_gemma":[0.07440202,0.00037929014,0.031920936,0.0004152814,0.000089032685,0.0006277326,0.88566595,0.0004908646,0.006008856],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.997938,0.00072819006,0.00022260641,0.00049223925,0.0004567746,0.00016221317],"domain_scores_gemma":[0.9963469,0.0012617671,0.00014457534,0.0010720008,0.00081063196,0.0003642148],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022593946,0.0014830853,0.00095965096,0.003028274,0.0015229551,0.0015203744,0.0028913787,0.0026140756,0.008990653],"category_scores_gemma":[0.0072116246,0.0003871883,0.0013068776,0.0021408857,0.0007013962,0.0019324815,0.0034271977,0.00162041,0.01079025],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020984907,0.0013609002,0.012442885,0.0023118274,0.00033804323,0.0010331065,0.00091439165,0.00557552,0.011299315,0.00324418,0.7636592,0.19572225],"study_design_scores_gemma":[0.0018197909,0.001246936,0.055122934,0.0007293297,0.00049628527,0.0026395975,0.0037042692,0.08521054,0.030142391,0.009699314,0.808658,0.00053060567],"about_ca_topic_score_codex":0.023031112,"about_ca_topic_score_gemma":0.061046764,"teacher_disagreement_score":0.023031112,"about_ca_system_score_codex":0.0011303293,"about_ca_system_score_gemma":0.0018348275,"threshold_uncertainty_score":0.04579413},"labels":[],"label_agreement":null},{"id":"W4404780891","doi":"10.18653/v1/2024.wmt-1.0","title":"Front Matter","year":2024,"lang":"en","type":"paratext","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Samsung; International Institute of Information Technology, Hyderabad; Université de Genève; Shanghai Jiao Tong University; Indian Institute of Technology Madras; Universidad de Alicante; McGill University","keywords":"Front (military); Computer science; Geology; Oceanography","score_opus":0.010019419453597547,"score_gpt":0.2801557049540283,"score_spread":0.2701362855004308,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404780891","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0006776587,0.0030916487,0.0007744358,0.0033786118,0.0033774164,0.00008773729,0.0064235874,0.0017773598,0.9804115],"genre_scores_gemma":[0.003075438,0.0017176571,0.0006963169,0.0011969967,0.0003403315,0.000054170283,0.0035603598,0.00084455987,0.98851424],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99857855,0.00018895873,0.000094505456,0.0005810508,0.00033915206,0.00021783807],"domain_scores_gemma":[0.9984282,0.00032768058,0.00010430837,0.00029916657,0.00028312716,0.0005575299],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.0008736602,0.0017555293,0.0015552597,0.0019957456,0.0022220248,0.009086626,0.0020910087,0.004552631,0.9233759],"category_scores_gemma":[0.0029721486,0.00050532026,0.0009467446,0.0017549612,0.001191402,0.0035537127,0.0039581847,0.0019279426,0.89817846],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023394648,0.000082901264,0.0005296222,0.001065078,0.000016275044,0.00049424346,0.0002140754,0.00011364267,0.001289342,0.01112191,0.7566352,0.22820374],"study_design_scores_gemma":[0.00001076558,0.000009806052,0.00023846829,0.00014102715,0.0000020774155,0.00012515076,0.00008036313,0.00002959565,0.0001073485,0.00061170873,0.9986401,0.0000034935892],"about_ca_topic_score_codex":0.0015233438,"about_ca_topic_score_gemma":0.0023578205,"teacher_disagreement_score":0.076624095,"about_ca_system_score_codex":0.0016038388,"about_ca_system_score_gemma":0.0018631421,"threshold_uncertainty_score":0.10929495},"labels":[],"label_agreement":null},{"id":"W4404780957","doi":"10.18653/v1/2024.wmt-1.2","title":"Are LLMs Breaking MT Metrics? Results of the WMT24 Metrics Shared Task","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"HORIZON EUROPE Framework Programme; Canadian Institute for Advanced Research","keywords":"Task (project management); Computer science; Engineering; Systems engineering","score_opus":0.025259916088926266,"score_gpt":0.28965755757766326,"score_spread":0.264397641488737,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404780957","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.48298395,0.06802027,0.040217523,0.040944725,0.008879437,0.0016729931,0.21472327,0.03563385,0.106924],"genre_scores_gemma":[0.6406183,0.0024041713,0.029835619,0.004097752,0.0017320858,0.0013470864,0.28726357,0.009027301,0.023674082],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9581701,0.025502075,0.0031683224,0.004237487,0.0073363227,0.0015857173],"domain_scores_gemma":[0.90397483,0.043697596,0.0045290682,0.022286713,0.019228423,0.006283388],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.032749847,0.0041107177,0.0034223997,0.0059768846,0.0037782749,0.005519695,0.004322258,0.005798603,0.0140930135],"category_scores_gemma":[0.11678567,0.0008803181,0.0016844574,0.0051501035,0.0024250906,0.009723001,0.0099624405,0.0037877741,0.018365864],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0035913426,0.0011166864,0.014871537,0.0029146457,0.0011663876,0.00044055647,0.0019526518,0.0072612944,0.0054452284,0.006016398,0.6173369,0.33788636],"study_design_scores_gemma":[0.0049444432,0.0040557985,0.12129384,0.0019865797,0.002573169,0.0025499992,0.0073971944,0.09796302,0.032242496,0.10384746,0.620051,0.0010950313],"about_ca_topic_score_codex":0.023683835,"about_ca_topic_score_gemma":0.035408393,"teacher_disagreement_score":0.96725017,"about_ca_system_score_codex":0.0029652992,"about_ca_system_score_gemma":0.003614292,"threshold_uncertainty_score":0.17319983},"labels":[],"label_agreement":null},{"id":"W4404781398","doi":"10.18653/v1/2024.wmt-1.34","title":"MSLC24: Further Challenges for Metrics on a Wide Landscape of Translation Quality","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Computer science; Quality (philosophy); Translation (biology)","score_opus":0.0743364805648641,"score_gpt":0.35327099193154154,"score_spread":0.27893451136667746,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404781398","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11414048,0.027463935,0.63353264,0.112728,0.006730686,0.00080762256,0.023066292,0.024424966,0.057105355],"genre_scores_gemma":[0.5071193,0.0031023992,0.40511706,0.010094204,0.0039088465,0.0018880032,0.03781725,0.016603775,0.014349158],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.8882637,0.05099655,0.009554603,0.009711482,0.038868245,0.0026054203],"domain_scores_gemma":[0.7100202,0.14067534,0.008153524,0.049303755,0.0848173,0.007029896],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.062368833,0.002524357,0.0038356087,0.008653986,0.0033319122,0.013841326,0.0052382876,0.005014976,0.007755368],"category_scores_gemma":[0.2833865,0.0008973526,0.0019020125,0.012603847,0.0048308885,0.018178254,0.011719384,0.008781168,0.0060622236],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005733952,0.0004209294,0.018988198,0.0015716725,0.00038703016,0.00023248345,0.002067593,0.0150185665,0.005030997,0.12009726,0.31231764,0.5232942],"study_design_scores_gemma":[0.00026916288,0.0012666071,0.022932494,0.001451932,0.00013909108,0.0013075079,0.0031287004,0.1491973,0.014873969,0.4329073,0.37173855,0.0007872418],"about_ca_topic_score_codex":0.0095443595,"about_ca_topic_score_gemma":0.008085733,"teacher_disagreement_score":0.9376312,"about_ca_system_score_codex":0.004589019,"about_ca_system_score_gemma":0.003662941,"threshold_uncertainty_score":0.3298418},"labels":[],"label_agreement":null},{"id":"W4404781442","doi":"10.18653/v1/2024.nlp4science-1.3","title":"What an Elegant Bridge: Multilingual LLMs are Biased Similarly in Different Languages","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"Engineering and Physical Sciences Research Council","keywords":"Bridge (graph theory); Computer science; Linguistics; Natural language processing; Medicine; Philosophy","score_opus":0.025035495069478882,"score_gpt":0.3316981056926324,"score_spread":0.30666261062315353,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404781442","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8392336,0.0014143547,0.10662487,0.015366378,0.00050024793,0.00006089509,0.00095441536,0.0008845079,0.034960732],"genre_scores_gemma":[0.98833174,0.00015468578,0.008092729,0.0016566891,0.00010434761,0.000025196025,0.00018048126,0.00025774314,0.0011963707],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9966072,0.0013843223,0.00015605507,0.0010865644,0.00050316553,0.00026263585],"domain_scores_gemma":[0.9857735,0.006759947,0.0017470617,0.0039596586,0.0011361302,0.00062375586],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006338582,0.00047678486,0.000688489,0.0008964378,0.0007676705,0.0035953359,0.00080935436,0.0010191717,0.0049597616],"category_scores_gemma":[0.045886435,0.00036704363,0.00052531285,0.00057758664,0.0030539553,0.008720725,0.0031229372,0.0018877103,0.0012002359],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017053046,0.00026172638,0.31403998,0.0006031058,0.0016773636,0.0006290831,0.035825748,0.007869855,0.048054706,0.24026185,0.0151950335,0.33387628],"study_design_scores_gemma":[0.000112567745,0.0003539186,0.119874835,0.00031017998,0.00042655895,0.001183068,0.01621232,0.028236084,0.020186136,0.7795327,0.033204,0.00036768836],"about_ca_topic_score_codex":0.002393961,"about_ca_topic_score_gemma":0.0024317894,"teacher_disagreement_score":0.006338582,"about_ca_system_score_codex":0.00064818724,"about_ca_system_score_gemma":0.00050912786,"threshold_uncertainty_score":0.03352201},"labels":[],"label_agreement":null},{"id":"W4404781502","doi":"10.18653/v1/2024.wmt-1.4","title":"Findings of the WMT 2024 Shared Task of the Open Language Data Initiative","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Task (project management); Open data; Natural language processing; Artificial intelligence; World Wide Web; Engineering","score_opus":0.046453455072534684,"score_gpt":0.3402527732463607,"score_spread":0.29379931817382604,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404781502","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.23029111,0.0077599175,0.022185402,0.13787729,0.0150914285,0.0068178996,0.5026994,0.005107357,0.072170146],"genre_scores_gemma":[0.23197682,0.0018568074,0.040680796,0.024595745,0.0059691817,0.0152964275,0.63658315,0.008391565,0.03464956],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.7895513,0.12240606,0.010743006,0.013089027,0.05574883,0.00846175],"domain_scores_gemma":[0.5153773,0.21360803,0.022257231,0.05937359,0.13942258,0.04996134],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.19236808,0.0022115884,0.0028361226,0.008266991,0.008994604,0.012286242,0.0051937494,0.005377011,0.012903517],"category_scores_gemma":[0.4065212,0.0011225225,0.0023161364,0.010065706,0.006508264,0.009779027,0.03246406,0.005786396,0.011615233],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0022293082,0.0011316945,0.017110044,0.002525672,0.00034899326,0.000623077,0.0072414703,0.0013939005,0.002104235,0.006921589,0.93083924,0.027530802],"study_design_scores_gemma":[0.0015042855,0.00088477426,0.13138993,0.0020681599,0.00044304258,0.0009580482,0.02056787,0.0045072692,0.005364373,0.018464362,0.8132006,0.0006473369],"about_ca_topic_score_codex":0.044391993,"about_ca_topic_score_gemma":0.05144248,"teacher_disagreement_score":0.19236808,"about_ca_system_score_codex":0.007322373,"about_ca_system_score_gemma":0.018505732,"threshold_uncertainty_score":0.9959539},"labels":[],"label_agreement":null},{"id":"W4404781603","doi":"10.18653/v1/2024.mrl-1.28","title":"McGill NLP Group Submission to the MRL 2024 Shared Task: Ensembling Enhances Effectiveness of Multilingual Small LMs","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"Canadian Institute for Advanced Research","keywords":"Computer science; Task (project management); Natural language processing; Artificial intelligence; Group (periodic table); Information retrieval; Engineering","score_opus":0.018091459720084805,"score_gpt":0.29730238933404657,"score_spread":0.27921092961396177,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404781603","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.27950746,0.0031858461,0.37076125,0.011526095,0.008211777,0.003788339,0.04278821,0.22682834,0.053402748],"genre_scores_gemma":[0.50601596,0.00040165303,0.3123868,0.002228486,0.0011507053,0.0024060144,0.10178485,0.015195407,0.058430113],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9922357,0.0036872546,0.00037549066,0.0018211947,0.0013099544,0.0005704207],"domain_scores_gemma":[0.97452956,0.010608725,0.0003600342,0.006479964,0.0058657946,0.0021559615],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.014620698,0.0029944542,0.002963825,0.0018832424,0.002415975,0.0028145888,0.0044922894,0.0034371042,0.031124087],"category_scores_gemma":[0.042158492,0.0011645961,0.0014657136,0.0013087534,0.0011951472,0.0047790264,0.008600516,0.0034247148,0.0144310985],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0034633735,0.00096688076,0.0037024599,0.00089255284,0.00036883258,0.0011249011,0.00083926384,0.028091034,0.03999147,0.002547129,0.5499987,0.36801338],"study_design_scores_gemma":[0.0022645432,0.0017770637,0.01111977,0.0001717201,0.0002792973,0.00088219467,0.000805578,0.6933139,0.08676913,0.017040698,0.18503547,0.0005406085],"about_ca_topic_score_codex":0.019707425,"about_ca_topic_score_gemma":0.03738779,"teacher_disagreement_score":0.031124087,"about_ca_system_score_codex":0.0021746205,"about_ca_system_score_gemma":0.003957653,"threshold_uncertainty_score":0.10412043},"labels":[],"label_agreement":null},{"id":"W4404781713","doi":"10.18653/v1/2024.findings-emnlp.203","title":"NormTab: Improving Symbolic Reasoning in LLMs Through Tabular Data Normalization","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Innovates","keywords":"Normalization (sociology); Computer science; Deductive reasoning; Artificial intelligence; Social science; Sociology","score_opus":0.02040650243535104,"score_gpt":0.3017446784371425,"score_spread":0.28133817600179145,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404781713","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011386005,0.00057644927,0.86553246,0.0008401489,0.00019247776,0.00023122014,0.004204101,0.11486517,0.0021719146],"genre_scores_gemma":[0.081865646,0.00032070116,0.8976169,0.0006485293,0.00008052157,0.0003440398,0.012190433,0.0046346,0.0022986275],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9932668,0.0019045764,0.0008787866,0.0012348037,0.002440441,0.00027455116],"domain_scores_gemma":[0.98414385,0.007084713,0.0010670209,0.005073857,0.0023920278,0.00023852498],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0043444997,0.0017235252,0.0012276577,0.0031233456,0.001041071,0.0042237747,0.003970212,0.0009803071,0.007639419],"category_scores_gemma":[0.033275105,0.0008223401,0.0023830181,0.0037947134,0.0015791412,0.008174901,0.004240025,0.0030335882,0.003974897],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00055925164,0.0005091411,0.008403797,0.0014789328,0.00028017838,0.00040727635,0.0011011248,0.049912788,0.025778698,0.050824556,0.07630887,0.7844355],"study_design_scores_gemma":[0.000120657096,0.00011219343,0.0013583845,0.00021387727,0.00013167651,0.00026872987,0.00048586848,0.7766272,0.0670978,0.09496956,0.058492128,0.00012204526],"about_ca_topic_score_codex":0.012049016,"about_ca_topic_score_gemma":0.021834128,"teacher_disagreement_score":0.012049016,"about_ca_system_score_codex":0.0022458772,"about_ca_system_score_gemma":0.0062019383,"threshold_uncertainty_score":0.025556386},"labels":[],"label_agreement":null},{"id":"W4404781890","doi":"10.18653/v1/2024.findings-emnlp.191","title":"Are Large Vision Language Models up to the Challenge of Chart Comprehension and Reasoning","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Compute Canada; Centre International de Recherche sur le Cancer","keywords":"Computer science; Comprehension; Chart; Natural language processing; Artificial intelligence; Programming language; Statistics; Mathematics","score_opus":0.01876255332833101,"score_gpt":0.3026275532543308,"score_spread":0.2838649999259998,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404781890","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07205354,0.0015781726,0.84195584,0.01079457,0.0005368591,0.0005907737,0.0034495024,0.050956875,0.018083964],"genre_scores_gemma":[0.45440507,0.0010265302,0.5250527,0.003260311,0.00019903488,0.00076636934,0.0071150404,0.0031725438,0.005002451],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99523336,0.0023483585,0.00027543708,0.00094752223,0.0009985287,0.00019683398],"domain_scores_gemma":[0.98055583,0.012874598,0.001072237,0.002810073,0.0021001764,0.00058706],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005869419,0.0009888357,0.000610455,0.00087540934,0.00048263487,0.0044972845,0.0024572122,0.0019878773,0.0071415147],"category_scores_gemma":[0.041153852,0.000638946,0.0015399592,0.00051243394,0.0016547155,0.0112336185,0.0029357865,0.0035494773,0.0034044485],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011203792,0.0007341995,0.007702484,0.002828121,0.0002990989,0.00079298497,0.009733825,0.056623194,0.039693028,0.13098687,0.11602056,0.63346523],"study_design_scores_gemma":[0.00026696132,0.00047916232,0.0030383111,0.00075003935,0.00023271039,0.0006301675,0.003798612,0.550606,0.033352047,0.24853632,0.1580599,0.00024979707],"about_ca_topic_score_codex":0.004917643,"about_ca_topic_score_gemma":0.004928255,"teacher_disagreement_score":0.0071415147,"about_ca_system_score_codex":0.0018569601,"about_ca_system_score_gemma":0.0029622838,"threshold_uncertainty_score":0.031040847},"labels":[],"label_agreement":null},{"id":"W4404782131","doi":"10.18653/v1/2024.findings-emnlp.787","title":"Exploring Quantization for Efficient Pre-Training of Transformer Language Models","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Fonds de recherche du Québec – Nature et technologies; Alliance de recherche numérique du Canada; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Computer science; Transformer; Quantization (signal processing); Language model; Artificial intelligence; Electrical engineering; Engineering; Algorithm; Voltage","score_opus":0.101602499662378,"score_gpt":0.32295577288032534,"score_spread":0.22135327321794734,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404782131","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.046553027,0.0005439056,0.9480037,0.00033762207,0.00006503148,0.000088863766,0.00009485298,0.0027982406,0.0015146958],"genre_scores_gemma":[0.70925885,0.00033651027,0.2874251,0.00034245625,0.00003060385,0.00019445062,0.000312283,0.0004997247,0.0015999787],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99935573,0.00022505988,0.00005317883,0.00016724505,0.000104967236,0.00009386644],"domain_scores_gemma":[0.99715686,0.0020156368,0.00012786209,0.0003184262,0.0002991032,0.00008205258],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017933429,0.0009925838,0.00065087137,0.00036049413,0.00036165796,0.0011833919,0.0014297795,0.00090111484,0.0030529057],"category_scores_gemma":[0.012468734,0.00060676725,0.0004312285,0.00037013475,0.00080807356,0.0024089038,0.0018068142,0.0028398726,0.0010373114],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00052963605,0.00031300483,0.0025691383,0.00038436527,0.000087253575,0.00018794237,0.000529661,0.52337056,0.04626537,0.019709915,0.0040630894,0.40199003],"study_design_scores_gemma":[0.000030280893,0.000087925946,0.00021729509,0.000029603874,0.000014697432,0.000034871347,0.00004799115,0.9762264,0.013495621,0.008946894,0.0008547641,0.000013649826],"about_ca_topic_score_codex":0.003885462,"about_ca_topic_score_gemma":0.006421485,"teacher_disagreement_score":0.003885462,"about_ca_system_score_codex":0.0006680988,"about_ca_system_score_gemma":0.0016144541,"threshold_uncertainty_score":0.010213017},"labels":[],"label_agreement":null},{"id":"W4404782142","doi":"10.18653/v1/2024.findings-emnlp.155","title":"MINERS: Multilingual Language Models as Semantic Retrievers","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Canadian Institute for Advanced Research","keywords":"Computer science; Natural language processing; Artificial intelligence; Semantic computing; Linguistics; Semantic Web","score_opus":0.014415151271030107,"score_gpt":0.2992104052329831,"score_spread":0.284795253961953,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404782142","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1053097,0.0044732103,0.84389293,0.0018307167,0.00049744896,0.0004630221,0.007445396,0.025091508,0.01099603],"genre_scores_gemma":[0.5069161,0.0020949969,0.45877853,0.0010098699,0.00029012197,0.0006210421,0.01768345,0.0014882042,0.011117667],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9982729,0.0007984776,0.00011560502,0.0004341819,0.0002745212,0.00010430397],"domain_scores_gemma":[0.99834085,0.0008780184,0.0001246961,0.00038490398,0.00020839635,0.0000631001],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023715748,0.0018134769,0.0010866756,0.001770518,0.0004954616,0.0021722512,0.0022590207,0.0013098258,0.005574932],"category_scores_gemma":[0.007845452,0.00064880593,0.0017668061,0.0013375755,0.0006379967,0.0056390734,0.0033479899,0.0023240778,0.0046291556],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012843453,0.0006715807,0.0072786165,0.0009970382,0.00078974274,0.00050659495,0.00076409715,0.17392972,0.011945107,0.042482007,0.032601725,0.72674954],"study_design_scores_gemma":[0.00006313453,0.00018328753,0.0006013774,0.00008257273,0.000095384225,0.00023599742,0.00022442878,0.9400545,0.0045440826,0.043188732,0.010682552,0.000043930228],"about_ca_topic_score_codex":0.002379948,"about_ca_topic_score_gemma":0.0057278536,"teacher_disagreement_score":0.005574932,"about_ca_system_score_codex":0.00059308886,"about_ca_system_score_gemma":0.0010315129,"threshold_uncertainty_score":0.018650055},"labels":[],"label_agreement":null},{"id":"W4404782170","doi":"10.18653/v1/2024.findings-emnlp.941","title":"Gazelle: An Instruction Dataset for Arabic Writing Assistance","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Arabic; Computer science; Natural language processing; Artificial intelligence; Linguistics; Philosophy","score_opus":0.02053628075947235,"score_gpt":0.32103529631440136,"score_spread":0.300499015554929,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404782170","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07907641,0.0018959567,0.011702667,0.0019149946,0.000653213,0.00087577605,0.8342808,0.041565634,0.028034559],"genre_scores_gemma":[0.04630038,0.00037134707,0.018946752,0.00031866276,0.000059214824,0.0008144707,0.92304265,0.0006229667,0.009523579],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9992361,0.00015973212,0.000120403674,0.00019155447,0.0002256439,0.0000665601],"domain_scores_gemma":[0.9982368,0.00042625557,0.000119069475,0.00043170215,0.0006411529,0.00014495778],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00057481055,0.0015728126,0.000500328,0.002601111,0.0010019243,0.00093796366,0.0016554381,0.0016257301,0.016614262],"category_scores_gemma":[0.0047501116,0.00023143385,0.00061907497,0.002098795,0.0005188049,0.0014103894,0.0019809024,0.0014665297,0.023749242],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00041259083,0.00038385476,0.005945729,0.0011806876,0.00005756443,0.0005335331,0.0007785686,0.0025392016,0.005681495,0.0014782485,0.84920067,0.13180792],"study_design_scores_gemma":[0.0002491655,0.00025959098,0.019378243,0.00031590427,0.0000482579,0.00064903224,0.0016403041,0.019431679,0.010841634,0.0028804326,0.94417214,0.00013366435],"about_ca_topic_score_codex":0.016443795,"about_ca_topic_score_gemma":0.040198922,"teacher_disagreement_score":0.016614262,"about_ca_system_score_codex":0.0009101345,"about_ca_system_score_gemma":0.0015168121,"threshold_uncertainty_score":0.05558026},"labels":[],"label_agreement":null},{"id":"W4404782543","doi":"10.18653/v1/2024.findings-emnlp.39","title":"LLM-supertagger: Categorial Grammar Supertagging via Large Language Models","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Combinatory categorial grammar; Categorial grammar; Computer science; Mildly context-sensitive grammar formalism; Grammar; Natural language processing; Link grammar; Generative grammar; Linguistics; Head-driven phrase structure grammar; Artificial intelligence; Programming language; Emergent grammar; Philosophy","score_opus":0.009230419870475823,"score_gpt":0.26237751352088967,"score_spread":0.2531470936504138,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404782543","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017131355,0.00047164832,0.9529677,0.0006370748,0.00019667482,0.00009444826,0.0011704788,0.025044164,0.0022864742],"genre_scores_gemma":[0.38600126,0.00038000694,0.59574705,0.0009834949,0.00019063972,0.00036871425,0.0052779773,0.0028280038,0.008222805],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992731,0.00028006305,0.000034926234,0.0002486181,0.000111754234,0.00005154187],"domain_scores_gemma":[0.99717134,0.0017205083,0.00010921451,0.0007048532,0.00019794081,0.00009612573],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016511069,0.0012029656,0.00081611227,0.0012999399,0.00053978845,0.0014380957,0.002138734,0.001415719,0.007692156],"category_scores_gemma":[0.0068747182,0.0006835714,0.0012469064,0.0010543413,0.0006896017,0.0047135083,0.002869858,0.0029205878,0.004358981],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003757526,0.00024758422,0.0036188175,0.000518368,0.00028239973,0.00057609513,0.0007507089,0.11997881,0.02438239,0.053384308,0.035578247,0.76030654],"study_design_scores_gemma":[0.000025698913,0.00003826768,0.000274416,0.000027062351,0.000029460294,0.000074466305,0.000053400578,0.9130385,0.0051211924,0.07524422,0.006049373,0.000023858265],"about_ca_topic_score_codex":0.0029954908,"about_ca_topic_score_gemma":0.009596803,"teacher_disagreement_score":0.007692156,"about_ca_system_score_codex":0.00099679,"about_ca_system_score_gemma":0.0013251536,"threshold_uncertainty_score":0.025732815},"labels":[],"label_agreement":null},{"id":"W4404782547","doi":"10.18653/v1/2024.emnlp-main.277","title":"KidLM: Advancing Language Models for Children – Early Insights and Future Directions","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science","score_opus":0.005206888338875873,"score_gpt":0.25324254869188434,"score_spread":0.24803566035300847,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404782547","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023832139,0.010029631,0.9229845,0.015543495,0.00033992616,0.00019950459,0.0042714113,0.015246696,0.007552756],"genre_scores_gemma":[0.10340747,0.0069227503,0.87609017,0.0012295168,0.00019175645,0.00043198024,0.0058047036,0.0015409915,0.0043806424],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99684274,0.0018517406,0.00018015163,0.00046541207,0.00051346124,0.00014653378],"domain_scores_gemma":[0.9862753,0.009824975,0.00027695377,0.0014608466,0.0015445236,0.0006173362],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008618346,0.001535817,0.0010850163,0.0013570376,0.00049303786,0.0037867832,0.0037481724,0.0020234808,0.00854301],"category_scores_gemma":[0.021852436,0.0012694563,0.0011387689,0.001040429,0.001333749,0.012003776,0.0034841627,0.0050664917,0.004409661],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005284741,0.0005321631,0.019360863,0.0019562226,0.0002980314,0.0006010419,0.0045698807,0.09952595,0.01063227,0.13112766,0.084239505,0.6466279],"study_design_scores_gemma":[0.00004932669,0.00021136407,0.0027182656,0.0012974299,0.00008120515,0.0004553456,0.0014469563,0.6252653,0.009639047,0.15043215,0.20824496,0.00015868459],"about_ca_topic_score_codex":0.013646511,"about_ca_topic_score_gemma":0.017913023,"teacher_disagreement_score":0.013646511,"about_ca_system_score_codex":0.0021632058,"about_ca_system_score_gemma":0.0027044257,"threshold_uncertainty_score":0.04557866},"labels":[],"label_agreement":null},{"id":"W4404782641","doi":"10.18653/v1/2024.emnlp-main.142","title":"Backward Lens: Projecting Language Model Gradients into the Vocabulary Space","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Azrieli Foundation; Tel Aviv University; Open Philanthropy Project","keywords":"Computer science; Space (punctuation); Vocabulary; Lens (geology); Artificial intelligence; Natural language processing; Linguistics; Optics; Physics; Philosophy","score_opus":0.015988609976201915,"score_gpt":0.28648215201607924,"score_spread":0.2704935420398773,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404782641","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0037981747,0.00008209582,0.9927953,0.00025474082,0.000047027544,0.000033461314,0.00019083629,0.0014457735,0.0013526352],"genre_scores_gemma":[0.2039491,0.0005687354,0.77370054,0.00056790136,0.00016270476,0.0003130034,0.0013332402,0.0017149398,0.017689912],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99940634,0.00015772425,0.000032171556,0.00014269038,0.00019952898,0.000061552724],"domain_scores_gemma":[0.99898165,0.00037505614,0.00009177158,0.00024354241,0.0002241011,0.00008372936],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013380788,0.0013567577,0.00060905673,0.0008282693,0.00050286774,0.0021246113,0.001532663,0.00087442016,0.011955009],"category_scores_gemma":[0.0074244747,0.000719122,0.0010842826,0.000712015,0.0013783908,0.0043960414,0.0029489123,0.0031084511,0.004712448],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00033003403,0.00016437031,0.0019374364,0.00034360192,0.00014041718,0.00029777785,0.00061871763,0.13507937,0.020532217,0.340041,0.02506898,0.47544608],"study_design_scores_gemma":[0.00003870963,0.0000748908,0.0003075616,0.000045688805,0.000019637058,0.000117747986,0.000071225884,0.77589834,0.010019013,0.20612803,0.00724708,0.000031987125],"about_ca_topic_score_codex":0.0060585756,"about_ca_topic_score_gemma":0.009224589,"teacher_disagreement_score":0.011955009,"about_ca_system_score_codex":0.00082419504,"about_ca_system_score_gemma":0.00196355,"threshold_uncertainty_score":0.039993525},"labels":[],"label_agreement":null},{"id":"W4404782770","doi":"10.18653/v1/2024.emnlp-main.422","title":"EAGLE-2: Faster Inference of Language Models with Dynamic Draft Trees","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; National Natural Science Foundation of China; Compute Canada","keywords":"Eagle; Computer science; Inference; Artificial intelligence; Natural language processing; Geology","score_opus":0.009488952044039406,"score_gpt":0.27436083384419935,"score_spread":0.26487188180015997,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404782770","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012330908,0.0002874288,0.9558314,0.00021360972,0.00009356834,0.00010935941,0.00067015906,0.02924705,0.0012165011],"genre_scores_gemma":[0.13942917,0.00016572374,0.8508897,0.00031772122,0.00008023214,0.00015398276,0.0034003316,0.0024425436,0.003120586],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99798846,0.0006519596,0.00014270974,0.00051233306,0.0005762788,0.00012820376],"domain_scores_gemma":[0.9926952,0.0048697507,0.0002099472,0.0014871423,0.00058203546,0.00015603335],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033668526,0.0013410284,0.0010477169,0.0011406227,0.00042346978,0.0018330895,0.002950358,0.0013537091,0.008322139],"category_scores_gemma":[0.017369697,0.0010079392,0.0015804704,0.0009393852,0.00061942614,0.0056528174,0.002046021,0.0025651741,0.0035611384],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00095117494,0.0002973112,0.0065732454,0.0006939112,0.00045963473,0.0004219501,0.0008140378,0.2204627,0.020924985,0.030093877,0.039299183,0.679008],"study_design_scores_gemma":[0.000072364426,0.000043064072,0.00030150323,0.000014393203,0.000024156894,0.000087288,0.000048837745,0.9776409,0.0037952373,0.013415104,0.004539743,0.000017447977],"about_ca_topic_score_codex":0.0074881753,"about_ca_topic_score_gemma":0.016039483,"teacher_disagreement_score":0.008322139,"about_ca_system_score_codex":0.00087907165,"about_ca_system_score_gemma":0.0017443872,"threshold_uncertainty_score":0.027840376},"labels":[],"label_agreement":null},{"id":"W4404782906","doi":"10.18653/v1/2024.emnlp-main.251","title":"Voices Unheard: NLP Resources and Models for Yorùbá Regional Dialects","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Canadian Institute for Advanced Research; National Science Foundation","keywords":"Natural language processing; Computer science; Artificial intelligence; Yoruba; Linguistics; Philosophy","score_opus":0.025282369405437523,"score_gpt":0.2852018835029734,"score_spread":0.25991951409753583,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404782906","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3952491,0.006121217,0.4873688,0.0065243016,0.00044876034,0.00036527263,0.046372406,0.020152684,0.03739757],"genre_scores_gemma":[0.8244544,0.0012520744,0.12229258,0.0002694114,0.00006500224,0.0004405944,0.034689095,0.0013587277,0.015178204],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99982303,0.00008748298,0.000010268371,0.00004291217,0.00001865902,0.000017598039],"domain_scores_gemma":[0.99942327,0.00039540825,0.000018380319,0.000070455404,0.00007249873,0.000019955129],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00075368996,0.0005295163,0.00041185456,0.00056269864,0.000734547,0.0014559396,0.0009049381,0.0007077074,0.008644782],"category_scores_gemma":[0.0031179623,0.0004155116,0.000561509,0.00067113515,0.00039762454,0.0020529365,0.0012090275,0.00089369423,0.0019354019],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014874554,0.00029247688,0.022888338,0.0007749447,0.00038218638,0.0014373593,0.004813419,0.3742349,0.007741345,0.07288107,0.117808565,0.3952579],"study_design_scores_gemma":[0.00009535162,0.000036983776,0.004994124,0.000091231916,0.000085192616,0.00016892656,0.0015115234,0.91398966,0.0024676754,0.03121584,0.045302317,0.00004112984],"about_ca_topic_score_codex":0.08771104,"about_ca_topic_score_gemma":0.13380226,"teacher_disagreement_score":0.08771104,"about_ca_system_score_codex":0.0010541551,"about_ca_system_score_gemma":0.0010226878,"threshold_uncertainty_score":0.17440099},"labels":[],"label_agreement":null},{"id":"W4404783043","doi":"10.18653/v1/2022.findings-aacl.22","title":"A Copy Mechanism for Handling Knowledge Base Elements in SPARQL Neural Machine Translation","year":2022,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"SPARQL; Computer science; Mechanism (biology); Machine translation; Knowledge base; Translation (biology); Artificial intelligence; Base (topology); Natural language processing; RDF; Semantic Web","score_opus":0.030341845944013603,"score_gpt":0.3053360348926012,"score_spread":0.27499418894858757,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404783043","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07318015,0.00058331614,0.8873952,0.0007296426,0.00013688192,0.00025365286,0.0009801089,0.030509021,0.006232051],"genre_scores_gemma":[0.64853716,0.00035004222,0.33920166,0.0006454504,0.000080915044,0.00033476425,0.002241065,0.0010142911,0.007594552],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99931514,0.00015966175,0.000086688786,0.00018725079,0.00018840564,0.000062724874],"domain_scores_gemma":[0.99772054,0.00064982614,0.000099583056,0.0011702823,0.00031214984,0.00004765701],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001510907,0.00060983165,0.0005951163,0.00075397914,0.000489873,0.0012898091,0.0026596386,0.0008682416,0.007376795],"category_scores_gemma":[0.00541784,0.00046873494,0.0006485642,0.00087397994,0.0011677253,0.0049433187,0.00256569,0.0014706623,0.0013307102],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006502896,0.00027763247,0.0035876574,0.0004774453,0.00013908319,0.0006080468,0.0011021231,0.073557325,0.046675064,0.07170736,0.028380971,0.77283704],"study_design_scores_gemma":[0.00012466116,0.00020797312,0.0008744563,0.000056732002,0.00012223427,0.00042157533,0.00024722773,0.7974399,0.09628875,0.07564729,0.028507125,0.000062214334],"about_ca_topic_score_codex":0.0065388875,"about_ca_topic_score_gemma":0.00822241,"teacher_disagreement_score":0.007376795,"about_ca_system_score_codex":0.0011450818,"about_ca_system_score_gemma":0.0013394954,"threshold_uncertainty_score":0.024677813},"labels":[],"label_agreement":null},{"id":"W4404783453","doi":"10.18653/v1/2024.emnlp-main.294","title":"Beyond Reference: Evaluating High Quality Translations Better than Human References","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"Institute for Information and Communications Technology Promotion; Ministry of Science and ICT, South Korea; Hanyang University","keywords":"Computer science; Quality (philosophy)","score_opus":0.09457329550499463,"score_gpt":0.41613552125445163,"score_spread":0.32156222574945703,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404783453","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6565755,0.016345134,0.2848978,0.000767251,0.000841756,0.0014330364,0.0047612065,0.00872114,0.025657196],"genre_scores_gemma":[0.8590074,0.0010325743,0.12846349,0.00037581238,0.00025353237,0.0004110953,0.005490112,0.0010307827,0.0039351666],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9834366,0.008458544,0.0016632027,0.00247955,0.003675452,0.00028665425],"domain_scores_gemma":[0.9619023,0.022380995,0.0033216528,0.0033769002,0.008065453,0.0009525765],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01244183,0.0020070826,0.001505727,0.0052858423,0.0009511839,0.0029964051,0.0013495146,0.0027505883,0.005184071],"category_scores_gemma":[0.055192836,0.00032710837,0.0007905761,0.003609765,0.0013020259,0.004516361,0.002151936,0.0007996605,0.002368082],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.005742533,0.0010536385,0.051814865,0.005270316,0.0018323336,0.0013808768,0.0041872477,0.036701452,0.09492311,0.007100932,0.01939841,0.7705943],"study_design_scores_gemma":[0.0016565728,0.016347168,0.18419243,0.0011792892,0.0035305908,0.0062029874,0.0052535236,0.4604274,0.20149982,0.039905764,0.0785658,0.0012386562],"about_ca_topic_score_codex":0.0024368172,"about_ca_topic_score_gemma":0.0046827784,"teacher_disagreement_score":0.01244183,"about_ca_system_score_codex":0.00083890575,"about_ca_system_score_gemma":0.0010428076,"threshold_uncertainty_score":0.065799475},"labels":[],"label_agreement":null},{"id":"W4404783575","doi":"10.18653/v1/2024.emnlp-main.176","title":"Zero-shot Cross-Lingual Transfer for Synthetic Data Generation in Grammatical Error Detection","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ubisoft (Canada)","funders":"","keywords":"Computer science; Zero (linguistics); Artificial intelligence; Transfer (computing); Shot (pellet); Natural language processing; Materials science; Linguistics","score_opus":0.08761482541301252,"score_gpt":0.37736527858875474,"score_spread":0.2897504531757422,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404783575","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.35769156,0.0007301037,0.6088636,0.00056740333,0.0004530235,0.0005254022,0.002710553,0.022967275,0.005491136],"genre_scores_gemma":[0.7312777,0.00014215433,0.25340325,0.00045981552,0.000055910565,0.0006024301,0.008948738,0.00166789,0.0034421468],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9944951,0.0029043288,0.00025589296,0.0013150232,0.0008095591,0.00022007429],"domain_scores_gemma":[0.9799789,0.012066902,0.00070391584,0.004412235,0.0025310188,0.00030698214],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006643828,0.0013847795,0.0008414244,0.0015128022,0.0008156679,0.0010447343,0.0024886262,0.0018366405,0.0025201726],"category_scores_gemma":[0.026715538,0.00054915773,0.0008135259,0.0010809597,0.0014220536,0.0020463704,0.003721606,0.0020946197,0.0016958937],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009682375,0.0011785254,0.021124698,0.0010386276,0.00042526037,0.0012469023,0.002158422,0.35921562,0.050244045,0.008303689,0.02526833,0.52882767],"study_design_scores_gemma":[0.00010192725,0.00045087046,0.004866909,0.00007193877,0.000057866146,0.0006919815,0.0005556954,0.9130561,0.060344994,0.010340829,0.009342464,0.00011841517],"about_ca_topic_score_codex":0.0035045375,"about_ca_topic_score_gemma":0.007051395,"teacher_disagreement_score":0.006643828,"about_ca_system_score_codex":0.0008319126,"about_ca_system_score_gemma":0.0013185511,"threshold_uncertainty_score":0.035136342},"labels":[],"label_agreement":null},{"id":"W4404791709","doi":"10.18653/v1/2022.aacl-demo.1","title":"VScript: Controllable Script Generation with Visual Presentation","year":2022,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Computer science; Presentation (obstetrics); Computer graphics (images); Artificial intelligence; Programming language","score_opus":0.013680847239153915,"score_gpt":0.2683348793399821,"score_spread":0.2546540321008282,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404791709","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0041820863,0.00017554684,0.7245986,0.00018112155,0.00054988655,0.00027861234,0.003179975,0.25452906,0.012325206],"genre_scores_gemma":[0.18367556,0.0005723828,0.6548564,0.000579848,0.00023905083,0.0017620571,0.020765195,0.08645977,0.051089715],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992022,0.00016311143,0.00006885836,0.00016792602,0.000306626,0.00009131843],"domain_scores_gemma":[0.998494,0.0005980711,0.00006300798,0.00042114966,0.0002978659,0.00012601097],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001274299,0.0019385326,0.0007408166,0.0009221319,0.00049471075,0.001876514,0.0034529262,0.0012154012,0.07602561],"category_scores_gemma":[0.0044498583,0.00097077485,0.0012002181,0.0005764779,0.00064832566,0.0021565321,0.0029827014,0.0013952499,0.025618082],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017600332,0.0004044468,0.0010539856,0.0010964901,0.00014256794,0.0011364897,0.0007018125,0.014699589,0.0777136,0.044222623,0.39220464,0.4648636],"study_design_scores_gemma":[0.000850082,0.00041105167,0.0007799899,0.00024751504,0.00011607653,0.0008963645,0.00022382927,0.35457352,0.1895762,0.056498192,0.39559403,0.00023324355],"about_ca_topic_score_codex":0.00088382995,"about_ca_topic_score_gemma":0.00093735167,"teacher_disagreement_score":0.07602561,"about_ca_system_score_codex":0.00042008015,"about_ca_system_score_gemma":0.0006122671,"threshold_uncertainty_score":0.254331},"labels":[],"label_agreement":null},{"id":"W4404792900","doi":"10.18653/v1/2024.mrl-1.2","title":"What an Elegant Bridge: Multilingual LLMs are Biased Similarly in Different Languages","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"Engineering and Physical Sciences Research Council","keywords":"Bridge (graph theory); Computer science; Linguistics; Philosophy; Medicine","score_opus":0.025035495069478882,"score_gpt":0.3316981056926324,"score_spread":0.30666261062315353,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404792900","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8392336,0.0014143547,0.10662487,0.015366378,0.00050024793,0.00006089509,0.00095441536,0.0008845079,0.034960732],"genre_scores_gemma":[0.98833174,0.00015468578,0.008092729,0.0016566891,0.00010434761,0.000025196025,0.00018048126,0.00025774314,0.0011963707],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9966072,0.0013843223,0.00015605507,0.0010865644,0.00050316553,0.00026263585],"domain_scores_gemma":[0.9857735,0.006759947,0.0017470617,0.0039596586,0.0011361302,0.00062375586],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006338582,0.00047678486,0.000688489,0.0008964378,0.0007676705,0.0035953359,0.00080935436,0.0010191717,0.0049597616],"category_scores_gemma":[0.045886435,0.00036704363,0.00052531285,0.00057758664,0.0030539553,0.008720725,0.0031229372,0.0018877103,0.0012002359],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017053046,0.00026172638,0.31403998,0.0006031058,0.0016773636,0.0006290831,0.035825748,0.007869855,0.048054706,0.24026185,0.0151950335,0.33387628],"study_design_scores_gemma":[0.000112567745,0.0003539186,0.119874835,0.00031017998,0.00042655895,0.001183068,0.01621232,0.028236084,0.020186136,0.7795327,0.033204,0.00036768836],"about_ca_topic_score_codex":0.002393961,"about_ca_topic_score_gemma":0.0024317894,"teacher_disagreement_score":0.006338582,"about_ca_system_score_codex":0.00064818724,"about_ca_system_score_gemma":0.00050912786,"threshold_uncertainty_score":0.03352201},"labels":[],"label_agreement":null},{"id":"W4404792915","doi":"10.18653/v1/2024.emnlp-main.695","title":"The Mystery of the Pathological Path-star Task for Language Models","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute","funders":"Strong; Government of Canada; Canadian Institute for Advanced Research","keywords":"Task (project management); Computer science; Path (computing); Star (game theory); Artificial intelligence; Natural language processing; Programming language; Engineering; Astrophysics; Physics","score_opus":0.015625217549799206,"score_gpt":0.2726373696601032,"score_spread":0.257012152110304,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404792915","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.032766372,0.0011843052,0.93359077,0.02155895,0.00035364527,0.000076166354,0.00072765653,0.0020424784,0.0076996624],"genre_scores_gemma":[0.64718044,0.0011003413,0.33055118,0.0049283504,0.0011625313,0.000495748,0.0020784878,0.0015116275,0.010991348],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9957942,0.0024962653,0.00017237969,0.0009614667,0.000385688,0.00019003113],"domain_scores_gemma":[0.970829,0.022877643,0.000675363,0.0041624466,0.0007749473,0.000680622],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007482992,0.0012827214,0.0018861478,0.000522353,0.0015293624,0.002726962,0.0033255527,0.004195142,0.006319315],"category_scores_gemma":[0.040517617,0.0010666095,0.0014487851,0.0006598862,0.0046003223,0.013602807,0.0035783825,0.011400768,0.0026251401],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006509951,0.00030820447,0.0019793771,0.00079529855,0.0002297721,0.00037522882,0.0012888697,0.15120572,0.0035110773,0.6279968,0.06208297,0.14957568],"study_design_scores_gemma":[0.000056383014,0.000038613325,0.0001375355,0.000029041987,0.0000105697345,0.00007606552,0.0000594748,0.21510766,0.0005374336,0.78097993,0.0029462192,0.000020984595],"about_ca_topic_score_codex":0.0033473645,"about_ca_topic_score_gemma":0.0044022608,"teacher_disagreement_score":0.007482992,"about_ca_system_score_codex":0.0014066497,"about_ca_system_score_gemma":0.0022392867,"threshold_uncertainty_score":0.039574325},"labels":[],"label_agreement":null},{"id":"W4404840841","doi":"10.4000/12spd","title":"Assessing human translation style with the help of NMT: a case study of French-language comic book translators","year":2024,"lang":"en","type":"article","venue":"Palimpsestes","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ministère de l’Emploi et de la Solidarité Sociale (Québec)","funders":"","keywords":"Comics; Style (visual arts); Translation (biology); Linguistics; Computer science; Natural language processing; Art; Literature; Philosophy; Chemistry","score_opus":0.023175509698278616,"score_gpt":0.32198046073950143,"score_spread":0.2988049510412228,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404840841","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98830533,0.00021867451,0.006581925,0.0002379914,0.000019722887,0.00014840746,0.00027374257,0.00011625972,0.004098009],"genre_scores_gemma":[0.98509,0.00016845837,0.011929716,0.00012360227,0.000024738467,0.000092763825,0.00041575267,0.000078038276,0.0020768235],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9925095,0.005591081,0.0003789544,0.00060959795,0.0007289391,0.00018191153],"domain_scores_gemma":[0.9711258,0.019441461,0.0019473479,0.0023535714,0.004633962,0.00049795205],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005757557,0.00062233873,0.000533846,0.0016971139,0.0016607232,0.0011632806,0.00073453854,0.0012776714,0.0019053166],"category_scores_gemma":[0.028490806,0.00023736402,0.00033912717,0.0022703467,0.0011376749,0.0011653934,0.0010350903,0.00062432804,0.0010156406],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020200596,0.002818323,0.21687908,0.0018112016,0.0004092172,0.014238635,0.22261469,0.011187912,0.03654172,0.0023307644,0.011492695,0.47765574],"study_design_scores_gemma":[0.00065518875,0.006569504,0.5163008,0.000625461,0.000591151,0.025455635,0.17072836,0.100761086,0.0906953,0.008323466,0.07874087,0.00055320945],"about_ca_topic_score_codex":0.006324917,"about_ca_topic_score_gemma":0.015641894,"teacher_disagreement_score":0.006324917,"about_ca_system_score_codex":0.0009328288,"about_ca_system_score_gemma":0.00069578364,"threshold_uncertainty_score":0.030449212},"labels":[],"label_agreement":null},{"id":"W4404883323","doi":"10.47810/clib.24.20","title":"Lexical Richness of French and Quebec Journalistic Texts","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Linguistics; Natural language processing; History; Artificial intelligence; Philosophy","score_opus":0.009868015370580152,"score_gpt":0.27934887595804764,"score_spread":0.2694808605874675,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404883323","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9923921,0.0007215403,0.00015647232,0.00018191726,0.000006554604,0.00001373179,0.0013211359,0.000018903174,0.005187674],"genre_scores_gemma":[0.9974208,0.00022304406,0.00023333724,0.000031920485,0.000019878107,0.00002316666,0.0009990357,0.00001117731,0.0010376113],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9978789,0.00075111864,0.00017019843,0.0003471275,0.00064496836,0.00020763451],"domain_scores_gemma":[0.989599,0.0062699933,0.0015218542,0.00019368125,0.0019734462,0.00044215302],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013941262,0.00035054836,0.00048153865,0.019206515,0.002482279,0.0032887235,0.00034481866,0.00046852857,0.0037894154],"category_scores_gemma":[0.0111527825,0.00014748561,0.0002684721,0.009648682,0.0016326769,0.0012677611,0.0010479535,0.00028909443,0.00022368642],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017155872,0.00019129041,0.60404694,0.0013156648,0.0004837059,0.002664492,0.23594536,0.00095017446,0.035171784,0.005494931,0.0049490687,0.10707102],"study_design_scores_gemma":[0.000014036585,0.00009117744,0.94595426,0.00007568448,0.00004600997,0.00092413276,0.03709878,0.0005785731,0.0011127106,0.00020784223,0.0138521325,0.000044666063],"about_ca_topic_score_codex":0.20665607,"about_ca_topic_score_gemma":0.24519135,"teacher_disagreement_score":0.7933439,"about_ca_system_score_codex":0.004695305,"about_ca_system_score_gemma":0.0010757609,"threshold_uncertainty_score":0.41090637},"labels":[],"label_agreement":null},{"id":"W4405077419","doi":"10.1075/scl.118.09mil","title":"Modeling fine-grained sociolinguistic variation","year":2024,"lang":"en","type":"book-chapter","venue":"Studies in corpus linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Variation (astronomy); Computer science; Linguistics; Astrophysics; Physics; Philosophy","score_opus":0.05598935463493161,"score_gpt":0.3361349588186589,"score_spread":0.2801456041837273,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405077419","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.75732505,0.0004555604,0.22903244,0.0005382867,0.000045098415,0.0000856229,0.0037187503,0.000646351,0.008152733],"genre_scores_gemma":[0.9693927,0.00008760555,0.02744446,0.000045462642,0.000010073519,0.000058318154,0.001587796,0.00009273202,0.0012807489],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999673,0.00012953972,0.000008943383,0.00009521322,0.000058544312,0.00003484916],"domain_scores_gemma":[0.9977036,0.0015995294,0.00017194256,0.0002308325,0.00023889124,0.000055133314],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007324419,0.00028983486,0.00026752337,0.0011016484,0.000492536,0.001177925,0.0005789989,0.00032663002,0.0023319698],"category_scores_gemma":[0.004581069,0.00021722006,0.0002568989,0.001230131,0.0009451016,0.0010101589,0.00068837084,0.0007475683,0.0003378739],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00027826885,0.0002337101,0.15381843,0.00053955143,0.00025903023,0.0009920141,0.007373995,0.50498366,0.042155694,0.11189312,0.011900185,0.16557235],"study_design_scores_gemma":[0.0000067716896,0.000022232227,0.050261576,0.000027258098,0.000012805292,0.000047298658,0.00078106456,0.91679496,0.0016198911,0.023725245,0.0066731516,0.00002776742],"about_ca_topic_score_codex":0.15589188,"about_ca_topic_score_gemma":0.2614704,"teacher_disagreement_score":0.15589188,"about_ca_system_score_codex":0.0019029025,"about_ca_system_score_gemma":0.0011891887,"threshold_uncertainty_score":0.309969},"labels":[],"label_agreement":null},{"id":"W4405168385","doi":"10.5753/stil.2024.31170","title":"LLM-SEMREL: Towards a Better Coreference Resolution for Portuguese","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Blutip (Canada)","funders":"","keywords":"Coreference; Portuguese; Computer science; Resolution (logic); Natural language processing; Artificial intelligence; Linguistics; Philosophy","score_opus":0.027608094266220472,"score_gpt":0.30722246624909727,"score_spread":0.2796143719828768,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405168385","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.028134711,0.0041038385,0.86917806,0.0023093561,0.0006142138,0.000632714,0.041731197,0.029592426,0.023703383],"genre_scores_gemma":[0.18397434,0.0038541541,0.7153931,0.0011977294,0.00022119368,0.0007806717,0.08269485,0.0050379327,0.0068460754],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9958295,0.001450269,0.00060391956,0.0009682111,0.00096428144,0.00018377307],"domain_scores_gemma":[0.9970341,0.0010652771,0.00031699875,0.0009869552,0.0004695709,0.00012705583],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0042894236,0.0016294321,0.0016019798,0.0056970464,0.001611305,0.0048783473,0.002254729,0.0016318852,0.0076101427],"category_scores_gemma":[0.01011505,0.0013498861,0.002401468,0.0036681395,0.0008429234,0.007483839,0.0037236414,0.0016642768,0.0053975405],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014569242,0.00057361624,0.007105936,0.0058091027,0.00048567177,0.0029504898,0.004762737,0.033279505,0.021834113,0.15780011,0.11260494,0.65133685],"study_design_scores_gemma":[0.00022225172,0.00022482095,0.0038089277,0.0011702749,0.00038951932,0.0019249193,0.0017658857,0.13993756,0.02010776,0.07180935,0.7583661,0.00027277117],"about_ca_topic_score_codex":0.0048774956,"about_ca_topic_score_gemma":0.007696984,"teacher_disagreement_score":0.0076101427,"about_ca_system_score_codex":0.001404635,"about_ca_system_score_gemma":0.0031448563,"threshold_uncertainty_score":0.025458455},"labels":[],"label_agreement":null},{"id":"W4405250561","doi":"10.32920/27997979","title":"Assessing the Performance of Automatic Speech Recognition Systems When Used by Native and Non-Native Speakers of Three Major Languages in Dictation Workflows","year":2024,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"European Commission; Copenhagen Business School","keywords":"Dictation; Computer science; Transcription (linguistics); Speech recognition; Workflow; Natural language processing; First language; Competence (human resources); Artificial intelligence; Linguistics; Psychology","score_opus":0.021145692516949997,"score_gpt":0.30139396009551644,"score_spread":0.28024826757856647,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405250561","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99377114,0.00008226504,0.0042711017,0.000051508487,0.000036910376,0.00017000371,0.00017995157,0.00017451787,0.0012626142],"genre_scores_gemma":[0.9764924,0.00011137988,0.019268047,0.00013983737,0.000054201835,0.00036228015,0.00094275485,0.00010194577,0.0025271159],"study_design_codex":"bench_or_experimental","study_design_gemma":"observational","domain_scores_codex":[0.9924684,0.0038253558,0.0009664291,0.0014804814,0.00087748,0.0003816992],"domain_scores_gemma":[0.960741,0.028727017,0.0015192119,0.0029796073,0.0050158296,0.0010174185],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0077255815,0.00096562755,0.0008721872,0.0006525609,0.00062409247,0.0017691338,0.0007825736,0.0012735218,0.0030869232],"category_scores_gemma":[0.03339739,0.0004627562,0.0006558723,0.00036905377,0.00086778007,0.0016060561,0.0015784957,0.0007420626,0.0027638439],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.017974863,0.004168434,0.14998591,0.0020539195,0.000851674,0.0012933238,0.03351869,0.011316906,0.43355072,0.00080134394,0.0027513236,0.34173286],"study_design_scores_gemma":[0.0015515554,0.041528348,0.511016,0.00020792698,0.0010001461,0.003552289,0.023240022,0.062298626,0.3417885,0.00136899,0.011641458,0.00080618006],"about_ca_topic_score_codex":0.0021720396,"about_ca_topic_score_gemma":0.0028240078,"teacher_disagreement_score":0.0077255815,"about_ca_system_score_codex":0.000449897,"about_ca_system_score_gemma":0.00067726587,"threshold_uncertainty_score":0.040857255},"labels":[],"label_agreement":null},{"id":"W4405304805","doi":"10.1109/besc64747.2024.10780612","title":"Enhancing Academic Title Drafting Through Abstractive Summarization","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Automatic summarization; Computer science; Natural language processing; Information retrieval; World Wide Web; Artificial intelligence","score_opus":0.01538067783087982,"score_gpt":0.31415540771153183,"score_spread":0.298774729880652,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405304805","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.043449547,0.002857648,0.9305898,0.0009431268,0.00066456565,0.0005577931,0.0015667216,0.013616685,0.0057541565],"genre_scores_gemma":[0.3687769,0.002951659,0.60542756,0.00040916057,0.00068969204,0.00043687475,0.008235109,0.00083787966,0.012235239],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.998262,0.000577146,0.0002207274,0.00034154978,0.0005189437,0.00007959983],"domain_scores_gemma":[0.99288154,0.0027686541,0.0009459316,0.0008005856,0.0023201997,0.00028307008],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034562834,0.0010885197,0.00084408926,0.0026881464,0.00057303056,0.0023046504,0.0014864652,0.0009361155,0.0037879974],"category_scores_gemma":[0.013875349,0.00042800154,0.001013235,0.0019445922,0.00046271936,0.0029572195,0.0012445932,0.00117166,0.0037118278],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036514172,0.00018770565,0.0034668637,0.0013017965,0.00011599952,0.0004048299,0.0008781162,0.049736936,0.04362517,0.0062332554,0.021509664,0.87217444],"study_design_scores_gemma":[0.00018427834,0.0007919548,0.00330228,0.00018077373,0.00036050854,0.00055594946,0.00044726612,0.8117011,0.10477733,0.014057133,0.06351062,0.00013082803],"about_ca_topic_score_codex":0.0019256336,"about_ca_topic_score_gemma":0.0025839193,"teacher_disagreement_score":0.0037879974,"about_ca_system_score_codex":0.0007430785,"about_ca_system_score_gemma":0.0020314741,"threshold_uncertainty_score":0.018278778},"labels":[],"label_agreement":null},{"id":"W4405491040","doi":"10.1109/me61309.2024.10789747","title":"Human 0, MLLM 1: Unlocking New Layers of Automation in Language-Conditioned Robotics with Multimodal LLMs","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Automation; Robotics; Artificial intelligence; Computer science; Natural language processing; Engineering; Robot; Mechanical engineering","score_opus":0.009887768186568531,"score_gpt":0.28555409611708116,"score_spread":0.2756663279305126,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405491040","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.058320038,0.00024671515,0.91223806,0.00054439233,0.00012217922,0.00018499942,0.00019171914,0.020852221,0.0072997645],"genre_scores_gemma":[0.5767055,0.00011449856,0.41631585,0.0005528761,0.00003949089,0.00025150497,0.0003654319,0.0008702674,0.004784655],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989114,0.0004194938,0.000044442564,0.00025848477,0.0002632596,0.00010298261],"domain_scores_gemma":[0.9982054,0.00089220476,0.0001328847,0.0004262294,0.00021393441,0.00012930554],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015432186,0.0010749891,0.00045690368,0.0003443612,0.00038503116,0.0013002426,0.0014349096,0.001053228,0.005909111],"category_scores_gemma":[0.0055563943,0.0003778603,0.0006033823,0.00017348623,0.0017846624,0.0030071393,0.0031147832,0.0018000279,0.001757254],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016178584,0.0006957812,0.0036343804,0.00088596315,0.0001307344,0.00057563395,0.0019492742,0.16734563,0.1687182,0.034619816,0.013559095,0.60626763],"study_design_scores_gemma":[0.00007618691,0.00074551685,0.0009238815,0.00007671821,0.000042877047,0.00021346941,0.00029284356,0.8980533,0.058920793,0.027284,0.013248629,0.000121886085],"about_ca_topic_score_codex":0.0025169523,"about_ca_topic_score_gemma":0.0038287132,"teacher_disagreement_score":0.005909111,"about_ca_system_score_codex":0.0006738324,"about_ca_system_score_gemma":0.0012459584,"threshold_uncertainty_score":0.01976794},"labels":[],"label_agreement":null},{"id":"W4405502038","doi":"10.2139/ssrn.5061381","title":"Beyond Vanilla Fine-Tuning: Leveraging Multistage, Multilingual, and Domain-Specific Methods for Low-Resource Machine Translation","year":2024,"lang":"en","type":"preprint","venue":"SSRN Electronic Journal","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Machine translation; Fine-tuning; Translation (biology); Resource (disambiguation); Domain (mathematical analysis); Artificial intelligence; Mathematics; Chemistry; Particle physics; Physics; Computer network","score_opus":0.020223359113123607,"score_gpt":0.3356685133325591,"score_spread":0.3154451542194355,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405502038","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023120284,0.00173854,0.9560596,0.00043634302,0.00025697594,0.00010632789,0.00024877596,0.012304992,0.0057282685],"genre_scores_gemma":[0.38267934,0.0008143943,0.60198104,0.000786002,0.0002937697,0.00018035874,0.0015660204,0.0029768148,0.00872222],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9973322,0.001184826,0.00016469073,0.0006429417,0.00043185835,0.00024349763],"domain_scores_gemma":[0.99593866,0.0023197597,0.00013977892,0.0010667384,0.0004119068,0.0001232705],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026521122,0.0015453642,0.002010724,0.0012692085,0.0011494018,0.0027503674,0.0023567507,0.0019204628,0.007772971],"category_scores_gemma":[0.008998176,0.00080647203,0.0012536156,0.0019549364,0.0010803858,0.0040308735,0.003468984,0.0028156156,0.0051275976],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000915026,0.00052475434,0.0012295864,0.00068832206,0.0003485792,0.00041800234,0.00037216564,0.08732976,0.04679329,0.020419838,0.014967356,0.8259933],"study_design_scores_gemma":[0.00009335285,0.00016167536,0.00041322826,0.00006923515,0.00012202314,0.00028916541,0.00011279044,0.89966,0.021807866,0.06573874,0.011455598,0.000076326345],"about_ca_topic_score_codex":0.0024809027,"about_ca_topic_score_gemma":0.008615821,"teacher_disagreement_score":0.007772971,"about_ca_system_score_codex":0.0005269899,"about_ca_system_score_gemma":0.0019406972,"threshold_uncertainty_score":0.026003242},"labels":[],"label_agreement":null},{"id":"W4405514180","doi":"10.3390/app142411790","title":"A Survey of Grapheme-to-Phoneme Conversion Methods","year":2024,"lang":"en","type":"article","venue":"Applied Sciences","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Fundamental Research Funds for the Central Universities","keywords":"Grapheme; Computer science; Natural language processing; Physics","score_opus":0.03871143839641628,"score_gpt":0.3668908844547667,"score_spread":0.3281794460583504,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405514180","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022388292,0.16718037,0.7541418,0.001442646,0.0016009378,0.00032023215,0.0037232137,0.010290219,0.03891234],"genre_scores_gemma":[0.1672178,0.17859004,0.5829332,0.0018943882,0.001294185,0.0005816189,0.019341934,0.0026550395,0.045491777],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9992563,0.00011896805,0.00008504428,0.00023660592,0.0002524238,0.000050666258],"domain_scores_gemma":[0.99915254,0.00037920236,0.0000430289,0.00014313862,0.0002527777,0.000029380039],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008112113,0.0014714593,0.0009372646,0.0026962082,0.00049635617,0.0012953386,0.0017571971,0.00088809937,0.00702421],"category_scores_gemma":[0.0025301618,0.00046809798,0.0010630393,0.0038072767,0.00043098372,0.0024433997,0.0008858023,0.001349895,0.004768295],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009821945,0.000063336,0.0006862694,0.0013266696,0.00006797133,0.00007893072,0.000085464766,0.0057362765,0.003985201,0.0042516943,0.014682823,0.9689371],"study_design_scores_gemma":[0.00007498763,0.00042477428,0.0070373924,0.0015766191,0.00040071987,0.0022104026,0.00059221656,0.25431427,0.064928964,0.033267967,0.6349242,0.00024754024],"about_ca_topic_score_codex":0.0043298653,"about_ca_topic_score_gemma":0.003967767,"teacher_disagreement_score":0.00702421,"about_ca_system_score_codex":0.00050368835,"about_ca_system_score_gemma":0.0011795571,"threshold_uncertainty_score":0.023498297},"labels":[],"label_agreement":null},{"id":"W4405601877","doi":"10.1109/scam63643.2024.00017","title":"Enhancing Identifier Naming Through Multi-Mask Fine-Tuning of Language Models of Code","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Identifier; Computer science; Code (set theory); Programming language; Language model; Natural language processing; Artificial intelligence","score_opus":0.025179247348711808,"score_gpt":0.31560278598394403,"score_spread":0.29042353863523224,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405601877","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.099030845,0.00041579636,0.86073023,0.00053478644,0.00021919934,0.0003059558,0.0006331005,0.035517108,0.0026130034],"genre_scores_gemma":[0.49826163,0.00020819287,0.49154243,0.0005198723,0.000055178778,0.0004101518,0.00217364,0.0034871318,0.0033418185],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99621224,0.0015042934,0.00032813026,0.0010829204,0.00070881075,0.00016352395],"domain_scores_gemma":[0.989087,0.0053998907,0.0011597584,0.0027505688,0.0013948858,0.00020788843],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004033016,0.0012741345,0.0007077755,0.0012261278,0.00053028174,0.0017889029,0.0020661445,0.0011935328,0.0024012339],"category_scores_gemma":[0.024129014,0.0005230134,0.0011631193,0.000733025,0.0009388077,0.0051687974,0.002637065,0.0018191707,0.0016948864],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010300747,0.0005557927,0.02105269,0.0010520151,0.00019652027,0.00080900994,0.0030877744,0.10505499,0.13182688,0.022665424,0.019285675,0.69338316],"study_design_scores_gemma":[0.00006344634,0.0002100328,0.0020554636,0.000055687557,0.00010173798,0.0005553752,0.00029361589,0.9029576,0.06422557,0.012436452,0.016957171,0.00008793108],"about_ca_topic_score_codex":0.0024021892,"about_ca_topic_score_gemma":0.0038456244,"teacher_disagreement_score":0.004033016,"about_ca_system_score_codex":0.001160056,"about_ca_system_score_gemma":0.0019402872,"threshold_uncertainty_score":0.021328866},"labels":[],"label_agreement":null},{"id":"W4405633760","doi":"10.1109/mlnlp63328.2024.10800442","title":"Improving Chinese Punctuation Restoration via External POS Tagger","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"The Scarborough Hospital; University of Toronto","funders":"","keywords":"Punctuation; Computer science; Natural language processing; Artificial intelligence; Speech recognition","score_opus":0.006599269837192123,"score_gpt":0.2681952532649079,"score_spread":0.2615959834277158,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405633760","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.30192244,0.0028201614,0.62128675,0.0006601863,0.0014579853,0.0005170937,0.007191541,0.055803556,0.008340309],"genre_scores_gemma":[0.49233833,0.0011871475,0.4634812,0.0005791718,0.00025015304,0.00030702783,0.02797938,0.0018286454,0.012048937],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.998317,0.00033723988,0.00022694179,0.0005746966,0.00037997117,0.00016410342],"domain_scores_gemma":[0.99695,0.0005579636,0.00026433085,0.0011593538,0.00093553134,0.00013290755],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019752213,0.0023094115,0.00157851,0.0029205075,0.001090377,0.0011119748,0.0014593634,0.0008748606,0.0030452095],"category_scores_gemma":[0.0057034595,0.00036996667,0.0014064133,0.0024041894,0.00077645754,0.00233761,0.0018573498,0.0013210475,0.0067570154],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00094968494,0.00037250912,0.008979482,0.00075849256,0.0001792417,0.0011788001,0.0006076607,0.023373535,0.1033502,0.002479389,0.035448905,0.8223222],"study_design_scores_gemma":[0.00022318146,0.0007993921,0.016532335,0.00014393855,0.0005214877,0.0016817838,0.0011715717,0.5322854,0.36827105,0.009135841,0.068971895,0.00026209472],"about_ca_topic_score_codex":0.005631352,"about_ca_topic_score_gemma":0.010098934,"teacher_disagreement_score":0.005631352,"about_ca_system_score_codex":0.00065190945,"about_ca_system_score_gemma":0.0025676726,"threshold_uncertainty_score":0.01119715},"labels":[],"label_agreement":null},{"id":"W4405653074","doi":"10.5755/j01.sal.1.45.38448","title":"Light verb constructions with deverbal nouns BITE and SNACK in native English varieties: a corpus-based study","year":2024,"lang":"en","type":"article","venue":"Studies About Languages","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Noun; Linguistics; Verb; Mathematics; Psychology; Natural language processing; Artificial intelligence; Computer science; Philosophy","score_opus":0.008797899029700997,"score_gpt":0.2900483563078339,"score_spread":0.28125045727813286,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405653074","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99050856,0.0009039735,0.00048759082,0.00007595693,0.000014008821,0.00005273097,0.00050777156,0.00000789265,0.0074415375],"genre_scores_gemma":[0.99517703,0.00088495726,0.0012451241,0.00007772829,0.000014263865,0.00011396904,0.0010760303,0.000034368,0.0013765245],"study_design_codex":"qualitative","study_design_gemma":"observational","domain_scores_codex":[0.9989311,0.00029356967,0.00012356107,0.00031143936,0.00024570827,0.000094649055],"domain_scores_gemma":[0.9957522,0.0025772222,0.0006159589,0.0002775419,0.00060270814,0.00017440646],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012292683,0.00038924487,0.0006235484,0.0028194352,0.0021793584,0.0019668702,0.0005361156,0.0004714199,0.0032683332],"category_scores_gemma":[0.0044472963,0.00040623563,0.0002670458,0.0032116903,0.00269025,0.00196967,0.0018376234,0.0008370381,0.00045386032],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00031346278,0.00031371412,0.19970445,0.0022603257,0.00008772809,0.0059815887,0.7016689,0.00015514174,0.021906385,0.006520742,0.0029370438,0.05815052],"study_design_scores_gemma":[0.000043186636,0.00012044067,0.6464555,0.000539634,0.000113701906,0.0075146174,0.28039688,0.00045087078,0.002440506,0.0009851949,0.060849044,0.00009038009],"about_ca_topic_score_codex":0.051275436,"about_ca_topic_score_gemma":0.13646601,"teacher_disagreement_score":0.051275436,"about_ca_system_score_codex":0.0014883773,"about_ca_system_score_gemma":0.0012370163,"threshold_uncertainty_score":0.10195392},"labels":[],"label_agreement":null},{"id":"W4405674993","doi":"10.24908/pceea.2024.18585","title":"Uses of Word Embeddings in Engineering Education","year":2024,"lang":"en","type":"article","venue":"Proceedings of the Canadian Engineering Education Association (CEEA)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Word (group theory); Computer science; Natural language processing; Linguistics; Philosophy","score_opus":0.00421729834196276,"score_gpt":0.2247444645672268,"score_spread":0.22052716622526403,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405674993","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06359327,0.034750126,0.8128423,0.014363988,0.001040304,0.00033878762,0.0013346677,0.00205839,0.06967825],"genre_scores_gemma":[0.44734007,0.020493804,0.51738787,0.00088439265,0.00030864758,0.00048312364,0.0012908767,0.00045813434,0.011353082],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.993324,0.0044907443,0.00055262685,0.0006069302,0.00092600804,0.00009975321],"domain_scores_gemma":[0.96881133,0.025528846,0.0015952368,0.0017220427,0.0020826068,0.00025989237],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005220297,0.0008934422,0.0005185849,0.0061195465,0.00077845127,0.0065312064,0.0007725286,0.0011836314,0.006739861],"category_scores_gemma":[0.03393962,0.0004660011,0.00057865295,0.007176226,0.0028140643,0.012751591,0.0028431963,0.0015442757,0.0015569311],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000059878756,0.000085204265,0.0030720904,0.0017821504,0.00006250828,0.00020253805,0.008510945,0.003018611,0.0018572318,0.15712485,0.004855332,0.8193688],"study_design_scores_gemma":[0.00003858529,0.00020936676,0.006256868,0.00410086,0.00011202909,0.0013761854,0.017768664,0.027070131,0.009908308,0.59745723,0.33550698,0.00019488923],"about_ca_topic_score_codex":0.0012597146,"about_ca_topic_score_gemma":0.002289576,"teacher_disagreement_score":0.006739861,"about_ca_system_score_codex":0.0013999663,"about_ca_system_score_gemma":0.001757136,"threshold_uncertainty_score":0.027607858},"labels":[],"label_agreement":null},{"id":"W4405750718","doi":"10.18280/isi.290603","title":"An Innovative Taxonomy-Driven Approach for Word Sense Disambiguation Using Conceptual Relatedness","year":2024,"lang":"en","type":"article","venue":"Ingénierie des systèmes d information","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Taxonomy (biology); Word-sense disambiguation; Computer science; Natural language processing; Word (group theory); Linguistics; Artificial intelligence; Biology; WordNet; Philosophy; Ecology","score_opus":0.03955928192299003,"score_gpt":0.2848120381334768,"score_spread":0.24525275621048676,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405750718","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004816784,0.000582018,0.98884356,0.00035431335,0.00020690697,0.00026063263,0.001032727,0.0024259428,0.0014770885],"genre_scores_gemma":[0.040782142,0.0003465925,0.95340776,0.00023927118,0.000090784284,0.0002861924,0.002843251,0.00027325758,0.0017308063],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9942846,0.0012510555,0.00071165554,0.0015030566,0.0019929772,0.0002567654],"domain_scores_gemma":[0.9939883,0.0018702293,0.00035857788,0.0011136775,0.0023834314,0.00028580686],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029879725,0.0013102026,0.0020447772,0.01103637,0.0022185536,0.0041182074,0.0035642767,0.002380436,0.004546159],"category_scores_gemma":[0.012725582,0.00093670894,0.002408064,0.010042034,0.001039618,0.007343393,0.005687627,0.0025128876,0.004707215],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002602547,0.00047173945,0.002910588,0.001036202,0.0003982322,0.00042756295,0.001651187,0.007144041,0.031248244,0.061006986,0.025917951,0.86752695],"study_design_scores_gemma":[0.0001665718,0.0003309098,0.004460907,0.0004835149,0.0004847356,0.0020457332,0.001999747,0.5386974,0.027673947,0.3092947,0.11399175,0.0003700049],"about_ca_topic_score_codex":0.0056745876,"about_ca_topic_score_gemma":0.013881895,"teacher_disagreement_score":0.01103637,"about_ca_system_score_codex":0.0009291858,"about_ca_system_score_gemma":0.004101265,"threshold_uncertainty_score":0.015802085},"labels":[],"label_agreement":null},{"id":"W4405812762","doi":"10.37236/12464","title":"The Lexicographically Least Binary Rich Word Achieving the Repetition Threshold","year":2024,"lang":"en","type":"article","venue":"The Electronic Journal of Combinatorics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Winnipeg","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Lexicographical order; Repetition (rhetorical device); Binary number; Word (group theory); Arithmetic; Mathematics; Computer science; Combinatorics; Linguistics; Philosophy","score_opus":0.006386845421117536,"score_gpt":0.24807258380892247,"score_spread":0.24168573838780494,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405812762","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7073993,0.00092027144,0.1984844,0.0014961758,0.00017850229,0.000092296825,0.0014542531,0.0015316654,0.08844301],"genre_scores_gemma":[0.8961159,0.00033128154,0.08803041,0.00043088154,0.00014390035,0.00014139242,0.0006996306,0.00033699142,0.013769721],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9994491,0.00010258604,0.00005832903,0.00014130476,0.00011576364,0.00013289072],"domain_scores_gemma":[0.9986046,0.0006371903,0.0001995047,0.00018162305,0.00021085606,0.00016624632],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00031071386,0.00040880684,0.00050722715,0.0008863351,0.0009536162,0.0016987653,0.00044685087,0.0008162915,0.0067566056],"category_scores_gemma":[0.00290714,0.0003321733,0.00037197876,0.00057820417,0.0015078923,0.00211336,0.0013408682,0.0007969486,0.002462967],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005813354,0.00007986302,0.0020578154,0.00044319013,0.000034346627,0.0010739675,0.0011418699,0.0066939825,0.117002845,0.8116349,0.006079539,0.05317638],"study_design_scores_gemma":[0.00006486658,0.00025700484,0.001424677,0.00011928525,0.00005183188,0.0015944643,0.00052873837,0.024567993,0.049409114,0.9035307,0.018341491,0.000109822526],"about_ca_topic_score_codex":0.000359238,"about_ca_topic_score_gemma":0.00060794066,"teacher_disagreement_score":0.0067566056,"about_ca_system_score_codex":0.0004888211,"about_ca_system_score_gemma":0.0005763532,"threshold_uncertainty_score":0.022603095},"labels":[],"label_agreement":null},{"id":"W4405835851","doi":"10.1089/cmb.2024.0635","title":"Generative Adversarial Networks for Neuroimage Translation","year":2024,"lang":"en","type":"article","venue":"Journal of Computational Biology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute","funders":"","keywords":"Adversarial system; Translation (biology); Generative grammar; Computer science; Artificial intelligence; Natural language processing; Biology","score_opus":0.022621946314076054,"score_gpt":0.32234251959595933,"score_spread":0.2997205732818833,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405835851","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022704111,0.000809772,0.9700336,0.00057274866,0.00007705172,0.000050574614,0.0001808987,0.00076135603,0.0048099337],"genre_scores_gemma":[0.8693393,0.00061276025,0.11920069,0.00037841572,0.00007313445,0.00018919316,0.00045527093,0.00024136635,0.009509779],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99975806,0.00009021892,0.000008364302,0.00005600031,0.0000587213,0.000028656248],"domain_scores_gemma":[0.9991611,0.00061394507,0.00006621974,0.00005573124,0.00007798992,0.000024990008],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00073085434,0.0007242337,0.0005223381,0.00036854765,0.0001875283,0.00047625022,0.00078650226,0.000793585,0.002678042],"category_scores_gemma":[0.0023365915,0.00040770348,0.0005751484,0.00032984032,0.00072469696,0.00056827086,0.0009529037,0.0015307057,0.0005348667],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000031309744,0.00001087273,0.0001507392,0.000021557365,0.00001535579,0.000034477594,0.00001651833,0.9806788,0.0012753968,0.0058680475,0.000639491,0.011257404],"study_design_scores_gemma":[0.0000016688199,0.0000059300714,0.00003402307,0.0000031847144,0.0000018092027,0.000008205544,0.0000015349701,0.99678326,0.00035605513,0.0025612388,0.00024064879,0.0000023505982],"about_ca_topic_score_codex":0.003359285,"about_ca_topic_score_gemma":0.0036967623,"teacher_disagreement_score":0.003359285,"about_ca_system_score_codex":0.0008718727,"about_ca_system_score_gemma":0.0005835162,"threshold_uncertainty_score":0.008958936},"labels":[],"label_agreement":null},{"id":"W4405935413","doi":"10.23919/cnsm62983.2024.10814616","title":"Evaluating the Robustness of ADVENT on the VeReMi-Extension Dataset","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"New York Institute of Technology; Dalhousie University","funders":"","keywords":"Robustness (evolution); Extension (predicate logic); Computer science; Programming language","score_opus":0.0751668046710298,"score_gpt":0.3836627307274215,"score_spread":0.3084959260563917,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405935413","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8310583,0.005604504,0.02667143,0.0015957206,0.0014265795,0.0010823213,0.10261049,0.015507426,0.014443217],"genre_scores_gemma":[0.5402078,0.0008568711,0.05856302,0.00054717343,0.00029258156,0.00031353667,0.39300296,0.0004919869,0.0057240375],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9952441,0.0010210783,0.000555507,0.0014838746,0.0013751438,0.00032033675],"domain_scores_gemma":[0.99555176,0.0019801185,0.0003565557,0.0011033774,0.0007708276,0.00023739776],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006058435,0.0022483552,0.0011842827,0.004323684,0.00097247027,0.0020507465,0.0020480207,0.0018100601,0.0009526972],"category_scores_gemma":[0.010512705,0.00025549435,0.0016303026,0.0016995927,0.00081076386,0.0028092086,0.0016262208,0.0013314366,0.0011407818],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0036606316,0.003552475,0.13002841,0.0025912006,0.00262784,0.0010441658,0.0006048789,0.25455648,0.030737586,0.0042059096,0.18243845,0.38395193],"study_design_scores_gemma":[0.00042670424,0.0017182921,0.07859764,0.00027691628,0.00040599925,0.0017775513,0.0009934738,0.80282414,0.02922257,0.003786833,0.079699054,0.00027082613],"about_ca_topic_score_codex":0.015843146,"about_ca_topic_score_gemma":0.026128491,"teacher_disagreement_score":0.015843146,"about_ca_system_score_codex":0.0011144939,"about_ca_system_score_gemma":0.0010656075,"threshold_uncertainty_score":0.032040477},"labels":[],"label_agreement":null},{"id":"W4405947445","doi":"10.3390/electronics14010120","title":"Hardware Design and Verification with Large Language Models: A Scoping Review, Challenges, and Open Issues","year":2024,"lang":"en","type":"article","venue":"Electronics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University; University of Victoria","funders":"","keywords":"Computer science; Software engineering; Systems engineering; Engineering","score_opus":0.035605549226581094,"score_gpt":0.32704535847348243,"score_spread":0.2914398092469013,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405947445","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0005970687,0.97410256,0.01743676,0.0036044337,0.00049844105,0.00011002062,0.00017681056,0.00010574184,0.0033681372],"genre_scores_gemma":[0.009058921,0.9736264,0.01398469,0.0015036291,0.00040765744,0.0003532787,0.00035203705,0.000087884495,0.0006254942],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9864038,0.0061794766,0.0028215183,0.0011119897,0.0031750682,0.00030808084],"domain_scores_gemma":[0.78504765,0.19400802,0.0048538814,0.0048915497,0.010737832,0.00046107694],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.024402363,0.0015981612,0.0020793097,0.008997066,0.0009803599,0.005182153,0.0034392036,0.0034848,0.0051523442],"category_scores_gemma":[0.10749003,0.001204076,0.003182898,0.008399277,0.00300787,0.009824204,0.0027521779,0.003502728,0.0017566093],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009913523,0.00008200764,0.0008732453,0.16681056,0.00059768575,0.0003247741,0.0016112513,0.006191321,0.0009142782,0.07835294,0.022781625,0.72136116],"study_design_scores_gemma":[0.00003178394,0.00015675041,0.0008705546,0.2757138,0.0015006022,0.0007830625,0.0015155522,0.0036529354,0.0016925556,0.057610855,0.65637225,0.00009925338],"about_ca_topic_score_codex":0.0057778,"about_ca_topic_score_gemma":0.005644191,"teacher_disagreement_score":0.024402363,"about_ca_system_score_codex":0.0036894686,"about_ca_system_score_gemma":0.017265666,"threshold_uncertainty_score":0.1290536},"labels":[],"label_agreement":null},{"id":"W4406193067","doi":"10.1007/978-981-97-1818-4_35-1","title":"Design, Development, and Annotation of the Persian Spoken Learner Corpus","year":2025,"lang":"en","type":"book-chapter","venue":"Springer handbooks in languages and linguistics.","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Memorial University of Newfoundland; Cégep de l'Outaouais","funders":"","keywords":"Annotation; Persian; Computer science; Natural language processing; Artificial intelligence; Linguistics; Philosophy","score_opus":0.013986580375575534,"score_gpt":0.25966980510593585,"score_spread":0.2456832247303603,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406193067","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10023653,0.0018714331,0.6079489,0.002092273,0.0012529698,0.012204017,0.13022356,0.045727998,0.09844235],"genre_scores_gemma":[0.10422336,0.0005959203,0.62167794,0.0005612686,0.0002307648,0.017741874,0.20712265,0.01188568,0.03596058],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.997273,0.0009465498,0.00029874832,0.000803367,0.00052275887,0.00015550085],"domain_scores_gemma":[0.9942866,0.0017701786,0.00019652637,0.0011143606,0.002198138,0.00043422784],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00441013,0.0010732404,0.0009447037,0.0048293457,0.002640402,0.0029367679,0.0025829058,0.0009807008,0.029449461],"category_scores_gemma":[0.0079291295,0.0011304036,0.00044425164,0.0032604996,0.0017048474,0.0039020923,0.005788731,0.0026006696,0.024590021],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000773176,0.00045931232,0.006215898,0.0024383615,0.000068340385,0.0013357162,0.012924975,0.0043404615,0.1351279,0.025638018,0.17263378,0.63804406],"study_design_scores_gemma":[0.00032474406,0.000565803,0.018748058,0.00041833508,0.00011740359,0.0016754584,0.00892303,0.027379373,0.13042037,0.015238641,0.7959465,0.00024228306],"about_ca_topic_score_codex":0.010371455,"about_ca_topic_score_gemma":0.01754645,"teacher_disagreement_score":0.029449461,"about_ca_system_score_codex":0.0014365981,"about_ca_system_score_gemma":0.0049298336,"threshold_uncertainty_score":0.09851825},"labels":[],"label_agreement":null},{"id":"W4406216908","doi":"10.15291/9789533315355","title":"Corpora in Language Learning, Translation and Research","year":2024,"lang":"en","type":"paratext","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Francophone University Association","funders":"","keywords":"Computer science; Natural language processing; Translation (biology); Artificial intelligence; Machine translation; Linguistics; Philosophy","score_opus":0.051783797285471125,"score_gpt":0.3845269411821286,"score_spread":0.3327431438966575,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406216908","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013412127,0.22181822,0.26720643,0.07093367,0.012266978,0.001472272,0.0064357123,0.0024472876,0.40400726],"genre_scores_gemma":[0.21991494,0.21110694,0.44497967,0.012009785,0.010916406,0.0056871814,0.011571689,0.004725778,0.07908753],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.95011425,0.036327727,0.0033503568,0.0024580604,0.0072511504,0.0004984817],"domain_scores_gemma":[0.83573884,0.13274756,0.0043080165,0.016414626,0.009386982,0.0014040554],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.035799526,0.000991105,0.0016159796,0.017798511,0.0059634764,0.024836067,0.0029150555,0.0035673922,0.017767005],"category_scores_gemma":[0.09811253,0.001141496,0.00053374877,0.03377065,0.012888135,0.030459313,0.009669325,0.0045766765,0.0043028677],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000083419814,0.000079368925,0.0012927665,0.0036495833,0.00007621989,0.00042060402,0.009494837,0.0015784344,0.0007078232,0.5868445,0.067168534,0.32860392],"study_design_scores_gemma":[0.000025158804,0.000033728513,0.0008145361,0.003932721,0.0000254729,0.0003129516,0.0048488025,0.001152614,0.001028385,0.2035128,0.78426397,0.000048876755],"about_ca_topic_score_codex":0.0037484306,"about_ca_topic_score_gemma":0.005691741,"teacher_disagreement_score":0.035799526,"about_ca_system_score_codex":0.005777438,"about_ca_system_score_gemma":0.008047697,"threshold_uncertainty_score":0.1893282},"labels":[],"label_agreement":null},{"id":"W4406220024","doi":"10.1111/lang.12702","title":"Meaning‐Inferencing Versus Meaning‐Given Procedures: The Case of Idioms","year":2025,"lang":"en","type":"article","venue":"Language Learning","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Meaning (existential); Psychology; Linguistics; Philosophy; Psychotherapist","score_opus":0.010544356441254342,"score_gpt":0.2894165473152566,"score_spread":0.2788721908740023,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406220024","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.97193223,0.00013532511,0.023029093,0.00027507532,0.000010850634,0.000105132305,0.000014126477,0.00004332818,0.0044548376],"genre_scores_gemma":[0.98542374,0.00009627762,0.014119162,0.00003992281,0.000007836245,0.00004070878,0.000010343585,0.00001140976,0.0002505505],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.98614174,0.011647907,0.00041537196,0.00040108853,0.0010673624,0.00032651896],"domain_scores_gemma":[0.854138,0.13452141,0.0041762395,0.004726783,0.0018386715,0.00059893506],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012290361,0.00037713657,0.00029378288,0.0005944949,0.0006140712,0.0013862534,0.00086243235,0.00078666495,0.0027881186],"category_scores_gemma":[0.099015996,0.00025157793,0.00022245431,0.00040395488,0.0029306307,0.003326245,0.001589163,0.0017441225,0.0002496211],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0037159529,0.002942664,0.049177337,0.0019035332,0.00011630729,0.005042481,0.1599542,0.0072533656,0.0825887,0.09382429,0.0009380641,0.59254307],"study_design_scores_gemma":[0.00071934,0.007672329,0.16597071,0.0014238274,0.00039185662,0.012189826,0.104921415,0.09159797,0.28755033,0.28415802,0.042775743,0.0006287115],"about_ca_topic_score_codex":0.0003637239,"about_ca_topic_score_gemma":0.00061009673,"teacher_disagreement_score":0.012290361,"about_ca_system_score_codex":0.0004349874,"about_ca_system_score_gemma":0.00073742925,"threshold_uncertainty_score":0.06499839},"labels":[],"label_agreement":null},{"id":"W4406250857","doi":"10.6018/ijes.584081","title":"Clause Initial Null Subjects in Web-based Written Language","year":2024,"lang":"en","type":"article","venue":"International Journal of English Studies","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"Ministerio de Ciencia, Innovación y Universidades","keywords":"Null (SQL); Linguistics; Computer science; Natural language processing; Web application; Psychology; Mathematics; World Wide Web; Philosophy; Database","score_opus":0.018683472693563047,"score_gpt":0.34764614673350297,"score_spread":0.32896267403993995,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406250857","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9916785,0.00028027207,0.0036202408,0.00008871189,0.00001270552,0.00003939116,0.0010942721,0.00010054043,0.0030854165],"genre_scores_gemma":[0.99589634,0.00007761936,0.00146053,0.000023914161,0.000008701272,0.000027191772,0.0017622648,0.000060909504,0.00068259233],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9940561,0.0024868587,0.0010508574,0.0009404565,0.0012369073,0.00022874112],"domain_scores_gemma":[0.9271622,0.053186335,0.010045608,0.005257619,0.0038971952,0.0004510999],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004543866,0.0002582407,0.00034491683,0.0024873186,0.00053420226,0.002161539,0.000557921,0.0003696104,0.0038568948],"category_scores_gemma":[0.037051406,0.00025866748,0.00022067853,0.0033342643,0.0013633599,0.0021847486,0.0014403255,0.00062137743,0.0010668366],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010302917,0.0002711126,0.84739596,0.000684305,0.000091906775,0.0027611228,0.029377904,0.00051040016,0.006477924,0.0065325676,0.0025590663,0.10230742],"study_design_scores_gemma":[0.00004037776,0.00027925745,0.91532755,0.0002829864,0.00012212218,0.004867298,0.030440906,0.0066637355,0.015060926,0.0077565084,0.019075738,0.00008261955],"about_ca_topic_score_codex":0.0028953487,"about_ca_topic_score_gemma":0.002803732,"teacher_disagreement_score":0.004543866,"about_ca_system_score_codex":0.00047512777,"about_ca_system_score_gemma":0.00050541706,"threshold_uncertainty_score":0.024030566},"labels":[],"label_agreement":null},{"id":"W4406273356","doi":"10.1075/ml.24021.tam","title":"Processing costs in Cantonese-Latin script-mixing","year":2024,"lang":"en","type":"article","venue":"The Mental Lexicon","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Windsor; The Scarborough Hospital; Brock University; University of Toronto","funders":"Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada","keywords":"Mixing (physics); Computer science; Natural language processing; Artificial intelligence; Linguistics; Physics; Philosophy","score_opus":0.017345400944147833,"score_gpt":0.29232999960306644,"score_spread":0.2749845986589186,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406273356","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99727696,0.000025899071,0.00047596404,0.00003422915,0.0000030988574,0.0000121031635,0.00004092394,0.000016950105,0.0021138682],"genre_scores_gemma":[0.99806494,0.000022769527,0.000960432,0.000021502932,0.0000027126978,0.000022749276,0.0001274198,0.000025427129,0.00075207476],"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99936813,0.00017501257,0.000060973027,0.00016901582,0.00015006997,0.000076787175],"domain_scores_gemma":[0.99344975,0.0043301205,0.00096763484,0.00057774544,0.00041057466,0.00026411764],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00089906255,0.00052567024,0.0003836078,0.0003988506,0.0004189512,0.0020842946,0.00043028998,0.00037939614,0.0074630096],"category_scores_gemma":[0.009195049,0.00037615563,0.00016165197,0.00030375863,0.0006596259,0.0016575058,0.0008002941,0.00067175296,0.0002640968],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004420531,0.00047360582,0.06288974,0.000546359,0.000105550185,0.0019376833,0.008383859,0.0019379337,0.8376044,0.006912068,0.0006770401,0.07411112],"study_design_scores_gemma":[0.00022751835,0.0009234327,0.82931405,0.00005868066,0.0002081059,0.0018743656,0.005162089,0.02179628,0.122711465,0.014932898,0.0026299385,0.00016116136],"about_ca_topic_score_codex":0.007545916,"about_ca_topic_score_gemma":0.008570659,"teacher_disagreement_score":0.007545916,"about_ca_system_score_codex":0.0009516158,"about_ca_system_score_gemma":0.00047891313,"threshold_uncertainty_score":0.02496624},"labels":[],"label_agreement":null},{"id":"W4406369570","doi":"10.1121/10.0035093","title":"The AnySpeech Project —Open-vocabulary keyword spotting and phonetic transcription in any language","year":2024,"lang":"en","type":"article","venue":"The Journal of the Acoustical Society of America","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Keyword spotting; Vocabulary; Transcription (linguistics); Spotting; Computer science; Linguistics; Natural language processing; Artificial intelligence","score_opus":0.010795820575793195,"score_gpt":0.28456061252893483,"score_spread":0.27376479195314163,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406369570","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.037909806,0.0021118354,0.52335024,0.0044517545,0.005117397,0.0013116967,0.13344596,0.26612705,0.026174298],"genre_scores_gemma":[0.08325773,0.0008428877,0.36706993,0.0016472663,0.00068666966,0.0023983635,0.49435556,0.023940966,0.02580058],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99198264,0.0027485087,0.0005604498,0.0021640952,0.0020671533,0.0004771641],"domain_scores_gemma":[0.98539585,0.004107795,0.00041581097,0.004902236,0.0036755693,0.0015026974],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007897486,0.0037885786,0.0019513264,0.0023122842,0.0017690656,0.004400435,0.004601331,0.0030099328,0.021624168],"category_scores_gemma":[0.018441675,0.0012340735,0.0021360514,0.0014488457,0.0018483863,0.008828889,0.009355859,0.004122543,0.041705403],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015162833,0.0006748181,0.0028003077,0.0011064884,0.00027969514,0.00040349795,0.0014215603,0.0073546525,0.024458569,0.008291413,0.6109818,0.3407109],"study_design_scores_gemma":[0.0010148789,0.0013576916,0.00814038,0.00054804713,0.00031646233,0.0012844176,0.0032249556,0.2559809,0.07429786,0.034992304,0.61828667,0.0005553967],"about_ca_topic_score_codex":0.017380362,"about_ca_topic_score_gemma":0.015592216,"teacher_disagreement_score":0.021624168,"about_ca_system_score_codex":0.0012487753,"about_ca_system_score_gemma":0.0047929874,"threshold_uncertainty_score":0.07234007},"labels":[],"label_agreement":null},{"id":"W4406499745","doi":"10.1109/cascon62161.2024.10837964","title":"Transformer-Based Text Highlighting for Medical Terms","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Computer science; Transformer; Electrical engineering; Engineering; Voltage","score_opus":0.010022330863076365,"score_gpt":0.2938376376285509,"score_spread":0.28381530676547456,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406499745","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13030593,0.0028651203,0.77910954,0.0009959241,0.00046579642,0.00046784608,0.008222705,0.07144418,0.006123006],"genre_scores_gemma":[0.57126004,0.0009829047,0.40637755,0.00044524213,0.00017994743,0.0002295421,0.011111407,0.0016533982,0.0077600116],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99956113,0.000116830204,0.000035970847,0.00015621894,0.00008464899,0.000045216104],"domain_scores_gemma":[0.99789447,0.00126227,0.00021933678,0.0002702176,0.00026217574,0.000091487775],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008210756,0.0010142395,0.00045238223,0.0020124125,0.00026523307,0.0008297406,0.00084871374,0.00057519367,0.0045940424],"category_scores_gemma":[0.0043370724,0.0002715683,0.0007937885,0.001138975,0.000348432,0.0021509358,0.0010247036,0.0010624476,0.0036020041],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014696813,0.00031272956,0.0068044635,0.00083810184,0.00011066737,0.00061809894,0.0007273754,0.063515835,0.057291716,0.0076931324,0.03581226,0.8248059],"study_design_scores_gemma":[0.00009420766,0.0004025782,0.003921776,0.00009023939,0.00009019775,0.00079685403,0.00030504484,0.8999173,0.048192646,0.014520997,0.03157904,0.000089104586],"about_ca_topic_score_codex":0.0025277692,"about_ca_topic_score_gemma":0.0054990086,"teacher_disagreement_score":0.0045940424,"about_ca_system_score_codex":0.0005624477,"about_ca_system_score_gemma":0.0010897077,"threshold_uncertainty_score":0.015368581},"labels":[],"label_agreement":null},{"id":"W4406499752","doi":"10.1109/cascon62161.2024.10837905","title":"Anaphora Resolution in Software Requirements Engineering: A Comparison of Generative NLP Pipelines and Encoder-Based Models","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Computer science; Artificial intelligence; Resolution (logic); Pipeline transport; Natural language processing; Generative grammar; Software; Anaphora (linguistics); Programming language; Engineering","score_opus":0.04183350212277246,"score_gpt":0.3194555275864681,"score_spread":0.27762202546369563,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406499752","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.22177853,0.0018467061,0.75807,0.0017534172,0.00007814283,0.00043004035,0.0008114134,0.007414929,0.007816817],"genre_scores_gemma":[0.77321464,0.0009837277,0.22113661,0.00029504657,0.00003655909,0.00021047614,0.0016094692,0.0002769122,0.0022364925],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99773955,0.0012552147,0.00012683628,0.0003601447,0.00041585203,0.00010250516],"domain_scores_gemma":[0.9864218,0.011086269,0.000368096,0.0011233335,0.0008401608,0.0001603192],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006151619,0.000742309,0.0006902418,0.0014676773,0.0005332322,0.001487021,0.0017164553,0.0012313155,0.0016603091],"category_scores_gemma":[0.014168875,0.0007017944,0.0010037892,0.0014747756,0.0006932585,0.003844846,0.0011963055,0.002004039,0.0008794373],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00077578746,0.00062684214,0.007200673,0.0005312523,0.00017270708,0.00014058754,0.0007634327,0.55955744,0.004779415,0.019421183,0.0032836099,0.40274706],"study_design_scores_gemma":[0.00001605074,0.000057809866,0.0003654259,0.0000117096015,0.000018338038,0.000026106472,0.000032471693,0.9950146,0.0013785559,0.0025762615,0.0004940036,0.000008694919],"about_ca_topic_score_codex":0.017356886,"about_ca_topic_score_gemma":0.018866574,"teacher_disagreement_score":0.017356886,"about_ca_system_score_codex":0.0024493635,"about_ca_system_score_gemma":0.00239479,"threshold_uncertainty_score":0.034511685},"labels":[],"label_agreement":null},{"id":"W4406499863","doi":"10.1109/cascon62161.2024.10837900","title":"Spelling Corrector for Turkish Product Search","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Turkish; Spelling; Computer science; Product (mathematics); Artificial intelligence; Mathematics; Linguistics; Geometry","score_opus":0.025800236222036387,"score_gpt":0.3189838370422317,"score_spread":0.29318360082019534,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406499863","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1476231,0.0024701853,0.76148343,0.0007700983,0.0005571553,0.0006126926,0.0044918777,0.07640379,0.005587714],"genre_scores_gemma":[0.5705067,0.0007883419,0.40519086,0.00035970233,0.00017320075,0.00027296445,0.009491375,0.0012962127,0.011920581],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9989511,0.00023772723,0.00013397739,0.00034280875,0.00023856483,0.00009585859],"domain_scores_gemma":[0.997294,0.0008045002,0.00037100774,0.0005096475,0.00094313035,0.000077605204],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011782796,0.0014028603,0.0008939876,0.0025328598,0.0006534502,0.00104289,0.0014405485,0.0008254876,0.006539104],"category_scores_gemma":[0.006489204,0.00033196848,0.0010422688,0.0017384504,0.00036366322,0.0016065784,0.00076058006,0.0011873823,0.0077463505],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010702471,0.00027530926,0.009221653,0.0004010376,0.00014533618,0.0004386875,0.00023285444,0.034040667,0.029435828,0.0023774132,0.028867563,0.89349335],"study_design_scores_gemma":[0.000097095544,0.00040988546,0.0053163143,0.00004689434,0.00014864808,0.0007515864,0.00015821424,0.93533915,0.03591277,0.002287477,0.019442843,0.00008913737],"about_ca_topic_score_codex":0.016533418,"about_ca_topic_score_gemma":0.026322696,"teacher_disagreement_score":0.016533418,"about_ca_system_score_codex":0.0009779345,"about_ca_system_score_gemma":0.002465261,"threshold_uncertainty_score":0.032874346},"labels":[],"label_agreement":null},{"id":"W4406580460","doi":"10.1016/s1541-9800(06)71046-8","title":"10.1016/s1541-9800(06)71046-8","year":2000,"lang":"en","type":"article","venue":"Time to knit","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Vitamin D and neurology; Calcium; Medicine; Endocrinology; Internal medicine","score_opus":0.005012501286981302,"score_gpt":0.1951778209311315,"score_spread":0.1901653196441502,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406580460","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00049293786,0.00042822855,0.0011413358,0.00039854753,0.00027886205,0.00012769134,0.0011111905,0.0015181189,0.99450314],"genre_scores_gemma":[0.0005849349,0.00018578612,0.0005292736,0.00021145004,0.000055484557,0.00005870266,0.0005934458,0.00023226412,0.99754864],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.999121,0.00006679985,0.00007333968,0.0003350073,0.00023147768,0.00017234094],"domain_scores_gemma":[0.9972132,0.00074474677,0.0001776172,0.00039803094,0.00066217355,0.00080435904],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.0013212815,0.003243163,0.0019751622,0.003070428,0.0026543934,0.004510912,0.0037231992,0.0063089365,0.99030834],"category_scores_gemma":[0.0019217655,0.0010757139,0.0015643105,0.003106045,0.001977984,0.00645438,0.003551672,0.0029226104,0.9935016],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00033885209,0.0002026418,0.0008768308,0.0005998314,0.00004195635,0.00026671137,0.00011447131,0.00043419088,0.0026682832,0.006506549,0.35620564,0.6317441],"study_design_scores_gemma":[0.00005039534,0.00010696465,0.0006020869,0.00028718633,0.000015400818,0.00029589765,0.00011444135,0.0002770235,0.000499011,0.0006243327,0.9971003,0.000026910087],"about_ca_topic_score_codex":0.004042583,"about_ca_topic_score_gemma":0.0033145722,"teacher_disagreement_score":0.009691656,"about_ca_system_score_codex":0.0013001999,"about_ca_system_score_gemma":0.0012306636,"threshold_uncertainty_score":0.013823986},"labels":[],"label_agreement":null},{"id":"W4406688098","doi":"10.1145/3714461","title":"Exploring Parameter-Efficient Fine-Tuning Techniques for Code Generation with Large Language Models","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":43,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Code generation; Code (set theory); Programming language; Software engineering; Key (lock); Operating system; Set (abstract data type)","score_opus":0.13664566024265196,"score_gpt":0.3367384357981753,"score_spread":0.20009277555552335,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406688098","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0520844,0.0018874794,0.9171027,0.00063370453,0.00014537963,0.00027946118,0.00069186505,0.023796469,0.0033785922],"genre_scores_gemma":[0.5033743,0.000656407,0.48738304,0.0008115688,0.00007192721,0.00066747743,0.0021002719,0.0029843948,0.0019506399],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987362,0.00045680944,0.00009652996,0.0003672533,0.00023434592,0.000108886015],"domain_scores_gemma":[0.9957652,0.0028300423,0.000196191,0.00079928496,0.00029651998,0.00011273708],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017527387,0.0016641556,0.0008798523,0.0008293414,0.000416937,0.0015001369,0.0025161195,0.0013135881,0.003332289],"category_scores_gemma":[0.015830526,0.00073559175,0.0012062811,0.000756048,0.0008858412,0.002860836,0.0019076803,0.0029362917,0.0023281712],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037426336,0.00038955032,0.0055670594,0.0007574429,0.0001892885,0.0002879374,0.00058380153,0.49560973,0.02405798,0.010345452,0.010918698,0.45091888],"study_design_scores_gemma":[0.00006119341,0.00006936251,0.00024622332,0.000032410095,0.00002515733,0.00006073649,0.00007670412,0.9784138,0.005627974,0.011804465,0.0035635352,0.000018474078],"about_ca_topic_score_codex":0.005018904,"about_ca_topic_score_gemma":0.012148481,"teacher_disagreement_score":0.005018904,"about_ca_system_score_codex":0.000963216,"about_ca_system_score_gemma":0.0018172908,"threshold_uncertainty_score":0.011147618},"labels":[],"label_agreement":null},{"id":"W4406704716","doi":"10.1007/s11049-024-09643-3","title":"Effects of uniqueness on extraction from definite NP objects","year":2025,"lang":"en","type":"article","venue":"Natural Language & Linguistic Theory","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"University of Delaware","keywords":"Uniqueness; Positive-definite matrix; Extraction (chemistry); Mathematics; Mathematical analysis; Chemistry; Physics; Chromatography","score_opus":0.0032187017155336185,"score_gpt":0.2718462062621694,"score_spread":0.26862750454663575,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406704716","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9703794,0.0020671505,0.013178952,0.000751586,0.00042018283,0.00008650261,0.00087750354,0.0023784451,0.0098603],"genre_scores_gemma":[0.97461265,0.0004939821,0.017539494,0.00031514146,0.00012770396,0.000042548156,0.0015251851,0.0022813645,0.003062039],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99235404,0.003357008,0.0013440037,0.0013968498,0.0009758287,0.0005722192],"domain_scores_gemma":[0.70629835,0.26758197,0.004145359,0.013760365,0.006579356,0.0016346719],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011757935,0.0013616399,0.0016737706,0.0011777182,0.0020997638,0.004400426,0.0014509092,0.0019704613,0.009779544],"category_scores_gemma":[0.11910201,0.0013165835,0.0010229547,0.0016619365,0.0018895854,0.010307837,0.004254595,0.0025270153,0.0022252738],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.06610518,0.0020703252,0.06772663,0.004030573,0.0014124658,0.004399984,0.002991652,0.029839829,0.32822385,0.012330293,0.013787222,0.4670821],"study_design_scores_gemma":[0.00087336596,0.004514383,0.100166045,0.00040891898,0.0037715416,0.0058565163,0.004406903,0.12141476,0.70592034,0.039153036,0.012996989,0.00051726005],"about_ca_topic_score_codex":0.0022676599,"about_ca_topic_score_gemma":0.0031734854,"teacher_disagreement_score":0.011757935,"about_ca_system_score_codex":0.0006741388,"about_ca_system_score_gemma":0.0015070497,"threshold_uncertainty_score":0.062182665},"labels":[],"label_agreement":null},{"id":"W4406735064","doi":"10.18653/v1/2025.emnlp-main.1413","title":"AFRIDOC-MT: Document-level MT Corpus for African Languages","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; Canadian Institute for Advanced Research","funders":"Agence Nationale de la Recherche; International Development Research Centre; Rockefeller Foundation","keywords":"Linguistics; Natural language processing; Computer science; History; Philosophy","score_opus":0.014788066328694727,"score_gpt":0.30713322854891384,"score_spread":0.2923451622202191,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406735064","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08262798,0.0062856036,0.080577105,0.0035271717,0.0017913097,0.0015542106,0.7210131,0.045620218,0.0570033],"genre_scores_gemma":[0.115919545,0.0013833902,0.085577115,0.00033080156,0.00024398872,0.0014188207,0.77946913,0.005397561,0.0102595575],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9990069,0.000289865,0.00016974306,0.00021733917,0.00021837426,0.00009777688],"domain_scores_gemma":[0.9961332,0.0014035909,0.00031388822,0.0007547064,0.0010650547,0.0003296046],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018020638,0.0011146187,0.0007198725,0.0045315647,0.0018477505,0.0015869311,0.0014935355,0.0013910573,0.022957217],"category_scores_gemma":[0.009171958,0.00057976035,0.0003352498,0.0041353917,0.00058225397,0.0032835042,0.0027446512,0.0016264337,0.012915328],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015381142,0.00032143088,0.003268547,0.0045098956,0.00007904164,0.0012166533,0.0016915472,0.0026646932,0.028992916,0.01585965,0.5896085,0.35024896],"study_design_scores_gemma":[0.00057527126,0.00016211081,0.011156169,0.0005355433,0.00010084099,0.0013498205,0.00082041416,0.0085859895,0.027318062,0.009254573,0.94001865,0.00012255079],"about_ca_topic_score_codex":0.005855454,"about_ca_topic_score_gemma":0.006708491,"teacher_disagreement_score":0.022957217,"about_ca_system_score_codex":0.00062867673,"about_ca_system_score_gemma":0.002522667,"threshold_uncertainty_score":0.07679951},"labels":[],"label_agreement":null},{"id":"W4406773963","doi":"10.1109/icassp49660.2025.10889681","title":"Adapting Without Seeing: Text-Aided Domain Adaptation for Adapting CLIP-like Models to Novel Domains","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Safran Electronics (Canada)","funders":"","keywords":"Computer science; Domain adaptation; Adaptation (eye); Domain (mathematical analysis); Human–computer interaction; Natural language processing; Artificial intelligence; Psychology; Mathematics; Neuroscience","score_opus":0.041586150460562986,"score_gpt":0.3052626088002666,"score_spread":0.26367645833970366,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406773963","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04023893,0.0007885924,0.9329891,0.00041273798,0.0005039416,0.00023716617,0.00091710855,0.020895915,0.0030164954],"genre_scores_gemma":[0.43221527,0.00060656015,0.5498113,0.0013553826,0.00027000826,0.00042130286,0.0048268815,0.0015540753,0.008939172],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9993587,0.00016394316,0.000028413346,0.00029238124,0.000089112764,0.00006738128],"domain_scores_gemma":[0.9984403,0.00052329816,0.00008808397,0.0005968219,0.00025012583,0.000101448444],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013410144,0.0015891411,0.0007593836,0.0007494736,0.00047074855,0.0012516826,0.0021532276,0.0014944153,0.0037548502],"category_scores_gemma":[0.0055514383,0.0004472937,0.0011360488,0.000815568,0.0007953429,0.002398777,0.002128024,0.0034764544,0.0031147122],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005469194,0.0005200093,0.0023362527,0.00030212558,0.00019933036,0.00040013532,0.00042711335,0.1788545,0.06474375,0.004973213,0.039501738,0.7071949],"study_design_scores_gemma":[0.000036848884,0.00011545798,0.0004864644,0.000018310331,0.000028415798,0.00012482078,0.00011838958,0.96544147,0.019172894,0.006469303,0.007954343,0.00003328346],"about_ca_topic_score_codex":0.003952108,"about_ca_topic_score_gemma":0.005764987,"teacher_disagreement_score":0.003952108,"about_ca_system_score_codex":0.0006857984,"about_ca_system_score_gemma":0.00073065126,"threshold_uncertainty_score":0.012561262},"labels":[],"label_agreement":null},{"id":"W4406808467","doi":"10.5430/wjel.v15n3p354","title":"Evaluating the Performance of Large Language Models on Arabic Lexical Ambiguities: A Comparative Study with Traditional Machine Translation Systems","year":2025,"lang":"en","type":"article","venue":"World Journal of English Language","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"Prince Sattam bin Abdulaziz University","keywords":"Computer science; Arabic; Machine translation; Natural language processing; Artificial intelligence; Translation (biology); Linguistics; Philosophy; Chemistry","score_opus":0.048740441268314234,"score_gpt":0.34244849435631014,"score_spread":0.2937080530879959,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406808467","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9309984,0.0062762606,0.04790936,0.0008320798,0.0002930174,0.00027699635,0.0010581008,0.0024605042,0.009895455],"genre_scores_gemma":[0.9626979,0.0009747654,0.033038806,0.00013584566,0.00008747058,0.000100226614,0.0016607788,0.00018081906,0.0011234331],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9949921,0.002818173,0.0005281407,0.000703622,0.00081907335,0.00013887843],"domain_scores_gemma":[0.9717172,0.023352373,0.0007025052,0.0017540937,0.0021533384,0.00032047936],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0068229213,0.001551457,0.0011958327,0.0022986575,0.0008351798,0.0022986694,0.0014129505,0.0013847881,0.0022626065],"category_scores_gemma":[0.027187588,0.00033345868,0.0009099724,0.002650884,0.00077048806,0.0037422264,0.0016622427,0.0010214876,0.0012661483],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0050954972,0.0017222126,0.039056387,0.0022290892,0.0016749616,0.0008368194,0.002777483,0.37070906,0.015009577,0.003978396,0.008515954,0.54839444],"study_design_scores_gemma":[0.00025681264,0.0022779782,0.019378336,0.00012533164,0.00056920195,0.0004736304,0.0015724055,0.9524044,0.013048473,0.0042766165,0.0054834737,0.00013335887],"about_ca_topic_score_codex":0.008466288,"about_ca_topic_score_gemma":0.007962662,"teacher_disagreement_score":0.008466288,"about_ca_system_score_codex":0.0012851097,"about_ca_system_score_gemma":0.001108038,"threshold_uncertainty_score":0.03608352},"labels":[],"label_agreement":null},{"id":"W4406866661","doi":"10.1145/3715109","title":"HumanEvalComm: Benchmarking the Communication Competence of Code Generation for LLMs and LLM Agent","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia, Okanagan Campus; Kelowna General Hospital; University of British Columbia","funders":"","keywords":"Benchmarking; Computer science; Competence (human resources); Knowledge management; Business; Psychology; Marketing","score_opus":0.08995787232149585,"score_gpt":0.34867155279438006,"score_spread":0.25871368047288423,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406866661","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.94323725,0.0005036569,0.035617195,0.00054686354,0.00016246836,0.0009730293,0.0016025918,0.0064333915,0.010923616],"genre_scores_gemma":[0.9118089,0.00015994349,0.07518152,0.00035409292,0.000029813818,0.0012553494,0.006818368,0.0006652083,0.003726848],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9849145,0.009136591,0.0013057032,0.0017967961,0.0021900837,0.00065632653],"domain_scores_gemma":[0.93659776,0.0412175,0.0031974225,0.009975861,0.0062996875,0.0027116719],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010986316,0.0011106329,0.0005602631,0.0018712084,0.00068466284,0.0018754794,0.0022461088,0.0019427042,0.0024312495],"category_scores_gemma":[0.054456152,0.00043119947,0.00061232666,0.0011072088,0.0017629663,0.002683057,0.0034729508,0.0024716798,0.0012187858],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.007538795,0.016112057,0.099203765,0.004483792,0.00066604,0.0010579686,0.018573293,0.14214353,0.042805314,0.017902922,0.049624592,0.5998879],"study_design_scores_gemma":[0.0022441417,0.013761508,0.11575112,0.00058277836,0.00025043747,0.0010702792,0.008640958,0.6559467,0.075216495,0.020470269,0.105456986,0.0006084432],"about_ca_topic_score_codex":0.0041897814,"about_ca_topic_score_gemma":0.004271389,"teacher_disagreement_score":0.010986316,"about_ca_system_score_codex":0.001540262,"about_ca_system_score_gemma":0.0019257758,"threshold_uncertainty_score":0.058101892},"labels":[],"label_agreement":null},{"id":"W4406892560","doi":"10.1109/fllm63129.2024.10852439","title":"Evaluating Large Language Models for Code Generation: Assessing Accuracy, Quality, and Performance","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Code (set theory); Quality (philosophy); Software quality; Reliability engineering; Programming language; Software; Engineering; Software development","score_opus":0.15856994438783298,"score_gpt":0.46352627942698466,"score_spread":0.30495633503915165,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406892560","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.75125664,0.0026578838,0.19823179,0.0010271452,0.00029810294,0.00095117773,0.0035496512,0.0360115,0.0060161795],"genre_scores_gemma":[0.77782935,0.000698267,0.20963575,0.00029824008,0.000041990963,0.00050150626,0.008222796,0.0014274599,0.0013446959],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9932459,0.00290285,0.00081618177,0.0010498603,0.0017182668,0.00026699738],"domain_scores_gemma":[0.9415139,0.04313912,0.0025525296,0.0062446184,0.005486796,0.0010630991],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010126821,0.0016473032,0.00087524834,0.0020671445,0.0005399233,0.0022488013,0.0022774464,0.0016067023,0.0014519279],"category_scores_gemma":[0.061346393,0.0006032639,0.0011895263,0.0014869757,0.0008268389,0.0032343578,0.0018436192,0.0018569997,0.0010170795],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0026076674,0.0019260143,0.04636565,0.0022162665,0.0008621772,0.000498298,0.0016126689,0.4576778,0.019825801,0.00454762,0.020721056,0.44113898],"study_design_scores_gemma":[0.0001437321,0.00057689054,0.0036505088,0.000086695814,0.00008810324,0.00011448869,0.00025298866,0.9769652,0.013467121,0.0018246924,0.0027793876,0.00005010535],"about_ca_topic_score_codex":0.012496194,"about_ca_topic_score_gemma":0.013623777,"teacher_disagreement_score":0.012496194,"about_ca_system_score_codex":0.0018098612,"about_ca_system_score_gemma":0.002825665,"threshold_uncertainty_score":0.053556323},"labels":[],"label_agreement":null},{"id":"W4406948310","doi":"10.1109/tai.2025.3535456","title":"SAMScore: A Content Structural Similarity Metric for Image Translation Evaluation","year":2025,"lang":"en","type":"article","venue":"IEEE Transactions on Artificial Intelligence","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"National Institute of Biomedical Imaging and Bioengineering; National Institutes of Health; National Cancer Institute; Foundation for the National Institutes of Health","keywords":"Translation (biology); Similarity (geometry); Structural similarity; Metric (unit); Content (measure theory); Artificial intelligence; Computer science; Image (mathematics); Mathematics; Pattern recognition (psychology); Natural language processing; Biology; Mathematical analysis; Engineering; Biochemistry","score_opus":0.12617588977865285,"score_gpt":0.38295597615255395,"score_spread":0.2567800863739011,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406948310","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.22298133,0.0080628935,0.7273244,0.0005195685,0.0009707484,0.0013495887,0.009002395,0.013640749,0.016148364],"genre_scores_gemma":[0.61586463,0.001213172,0.3578304,0.00034810434,0.00022727642,0.0010016618,0.017052898,0.0015853028,0.0048764544],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99499243,0.0012873407,0.000628116,0.0005831654,0.002314756,0.00019420049],"domain_scores_gemma":[0.9926615,0.0033332568,0.0006455069,0.00083903177,0.0022445542,0.00027617376],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0038077,0.0018725314,0.0015370022,0.00635851,0.00068324123,0.0019385279,0.0015017599,0.0018248248,0.0045653186],"category_scores_gemma":[0.01899596,0.00023609822,0.0010648724,0.0040817456,0.00085964444,0.003090767,0.001856428,0.00096928177,0.0020558957],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001530589,0.00043048288,0.014543998,0.002364534,0.0009588961,0.00039658128,0.00048657533,0.043600626,0.050388016,0.008069257,0.046026763,0.83120364],"study_design_scores_gemma":[0.00029059665,0.0027782347,0.029394167,0.00029588258,0.00043911,0.002225346,0.0007412342,0.82447,0.084423475,0.019324178,0.035358246,0.00025951248],"about_ca_topic_score_codex":0.002129144,"about_ca_topic_score_gemma":0.0032706656,"teacher_disagreement_score":0.00635851,"about_ca_system_score_codex":0.0010129344,"about_ca_system_score_gemma":0.0009074157,"threshold_uncertainty_score":0.02013725},"labels":[],"label_agreement":null},{"id":"W4407083508","doi":"10.1038/s41586-025-08655-2","title":"Author Correction: Joint speech and text machine translation for up to 100 languages","year":2025,"lang":"en","type":"erratum","venue":"Nature","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"McGill University","funders":"","keywords":"Machine translation; Computer science; Translation (biology); Joint (building); Natural language processing; Linguistics; Artificial intelligence; Speech translation; Speech recognition; Biology; Philosophy; Engineering","score_opus":0.014437996823840641,"score_gpt":0.31977168904076236,"score_spread":0.30533369221692175,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407083508","genre_codex":"editorial","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0008297123,0.0011321539,0.008775522,0.043268815,0.92753303,0.000054933018,0.0041026566,0.0019570242,0.012346188],"genre_scores_gemma":[0.046590492,0.006403147,0.049010824,0.05900041,0.12288047,0.0003631622,0.015573966,0.008741124,0.6914364],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99456465,0.00080545933,0.0009582432,0.00084718567,0.0025341846,0.0002902935],"domain_scores_gemma":[0.9679289,0.006976814,0.0010129745,0.002681691,0.020661157,0.00073843164],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00336501,0.0017957909,0.0013026301,0.003580517,0.0035389857,0.0032583768,0.0025516453,0.0034512088,0.084736034],"category_scores_gemma":[0.046217956,0.00079366594,0.0010582882,0.003219023,0.0016733977,0.002337462,0.0023350436,0.006217096,0.04592335],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003658,0.0000065289523,0.00009822009,0.00011017251,0.00001457403,0.0004894073,0.0000715867,0.00007920253,0.0001888281,0.001770126,0.9830294,0.014105329],"study_design_scores_gemma":[0.0000256985,0.000017133249,0.0004381761,0.00022060318,0.000043393833,0.0012053987,0.000104098486,0.00055123534,0.0011798239,0.002741678,0.99343807,0.000034682173],"about_ca_topic_score_codex":0.016320229,"about_ca_topic_score_gemma":0.0282754,"teacher_disagreement_score":0.084736034,"about_ca_system_score_codex":0.002613325,"about_ca_system_score_gemma":0.004920682,"threshold_uncertainty_score":0.28347027},"labels":[],"label_agreement":null},{"id":"W4407105429","doi":"10.1075/rs.24009.bot","title":"Review of Seoane &amp; Biber (2021): Corpus-based Approaches to Register Variation","year":2024,"lang":"en","type":"article","venue":"Register Studies","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Register (sociolinguistics); Variation (astronomy); Natural language processing; Biology; Computer science; Linguistics; Philosophy; Physics; Astrophysics","score_opus":0.1655578729349414,"score_gpt":0.3473067979953762,"score_spread":0.18174892506043483,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407105429","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0040589683,0.89838153,0.04393573,0.014604069,0.005317011,0.00021707072,0.0039581195,0.00027914034,0.029248342],"genre_scores_gemma":[0.03697805,0.86213356,0.063680045,0.007378565,0.0027253532,0.0006729273,0.006163382,0.0004405041,0.019827584],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99776566,0.0008622256,0.00032995085,0.00032714565,0.00065597665,0.000059006717],"domain_scores_gemma":[0.98704475,0.008913795,0.00050031743,0.0008253203,0.0025306384,0.00018523895],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0050626784,0.0007878672,0.0013696834,0.013450662,0.0014722252,0.0029292013,0.0016270602,0.001322654,0.008436013],"category_scores_gemma":[0.02146262,0.00050354097,0.0004652193,0.015271171,0.0021611236,0.0039597303,0.0027059782,0.0013933768,0.0034714625],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000075729855,0.00003652639,0.0012281811,0.011778332,0.000118860946,0.00017798359,0.0017654529,0.00041085453,0.0015103328,0.034056433,0.09634912,0.8524922],"study_design_scores_gemma":[0.0000068873683,0.000028404424,0.002953942,0.0056376895,0.00015230953,0.00030263688,0.0006218585,0.00021633059,0.0010854087,0.0065776696,0.9823861,0.00003069499],"about_ca_topic_score_codex":0.014121863,"about_ca_topic_score_gemma":0.035213225,"teacher_disagreement_score":0.014121863,"about_ca_system_score_codex":0.0017788826,"about_ca_system_score_gemma":0.0048886915,"threshold_uncertainty_score":0.02822131},"labels":[],"label_agreement":null},{"id":"W4407130178","doi":"10.1109/smap63474.2024.00028","title":"MMREC: LLM Based Multi-Modal Recommender System","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Recommender system; Modal; Computer science; Information retrieval","score_opus":0.02116551148508732,"score_gpt":0.290798358730528,"score_spread":0.2696328472454407,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407130178","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025594091,0.0016017071,0.9577745,0.0011127697,0.00025921117,0.00017272292,0.0010420966,0.006898843,0.005544037],"genre_scores_gemma":[0.5555111,0.00091013266,0.42046583,0.0015299867,0.00025394213,0.00033377804,0.0025586681,0.00022880819,0.018207714],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99898595,0.00031050266,0.000056088815,0.0002845604,0.00026952891,0.000093390336],"domain_scores_gemma":[0.9989742,0.00035505986,0.00006982054,0.00022037214,0.00031821197,0.00006234856],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012109412,0.0007175978,0.001360159,0.0008817145,0.00080545165,0.0010150544,0.0020420689,0.0015573744,0.004633589],"category_scores_gemma":[0.002933247,0.00037983185,0.0010023043,0.0009420256,0.00039438347,0.001678701,0.0013259661,0.0016888679,0.0027236692],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008692931,0.0007389924,0.0061978553,0.00049352925,0.00050589733,0.0006012947,0.00036000297,0.19256395,0.027352165,0.017365595,0.057775144,0.6951763],"study_design_scores_gemma":[0.000029764029,0.00009459764,0.0005498176,0.000013603178,0.000038544742,0.00015023244,0.000028209108,0.98912174,0.0019712956,0.0034826377,0.0044844747,0.0000350385],"about_ca_topic_score_codex":0.017237704,"about_ca_topic_score_gemma":0.034772087,"teacher_disagreement_score":0.017237704,"about_ca_system_score_codex":0.00089334603,"about_ca_system_score_gemma":0.0009911305,"threshold_uncertainty_score":0.034274697},"labels":[],"label_agreement":null},{"id":"W4407184660","doi":"10.48550/arxiv.2502.01657","title":"Improving Rule-based Reasoning in LLMs using Neurosymbolic Representations","year":2025,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Air Force Office of Scientific Research; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Rule-based system; Computer science; Artificial intelligence","score_opus":0.05274446377624193,"score_gpt":0.23389251196064667,"score_spread":0.18114804818440475,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407184660","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02174363,0.00031965788,0.9670382,0.0005496584,0.00008260722,0.00009229627,0.00041428668,0.00723093,0.002528591],"genre_scores_gemma":[0.33524525,0.00035046646,0.65783584,0.00049953785,0.00009040318,0.00022451543,0.0012107623,0.000625154,0.003918101],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99918455,0.000236155,0.00006554248,0.00026472379,0.000199244,0.000049805585],"domain_scores_gemma":[0.9976326,0.0014179378,0.00022097715,0.00041475517,0.0002304927,0.00008324433],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011039891,0.0010634622,0.00071427505,0.0010000413,0.00046021902,0.0019278921,0.0018256579,0.00092435646,0.005099668],"category_scores_gemma":[0.008459486,0.00047947405,0.001407388,0.0006853463,0.0010520116,0.0032923832,0.0017547322,0.0021319715,0.0017402157],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029447977,0.00026107134,0.002393518,0.0005052941,0.00019164906,0.00033269075,0.00038961368,0.2683069,0.023360396,0.046426427,0.0088044,0.64873356],"study_design_scores_gemma":[0.000022920067,0.000035731482,0.00017728297,0.000027107639,0.000023047392,0.000048578408,0.00003434117,0.92322093,0.008467612,0.06554753,0.0023809087,0.000013912705],"about_ca_topic_score_codex":0.0031084307,"about_ca_topic_score_gemma":0.0074328007,"teacher_disagreement_score":0.005099668,"about_ca_system_score_codex":0.0011965978,"about_ca_system_score_gemma":0.0016072203,"threshold_uncertainty_score":0.017060041},"labels":[],"label_agreement":null},{"id":"W4407264829","doi":"10.29173/cais1997","title":"Writing Practice in LIS","year":2025,"lang":"en","type":"article","venue":"Proceedings of the Annual Conference of CAIS / Actes du congrès annuel de l ACSI","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"University of Toronto; Western University","funders":"","keywords":"Library science; Computer science","score_opus":0.012983590035642738,"score_gpt":0.27969711354758986,"score_spread":0.2667135235119471,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407264829","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.51074696,0.008328443,0.037064087,0.03826555,0.00042428845,0.00026582723,0.00027184075,0.0024752587,0.4021577],"genre_scores_gemma":[0.9326154,0.0029339576,0.017980093,0.0011166665,0.00008922293,0.0001406788,0.00013633602,0.00015148093,0.04483611],"study_design_codex":"qualitative","study_design_gemma":"observational","domain_scores_codex":[0.98214793,0.013692972,0.000789015,0.0010539696,0.0016919661,0.0006241779],"domain_scores_gemma":[0.9643537,0.02504164,0.0026405489,0.0028635797,0.0028892527,0.0022112355],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.020081002,0.000379838,0.00046570777,0.0025556996,0.0111551955,0.011351323,0.0016097786,0.0020438605,0.01311158],"category_scores_gemma":[0.047647662,0.00033566312,0.00027380206,0.003609645,0.009539811,0.005989561,0.0069305203,0.0019502967,0.0045718257],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019566051,0.00026760792,0.012664994,0.00096359843,0.000022742744,0.0014311548,0.5614356,0.0012250295,0.0019764367,0.057895366,0.021273391,0.34064847],"study_design_scores_gemma":[0.000061863444,0.00041883613,0.016204966,0.0024237733,0.0000252674,0.0019121681,0.38263696,0.0039812056,0.004151365,0.053390168,0.53469217,0.000101297046],"about_ca_topic_score_codex":0.00372804,"about_ca_topic_score_gemma":0.0061431904,"teacher_disagreement_score":0.020081002,"about_ca_system_score_codex":0.00809512,"about_ca_system_score_gemma":0.009131214,"threshold_uncertainty_score":0.10619974},"labels":[],"label_agreement":null},{"id":"W4407362568","doi":"10.1109/rsp64122.2024.10870993","title":"Advancing Formal Verification: Fine-Tuning LLMs for Translating Natural Language Requirements to CTL Specifications","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Mitacs","keywords":"Computer science; CTL*; Natural language; Programming language; Formal methods; Formal verification; Natural language processing; Chemistry","score_opus":0.03252903741174228,"score_gpt":0.32889658355676027,"score_spread":0.296367546145018,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407362568","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12912108,0.0010988055,0.7781091,0.0014951656,0.00029759924,0.0006429419,0.003648976,0.07815335,0.0074329544],"genre_scores_gemma":[0.35792038,0.00033352958,0.6233119,0.0007548517,0.000037728943,0.00047141846,0.011195292,0.003646131,0.0023287772],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99441355,0.0027809865,0.00052611984,0.0010193762,0.0010544593,0.00020559714],"domain_scores_gemma":[0.9818512,0.010516229,0.0007849963,0.004853334,0.0017048871,0.0002892952],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0058154063,0.0012469479,0.00057369616,0.0016491015,0.00060428045,0.0020023077,0.0020315198,0.0013212198,0.003962184],"category_scores_gemma":[0.036393918,0.00051034277,0.0015315112,0.00091497146,0.0009704826,0.0037666017,0.0031605947,0.0018428374,0.0024852834],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009976387,0.0008401212,0.01554317,0.0026856458,0.0002976465,0.0007657788,0.002063662,0.15722936,0.06926198,0.02459454,0.043709416,0.6820111],"study_design_scores_gemma":[0.00019119073,0.00035404,0.0019724465,0.00021995344,0.00009084491,0.00038632896,0.00070514967,0.8861585,0.04580426,0.02738575,0.036653236,0.00007828976],"about_ca_topic_score_codex":0.0063789813,"about_ca_topic_score_gemma":0.012845051,"teacher_disagreement_score":0.0063789813,"about_ca_system_score_codex":0.0015111385,"about_ca_system_score_gemma":0.0035626716,"threshold_uncertainty_score":0.030755162},"labels":[],"label_agreement":null},{"id":"W4407572083","doi":"10.52041/iase2023.606","title":"Pedagogical considerations for teaching data translators","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University; Occupational Cancer Research Centre","funders":"","keywords":"Computer science; Mathematics education; Psychology","score_opus":0.16509057597244955,"score_gpt":0.4195190043006851,"score_spread":0.25442842832823553,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407572083","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0232878,0.0026916254,0.6758227,0.17615749,0.0024266918,0.0007249961,0.00017158267,0.0041066995,0.11461042],"genre_scores_gemma":[0.28178757,0.0029378827,0.6554932,0.013564172,0.0014234068,0.0014922959,0.00019213793,0.0010176967,0.04209162],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9869775,0.008626791,0.00047718026,0.0009417736,0.002160915,0.0008158039],"domain_scores_gemma":[0.9363615,0.045081932,0.0019215745,0.0029904826,0.008303438,0.0053410474],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.015877912,0.00073494366,0.0006015486,0.0012886432,0.00404536,0.008573176,0.0031721971,0.004588272,0.028082129],"category_scores_gemma":[0.06617095,0.0009058525,0.0005369257,0.0009085758,0.0023082625,0.010594358,0.0056756856,0.008884924,0.009352214],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00053911394,0.0022631707,0.0039319233,0.0012569502,0.000027684317,0.0011071736,0.017925303,0.0029788392,0.013459559,0.41118148,0.10521748,0.4401113],"study_design_scores_gemma":[0.00032862945,0.00041210977,0.0018342866,0.0016096436,0.00005607082,0.0014826447,0.01797698,0.019989371,0.016622476,0.5250741,0.41449854,0.00011510089],"about_ca_topic_score_codex":0.0020041857,"about_ca_topic_score_gemma":0.005675263,"teacher_disagreement_score":0.028082129,"about_ca_system_score_codex":0.0036714452,"about_ca_system_score_gemma":0.008486002,"threshold_uncertainty_score":0.09394407},"labels":[],"label_agreement":null},{"id":"W4407580437","doi":"10.1145/3717061","title":"An Empirical Study of Retrieval-Augmented Code Generation: Challenges and Opportunities","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Code (set theory); Data science; Information retrieval; Programming language","score_opus":0.20914820288916802,"score_gpt":0.39155547964417226,"score_spread":0.18240727675500423,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407580437","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.95658207,0.003781275,0.027922008,0.0013471126,0.000153639,0.00042596567,0.0021728827,0.0016086135,0.006006331],"genre_scores_gemma":[0.96812695,0.0008456453,0.022531573,0.00036852618,0.000057254514,0.00034915557,0.0057207197,0.0004035268,0.0015967109],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9808797,0.010143108,0.0013409278,0.0026569571,0.0045164954,0.00046276787],"domain_scores_gemma":[0.785157,0.17002147,0.010213731,0.019410731,0.013425196,0.0017718404],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.021476127,0.0010584434,0.00060903176,0.0021253352,0.000996602,0.0022297194,0.0018870052,0.0014627165,0.0025550772],"category_scores_gemma":[0.17397694,0.00067466294,0.00083370804,0.0022852167,0.001963749,0.008249703,0.0024328157,0.0030750628,0.0012688381],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002260276,0.002491333,0.42510787,0.0037320955,0.0005470207,0.0011239925,0.008799589,0.030044839,0.009897376,0.0075148814,0.029920008,0.47856075],"study_design_scores_gemma":[0.0007413123,0.0062406515,0.4096999,0.0019310837,0.0007746839,0.004658395,0.014853172,0.4106149,0.031851828,0.01888646,0.09926211,0.0004855858],"about_ca_topic_score_codex":0.003689805,"about_ca_topic_score_gemma":0.004335057,"teacher_disagreement_score":0.021476127,"about_ca_system_score_codex":0.001069858,"about_ca_system_score_gemma":0.0014557012,"threshold_uncertainty_score":0.11357802},"labels":[],"label_agreement":null},{"id":"W4407590589","doi":"10.48550/arxiv.2502.09589","title":"Logical forms complement probability in understanding language model (and human) performance","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Complement (music); Computer science; Logical conjunction; Natural language processing; Programming language; Chemistry","score_opus":0.1211045059254114,"score_gpt":0.33799592959226704,"score_spread":0.21689142366685565,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407590589","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7431851,0.0020854422,0.22802225,0.0027485471,0.00010081937,0.00020565094,0.003558177,0.0019189671,0.018175054],"genre_scores_gemma":[0.9475165,0.00037891147,0.048054647,0.00023864827,0.000046769135,0.00008602226,0.002964741,0.00018639109,0.0005273308],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9922414,0.004593046,0.0004359899,0.0015567775,0.0009761191,0.00019674333],"domain_scores_gemma":[0.8829156,0.09794862,0.0059336494,0.009970981,0.002099101,0.0011319644],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010532664,0.00090170273,0.00048614165,0.0021552683,0.0005764958,0.0050959126,0.0009767156,0.0013303352,0.0047427276],"category_scores_gemma":[0.095273316,0.0005808253,0.0008801331,0.0014551558,0.0022773868,0.011212458,0.0026037449,0.0022181505,0.0012318505],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021474476,0.0011390658,0.3505294,0.0018518686,0.0011028877,0.00044257473,0.013069919,0.09607648,0.017137337,0.06623803,0.011585856,0.43867922],"study_design_scores_gemma":[0.00015827893,0.0009525608,0.12764114,0.00030375633,0.00028383386,0.00074069644,0.0036717416,0.5478312,0.011049179,0.28756303,0.019511886,0.0002927627],"about_ca_topic_score_codex":0.0046092137,"about_ca_topic_score_gemma":0.0044095987,"teacher_disagreement_score":0.010532664,"about_ca_system_score_codex":0.0010118213,"about_ca_system_score_gemma":0.0009044782,"threshold_uncertainty_score":0.055702686},"labels":[],"label_agreement":null},{"id":"W4407719212","doi":"10.1007/978-981-96-2292-4_7","title":"Joint Multi-modal Modeling for Speech-to-Text Translation as Multilingual Neural Machine Translation","year":2025,"lang":"en","type":"book-chapter","venue":"Communications in computer and information science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Computer science; Machine translation; Translation (biology); Natural language processing; Joint (building); Modal; Artificial intelligence; Speech recognition; Engineering; Biology; Chemistry; Messenger RNA; Structural engineering","score_opus":0.07970553934307349,"score_gpt":0.3535581530429059,"score_spread":0.2738526136998324,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407719212","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004554517,0.001130586,0.9885351,0.00028646245,0.00014786774,0.000019024103,0.0002681991,0.0015193927,0.0035388155],"genre_scores_gemma":[0.3802156,0.0028530809,0.57988995,0.0003891125,0.00038781198,0.00024859217,0.0026189706,0.0012482185,0.03214863],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99948585,0.0001946059,0.000038405527,0.00012884394,0.000106359286,0.000045851786],"domain_scores_gemma":[0.9993099,0.0003735567,0.000048494414,0.000111123125,0.00013600764,0.000020921034],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00077326,0.00085672963,0.0009540701,0.00053423183,0.0005079427,0.0015055692,0.0011618033,0.0012579395,0.0058990745],"category_scores_gemma":[0.001773357,0.0005328905,0.0015495886,0.001259672,0.0004313446,0.0018604995,0.0012908366,0.0015551301,0.0038875202],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00031170514,0.00019269373,0.00046203585,0.00043191225,0.0002567102,0.0003241352,0.0002503808,0.42129782,0.023457622,0.07657653,0.015633091,0.46080533],"study_design_scores_gemma":[0.0000026927244,0.00001826962,0.00009120622,0.000010348446,0.000021510641,0.00005284776,0.00001556134,0.97153556,0.0033060599,0.021762263,0.0031702688,0.000013445838],"about_ca_topic_score_codex":0.0040729456,"about_ca_topic_score_gemma":0.0074883276,"teacher_disagreement_score":0.0058990745,"about_ca_system_score_codex":0.00068863237,"about_ca_system_score_gemma":0.0007247588,"threshold_uncertainty_score":0.019734383},"labels":[],"label_agreement":null},{"id":"W4407770995","doi":"10.1145/3641554.3701934","title":"Exploring Student Reactions to LLM-Generated Feedback on Explain in Plain English Problems","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Plain English; Computer science; Mathematics education; Psychology; Linguistics; Philosophy","score_opus":0.05100495558707917,"score_gpt":0.2998898895530943,"score_spread":0.24888493396601513,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407770995","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9969188,0.000024161372,0.0023160193,0.000070331545,0.000008063024,0.000052420364,0.000054662694,0.00021332252,0.00034221445],"genre_scores_gemma":[0.99394965,0.000044520984,0.0046668206,0.0000865712,0.000010038581,0.000116026844,0.00016201325,0.000068806854,0.0008955349],"study_design_codex":"observational","study_design_gemma":"qualitative","domain_scores_codex":[0.9899065,0.0053249025,0.00071713777,0.0012457297,0.0022491436,0.00055661844],"domain_scores_gemma":[0.8499307,0.11433831,0.016281204,0.006245063,0.010448045,0.0027565472],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0068621403,0.0012009775,0.00083849864,0.0016265484,0.0005228629,0.0017762765,0.001399849,0.0017613112,0.0018899163],"category_scores_gemma":[0.11025841,0.00045041688,0.00044958753,0.0006514728,0.000899012,0.0012170533,0.0019645456,0.0017366267,0.00090978685],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0034910375,0.003800024,0.47091883,0.0012855273,0.00034507047,0.003556268,0.1553638,0.007218052,0.098278254,0.0005323306,0.0035253926,0.25168532],"study_design_scores_gemma":[0.0002743121,0.016243655,0.70462126,0.0007627455,0.00033500444,0.003782703,0.0711055,0.06542574,0.12146505,0.0023429282,0.012947776,0.0006933075],"about_ca_topic_score_codex":0.00060431357,"about_ca_topic_score_gemma":0.0008923617,"teacher_disagreement_score":0.0068621403,"about_ca_system_score_codex":0.00078324054,"about_ca_system_score_gemma":0.0004299548,"threshold_uncertainty_score":0.036290884},"labels":[],"label_agreement":null},{"id":"W4407780942","doi":"10.1098/rsos.241091/v1/review1","title":"Review for \"Projected speaker numbers and dormancy risks of Canada’s Indigenous languages\"","year":2024,"lang":"en","type":"peer-review","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Indigenous; Dormancy; Linguistics; Computer science; Geography; Political science; Environmental science; History; Biology; Agronomy; Philosophy; Ecology","score_opus":0.037864424919380746,"score_gpt":0.3635824775299149,"score_spread":0.32571805261053416,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407780942","genre_codex":"commentary","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005180913,0.34999812,0.0034339484,0.3596867,0.106245704,0.00062271056,0.024743846,0.0006767709,0.14941128],"genre_scores_gemma":[0.07219655,0.43058807,0.007901427,0.11634802,0.029163755,0.0010742876,0.01993524,0.00070197036,0.32209072],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99596715,0.00041419468,0.0002643063,0.00025318775,0.0026429626,0.00045819196],"domain_scores_gemma":[0.9118097,0.0074953926,0.0014843446,0.0010294489,0.07572513,0.002456001],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0070153186,0.0005095182,0.00075453543,0.00515967,0.0025492788,0.0032012796,0.0025895822,0.0019317452,0.022218332],"category_scores_gemma":[0.040486112,0.00032176063,0.00083034096,0.003906335,0.0018764365,0.0013792121,0.0012390197,0.0023715396,0.0051814844],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000037957772,0.0000058448213,0.00038841562,0.00093739084,0.00002688741,0.000028398981,0.0001115673,0.000044101722,0.00009203218,0.0026721319,0.9301374,0.06551786],"study_design_scores_gemma":[0.000015319123,0.000012856126,0.004900965,0.002533165,0.00007977508,0.000034444765,0.00018423113,0.000041853542,0.00015724856,0.0005096535,0.9915133,0.000017208107],"about_ca_topic_score_codex":0.778413,"about_ca_topic_score_gemma":0.849023,"teacher_disagreement_score":0.221587,"about_ca_system_score_codex":0.01627937,"about_ca_system_score_gemma":0.093598925,"threshold_uncertainty_score":0.4457839},"labels":[],"label_agreement":null},{"id":"W4407780943","doi":"10.1098/rsos.241091/v1/review2","title":"Review for \"Projected speaker numbers and dormancy risks of Canada’s Indigenous languages\"","year":2024,"lang":"en","type":"peer-review","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Indigenous; Dormancy; Linguistics; History; Geography; Computer science; Biology; Agronomy; Philosophy; Ecology","score_opus":0.037864424919380746,"score_gpt":0.3635824775299149,"score_spread":0.32571805261053416,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407780943","genre_codex":"commentary","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005180913,0.34999812,0.0034339484,0.3596867,0.106245704,0.00062271056,0.024743846,0.0006767709,0.14941128],"genre_scores_gemma":[0.07219655,0.43058807,0.007901427,0.11634802,0.029163755,0.0010742876,0.01993524,0.00070197036,0.32209072],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99596715,0.00041419468,0.0002643063,0.00025318775,0.0026429626,0.00045819196],"domain_scores_gemma":[0.9118097,0.0074953926,0.0014843446,0.0010294489,0.07572513,0.002456001],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0070153186,0.0005095182,0.00075453543,0.00515967,0.0025492788,0.0032012796,0.0025895822,0.0019317452,0.022218332],"category_scores_gemma":[0.040486112,0.00032176063,0.00083034096,0.003906335,0.0018764365,0.0013792121,0.0012390197,0.0023715396,0.0051814844],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000037957772,0.0000058448213,0.00038841562,0.00093739084,0.00002688741,0.000028398981,0.0001115673,0.000044101722,0.00009203218,0.0026721319,0.9301374,0.06551786],"study_design_scores_gemma":[0.000015319123,0.000012856126,0.004900965,0.002533165,0.00007977508,0.000034444765,0.00018423113,0.000041853542,0.00015724856,0.0005096535,0.9915133,0.000017208107],"about_ca_topic_score_codex":0.778413,"about_ca_topic_score_gemma":0.849023,"teacher_disagreement_score":0.221587,"about_ca_system_score_codex":0.01627937,"about_ca_system_score_gemma":0.093598925,"threshold_uncertainty_score":0.4457839},"labels":[],"label_agreement":null},{"id":"W4407781371","doi":"10.1098/rsos.241091/v2/response1","title":"Author response for \"Projected speaker numbers and dormancy risks of Canada’s Indigenous languages\"","year":2024,"lang":"en","type":"peer-review","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Indigenous; Dormancy; Linguistics; History; Computer science; Geography; Biology; Botany; Ecology; Philosophy","score_opus":0.044937253625454386,"score_gpt":0.3772229466575765,"score_spread":0.33228569303212213,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407781371","genre_codex":"commentary","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0059618857,0.00077325513,0.000953356,0.78524923,0.13169314,0.0005233853,0.015845327,0.0012358057,0.05776453],"genre_scores_gemma":[0.029264925,0.000892853,0.0010569278,0.27231288,0.015561843,0.00082770624,0.0045607234,0.00060055184,0.6749216],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9954796,0.00045039697,0.0002926139,0.00036119416,0.0025098089,0.0009063839],"domain_scores_gemma":[0.958032,0.008440744,0.00092242,0.0009243,0.02884079,0.0028397068],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00456702,0.00078653905,0.00093733205,0.0017232922,0.006387062,0.004077069,0.0015835416,0.010632443,0.13260013],"category_scores_gemma":[0.041352045,0.000460916,0.0006371897,0.0012693162,0.0014814368,0.001238063,0.0025926584,0.005997862,0.036721986],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000027486712,0.000003234974,0.00017300433,0.000019982463,0.0000015224946,0.000040691626,0.000103995655,0.000016211268,0.000060377275,0.00022989401,0.99828905,0.0010345023],"study_design_scores_gemma":[0.00005471529,0.000022326583,0.004267655,0.00015901729,0.000011972678,0.0000665712,0.0024492745,0.0002648483,0.00040041676,0.0003213604,0.9919309,0.000050939354],"about_ca_topic_score_codex":0.28121558,"about_ca_topic_score_gemma":0.39034218,"teacher_disagreement_score":0.71878445,"about_ca_system_score_codex":0.010731335,"about_ca_system_score_gemma":0.02112459,"threshold_uncertainty_score":0.55915743},"labels":[],"label_agreement":null},{"id":"W4407781428","doi":"10.1098/rsos.241091/v2/review1","title":"Review for \"Projected speaker numbers and dormancy risks of Canada’s Indigenous languages\"","year":2024,"lang":"en","type":"peer-review","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Indigenous; Dormancy; Linguistics; History; Geography; Political science; Biology; Agronomy; Philosophy; Ecology","score_opus":0.037864424919380746,"score_gpt":0.3635824775299149,"score_spread":0.32571805261053416,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407781428","genre_codex":"commentary","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005180913,0.34999812,0.0034339484,0.3596867,0.106245704,0.00062271056,0.024743846,0.0006767709,0.14941128],"genre_scores_gemma":[0.07219655,0.43058807,0.007901427,0.11634802,0.029163755,0.0010742876,0.01993524,0.00070197036,0.32209072],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99596715,0.00041419468,0.0002643063,0.00025318775,0.0026429626,0.00045819196],"domain_scores_gemma":[0.9118097,0.0074953926,0.0014843446,0.0010294489,0.07572513,0.002456001],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0070153186,0.0005095182,0.00075453543,0.00515967,0.0025492788,0.0032012796,0.0025895822,0.0019317452,0.022218332],"category_scores_gemma":[0.040486112,0.00032176063,0.00083034096,0.003906335,0.0018764365,0.0013792121,0.0012390197,0.0023715396,0.0051814844],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000037957772,0.0000058448213,0.00038841562,0.00093739084,0.00002688741,0.000028398981,0.0001115673,0.000044101722,0.00009203218,0.0026721319,0.9301374,0.06551786],"study_design_scores_gemma":[0.000015319123,0.000012856126,0.004900965,0.002533165,0.00007977508,0.000034444765,0.00018423113,0.000041853542,0.00015724856,0.0005096535,0.9915133,0.000017208107],"about_ca_topic_score_codex":0.778413,"about_ca_topic_score_gemma":0.849023,"teacher_disagreement_score":0.221587,"about_ca_system_score_codex":0.01627937,"about_ca_system_score_gemma":0.093598925,"threshold_uncertainty_score":0.4457839},"labels":[],"label_agreement":null},{"id":"W4407782326","doi":"10.1098/rsos.241091/v2/review2","title":"Review for \"Projected speaker numbers and dormancy risks of Canada’s Indigenous languages\"","year":2024,"lang":"en","type":"peer-review","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Indigenous; Dormancy; Linguistics; Geography; History; Biology; Botany; Ecology; Philosophy","score_opus":0.037864424919380746,"score_gpt":0.3635824775299149,"score_spread":0.32571805261053416,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407782326","genre_codex":"commentary","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005180913,0.34999812,0.0034339484,0.3596867,0.106245704,0.00062271056,0.024743846,0.0006767709,0.14941128],"genre_scores_gemma":[0.07219655,0.43058807,0.007901427,0.11634802,0.029163755,0.0010742876,0.01993524,0.00070197036,0.32209072],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99596715,0.00041419468,0.0002643063,0.00025318775,0.0026429626,0.00045819196],"domain_scores_gemma":[0.9118097,0.0074953926,0.0014843446,0.0010294489,0.07572513,0.002456001],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0070153186,0.0005095182,0.00075453543,0.00515967,0.0025492788,0.0032012796,0.0025895822,0.0019317452,0.022218332],"category_scores_gemma":[0.040486112,0.00032176063,0.00083034096,0.003906335,0.0018764365,0.0013792121,0.0012390197,0.0023715396,0.0051814844],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000037957772,0.0000058448213,0.00038841562,0.00093739084,0.00002688741,0.000028398981,0.0001115673,0.000044101722,0.00009203218,0.0026721319,0.9301374,0.06551786],"study_design_scores_gemma":[0.000015319123,0.000012856126,0.004900965,0.002533165,0.00007977508,0.000034444765,0.00018423113,0.000041853542,0.00015724856,0.0005096535,0.9915133,0.000017208107],"about_ca_topic_score_codex":0.778413,"about_ca_topic_score_gemma":0.849023,"teacher_disagreement_score":0.221587,"about_ca_system_score_codex":0.01627937,"about_ca_system_score_gemma":0.093598925,"threshold_uncertainty_score":0.4457839},"labels":[],"label_agreement":null},{"id":"W4407857946","doi":"10.1145/3696443.3708957","title":"DialEgg: Dialect-Agnostic MLIR Optimizer using Equality Saturation with Egglog","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Mila - Quebec Artificial Intelligence Institute; McGill University","funders":"","keywords":"Saturation (graph theory); Computer science; Mathematics; Combinatorics","score_opus":0.016994546623230387,"score_gpt":0.3004577194410816,"score_spread":0.2834631728178512,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407857946","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012036879,0.00027856833,0.8117045,0.00032405794,0.0001624495,0.00017624414,0.0010022222,0.16471823,0.009596897],"genre_scores_gemma":[0.25148502,0.0004722825,0.64345413,0.0018967012,0.000114156865,0.00049177895,0.004429816,0.077411056,0.020245045],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985721,0.0002954588,0.00014891112,0.00033105817,0.0005085759,0.00014391124],"domain_scores_gemma":[0.997775,0.0011347323,0.000118465076,0.0006581886,0.00024941028,0.00006411901],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017120921,0.0009631233,0.0005545526,0.00074456306,0.00042056738,0.0016670021,0.0019735696,0.0008975818,0.011242308],"category_scores_gemma":[0.0053729406,0.0009544316,0.0011741107,0.00035068239,0.0012173343,0.0033810537,0.0033412266,0.002454953,0.0057385764],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001307184,0.0005552192,0.011490708,0.0021873533,0.0002837232,0.0016681623,0.00284142,0.028693862,0.13002124,0.16253777,0.12679867,0.5316147],"study_design_scores_gemma":[0.00027877392,0.0003358739,0.0027019838,0.00034883982,0.00019964232,0.0016097428,0.0004932386,0.27028263,0.24201,0.108781636,0.37267047,0.00028717832],"about_ca_topic_score_codex":0.0014750339,"about_ca_topic_score_gemma":0.0029068678,"teacher_disagreement_score":0.011242308,"about_ca_system_score_codex":0.00080288306,"about_ca_system_score_gemma":0.0012144233,"threshold_uncertainty_score":0.03760922},"labels":[],"label_agreement":null},{"id":"W4407944201","doi":"10.3390/app15052476","title":"A Small-Scale Evaluation of Large Language Models Used for Grammatical Error Correction in a German Children’s Literature Corpus: A Comparative Study","year":2025,"lang":"en","type":"article","venue":"Applied Sciences","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"German; Computer science; Linguistics; Natural language processing; Artificial intelligence; Philosophy","score_opus":0.04146972705040574,"score_gpt":0.3714450331222936,"score_spread":0.32997530607188785,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407944201","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.92767596,0.006126051,0.03920004,0.0005138405,0.00042201456,0.0007696268,0.0063148253,0.012854753,0.006122862],"genre_scores_gemma":[0.8728984,0.0021677571,0.090587705,0.00031564073,0.000090392925,0.00075374363,0.027466604,0.0012051585,0.0045145266],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.996627,0.0016195544,0.00033710338,0.0007371248,0.000560744,0.00011856428],"domain_scores_gemma":[0.9865181,0.009260375,0.0004168204,0.0015914532,0.0018811261,0.00033208236],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004951828,0.0019500804,0.0009059155,0.0028873878,0.0005928888,0.0015565857,0.0020946397,0.0013112499,0.0022494467],"category_scores_gemma":[0.017797986,0.0007430557,0.001093981,0.0013891588,0.0008223539,0.0025878425,0.0016964329,0.0015669612,0.0018899675],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020487814,0.0018096373,0.035250667,0.004195768,0.0017263546,0.0017871357,0.004672378,0.13078171,0.046279684,0.0024256716,0.030744065,0.7382782],"study_design_scores_gemma":[0.000658012,0.0026694792,0.055518445,0.00077176304,0.0014337695,0.002056372,0.004335125,0.80821216,0.078701094,0.0016327522,0.043642953,0.0003681345],"about_ca_topic_score_codex":0.0289025,"about_ca_topic_score_gemma":0.039402116,"teacher_disagreement_score":0.0289025,"about_ca_system_score_codex":0.0019255907,"about_ca_system_score_gemma":0.0019343431,"threshold_uncertainty_score":0.057468534},"labels":[],"label_agreement":null},{"id":"W4408062977","doi":"10.5220/0013261100003905","title":"Zeroth Order Optimization for Pretraining Language Models","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; Group for Research in Decision Analysis","funders":"","keywords":"Computer science; Order (exchange); Zeroth law of thermodynamics; Artificial intelligence; Natural language processing; Physics; Economics","score_opus":0.013470139288254095,"score_gpt":0.29340706085176305,"score_spread":0.27993692156350897,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4408062977","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023164427,0.00034479823,0.97242075,0.00037628144,0.0000730789,0.0000408194,0.00007722094,0.001431885,0.0020707783],"genre_scores_gemma":[0.4280819,0.00042235732,0.55878127,0.0006809334,0.00009631385,0.00025712256,0.0006010252,0.001141572,0.009937478],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993413,0.00022636041,0.00004100646,0.00013203523,0.00016236817,0.000096937634],"domain_scores_gemma":[0.9979171,0.0015151545,0.00009028239,0.00018268843,0.00022768619,0.0000670948],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010990724,0.0010205718,0.0007775942,0.0005460227,0.00045901682,0.0010057162,0.0011968509,0.0012095268,0.004984705],"category_scores_gemma":[0.0054505155,0.0006768752,0.0006864148,0.0004995065,0.0012809522,0.0019328329,0.0013749243,0.002698543,0.0015853817],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025679747,0.00013203757,0.00086522324,0.00029464278,0.00007246041,0.00010549807,0.00022595139,0.7419505,0.01199409,0.034328803,0.0054898127,0.2042843],"study_design_scores_gemma":[0.000005624923,0.000017995875,0.00005147059,0.0000075403996,0.000004058212,0.000009604502,0.00000849044,0.9914682,0.0019737722,0.00596565,0.0004833481,0.000004223525],"about_ca_topic_score_codex":0.00593728,"about_ca_topic_score_gemma":0.0095149195,"teacher_disagreement_score":0.00593728,"about_ca_system_score_codex":0.0012330943,"about_ca_system_score_gemma":0.0022976275,"threshold_uncertainty_score":0.016675532},"labels":[],"label_agreement":null},{"id":"W4408155921","doi":"10.1163/18773109-01602002","title":"The cognitive mechanisms involved in the “DEGREE ADVERB + PROPER NAME” construction","year":2024,"lang":"en","type":"article","venue":"International Review of Pragmatics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Adverb; Sociolinguistics; Linguistics; Degree (music); Cognition; Psychology; Philosophy; Noun; Physics","score_opus":0.026975858864692073,"score_gpt":0.31908749145571824,"score_spread":0.2921116325910262,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4408155921","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4591204,0.0007937909,0.4188103,0.0059624123,0.00021923089,0.00014681679,0.00012963582,0.00074400404,0.114073366],"genre_scores_gemma":[0.9757706,0.0001715813,0.022437818,0.00013410051,0.000038782604,0.000026503794,0.000051130635,0.000097446566,0.0012720348],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.99559915,0.002283671,0.00019725515,0.00069185445,0.0010062109,0.0002218673],"domain_scores_gemma":[0.9910739,0.005643636,0.0008222031,0.0014517488,0.00077698217,0.00023151087],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007282735,0.00037418335,0.00041235506,0.0013441541,0.0012737013,0.0045059216,0.0011598041,0.001631618,0.0050918697],"category_scores_gemma":[0.016352726,0.00076440634,0.00071350357,0.0008842866,0.013358146,0.014383613,0.0037763498,0.0024770778,0.00063027523],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006656399,0.000030020112,0.0023575441,0.00016341331,0.00002509138,0.00019084573,0.014113088,0.00048497453,0.010622991,0.95644784,0.0005402087,0.0149574205],"study_design_scores_gemma":[0.000035498346,0.00006448821,0.008814738,0.00012076133,0.000067427914,0.0007523498,0.009089237,0.011675341,0.006428715,0.9444154,0.018458422,0.0000775333],"about_ca_topic_score_codex":0.0012388062,"about_ca_topic_score_gemma":0.00097554375,"teacher_disagreement_score":0.007282735,"about_ca_system_score_codex":0.0010858988,"about_ca_system_score_gemma":0.0010022059,"threshold_uncertainty_score":0.03851521},"labels":[],"label_agreement":null},{"id":"W4408361647","doi":"10.5206/cjils-rcsib.v48i1.19419","title":"Investigating the Use of Plain Language Summaries in Canadian Science Journals","year":2025,"lang":"en","type":"article","venue":"Canadian Journal of Information and Library Science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Plain language; Plain English; Computer science; Library science; Data science; Linguistics","score_opus":0.014081835830692743,"score_gpt":0.24303750817165296,"score_spread":0.2289556723409602,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4408361647","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"reporting","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"reporting","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8667034,0.0072953394,0.0050355676,0.025536397,0.0009387509,0.0018335836,0.006886987,0.00095177104,0.08481827],"genre_scores_gemma":[0.96249515,0.006387446,0.01067281,0.0029712105,0.0003436842,0.0006943399,0.0036367716,0.00021951445,0.012579126],"study_design_codex":"qualitative","study_design_gemma":"observational","domain_scores_codex":[0.8910738,0.032220643,0.01278591,0.0038669396,0.055747475,0.0043052947],"domain_scores_gemma":[0.45118487,0.25863102,0.08400607,0.017962167,0.17363682,0.014578967],"candidate_categories":["metaresearch","scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.049305193,0.0006227339,0.0008721981,0.021841327,0.014601282,0.017921532,0.0035727748,0.0018346013,0.004780537],"category_scores_gemma":[0.45860258,0.0009401965,0.0005477246,0.038501408,0.004880093,0.008760559,0.007737863,0.0022298384,0.0013098171],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":true,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00073410757,0.00035607882,0.16405262,0.0047920663,0.00020783118,0.0016332556,0.53728783,0.0006375608,0.0040231342,0.016827608,0.04696666,0.22248127],"study_design_scores_gemma":[0.00013938101,0.0005849871,0.2854208,0.002804573,0.0003949978,0.0010995268,0.33780983,0.0025320426,0.004682127,0.0030874703,0.3607807,0.0006635382],"about_ca_topic_score_codex":0.71829855,"about_ca_topic_score_gemma":0.7945923,"teacher_disagreement_score":0.9820785,"about_ca_system_score_codex":0.048267286,"about_ca_system_score_gemma":0.108057804,"threshold_uncertainty_score":0.56672084},"labels":[],"label_agreement":null},{"id":"W4408504061","doi":"10.1007/978-3-031-51447-0_258-1","title":"Machine Translation Literacy","year":2025,"lang":"en","type":"book-chapter","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Translation (biology); Literacy; Computer science; Linguistics; Natural language processing; Psychology; Pedagogy; Philosophy; Chemistry","score_opus":0.012243346810040977,"score_gpt":0.2750411579495608,"score_spread":0.26279781113951983,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4408504061","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00093792396,0.0075665554,0.02243108,0.0049417657,0.0012596017,0.000038310038,0.0003206094,0.00066021626,0.961844],"genre_scores_gemma":[0.012789255,0.007370778,0.009129789,0.0015106923,0.00091438874,0.00006430495,0.00073465216,0.0006985294,0.96678746],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9994455,0.00015087677,0.000032681957,0.00010532885,0.00021629747,0.000049440754],"domain_scores_gemma":[0.9992167,0.00041848546,0.000025566585,0.0001441363,0.00015821749,0.00003690394],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00046715324,0.00080631376,0.00067411823,0.0019995472,0.0014829363,0.0050874003,0.0007533151,0.0013021054,0.13075167],"category_scores_gemma":[0.0024071513,0.000501261,0.00032404577,0.0020725352,0.0017319121,0.007570295,0.001984417,0.0027691124,0.06860723],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00001673351,0.00003374812,0.000072121264,0.00027298217,0.0000049041937,0.000091366615,0.00073564606,0.00016158834,0.0008147706,0.36856958,0.3497189,0.27950764],"study_design_scores_gemma":[0.0000020775976,0.00000816037,0.00010292911,0.000172947,0.0000024971312,0.00018221188,0.0001137722,0.0003015852,0.0006458475,0.060400184,0.9380629,0.0000050032177],"about_ca_topic_score_codex":0.0011102527,"about_ca_topic_score_gemma":0.002210564,"teacher_disagreement_score":0.13075167,"about_ca_system_score_codex":0.0012727374,"about_ca_system_score_gemma":0.0011890722,"threshold_uncertainty_score":0.4374079},"labels":[],"label_agreement":null},{"id":"W4408592579","doi":"10.1109/taslpro.2025.3552936","title":"A Multilingual Dataset (MultiMWP) and Benchmark for Math Word Problem Generation","year":2025,"lang":"en","type":"article","venue":"IEEE Transactions on Audio Speech and Language Processing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Benchmark (surveying); Natural language processing; Artificial intelligence; Word (group theory); Speech recognition; Mathematics","score_opus":0.013684582970709907,"score_gpt":0.2964765663066384,"score_spread":0.2827919833359285,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4408592579","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09695839,0.0029073504,0.029295983,0.0025414904,0.001129861,0.0018048258,0.79284996,0.03722206,0.035289995],"genre_scores_gemma":[0.03422175,0.00029483775,0.041184787,0.00041819457,0.000093153554,0.0010965129,0.91722834,0.00084085466,0.004621535],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99789894,0.00060527556,0.00026259306,0.000625466,0.00045421842,0.00015358576],"domain_scores_gemma":[0.99673223,0.0014514369,0.00017331983,0.00075046124,0.0006352776,0.00025715216],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017087562,0.0032281426,0.0009619985,0.0028379639,0.0013361464,0.0015936755,0.0042855944,0.0026362624,0.020215694],"category_scores_gemma":[0.009353977,0.00061797263,0.0022278077,0.0036775582,0.00073312555,0.002802299,0.002164874,0.00282595,0.0145712085],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007421049,0.0012866337,0.0042664646,0.0030423715,0.00030807994,0.0005915219,0.00026568308,0.019610858,0.0035891926,0.0042024683,0.8366671,0.12542742],"study_design_scores_gemma":[0.002832528,0.0011035461,0.019401705,0.0004947373,0.00026509073,0.00150214,0.001374172,0.24962251,0.020855278,0.019548114,0.6827451,0.0002550452],"about_ca_topic_score_codex":0.018023368,"about_ca_topic_score_gemma":0.03807944,"teacher_disagreement_score":0.020215694,"about_ca_system_score_codex":0.0018219366,"about_ca_system_score_gemma":0.0024130025,"threshold_uncertainty_score":0.067628205},"labels":[],"label_agreement":null},{"id":"W4408800806","doi":"10.1016/j.neucom.2025.130042","title":"Unifying the syntax and semantics for math word problem solving","year":2025,"lang":"en","type":"article","venue":"Neurocomputing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Natural Science Foundation for Distinguished Young Scholars of Hunan Province; Central China Normal University; China Postdoctoral Science Foundation; Ministry of Education of the People's Republic of China","keywords":"Syntax; Semantics (computer science); Computer science; Word (group theory); Abstract syntax tree; Natural language processing; Artificial intelligence; Linguistics; Programming language; Mathematics; Algebra over a field; Pure mathematics; Philosophy","score_opus":0.012449409954020493,"score_gpt":0.2720112182723778,"score_spread":0.2595618083183573,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4408800806","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.037232373,0.00049355614,0.94446343,0.0018202714,0.00022859118,0.00012610323,0.0008049061,0.0019341878,0.012896658],"genre_scores_gemma":[0.5439271,0.0004479712,0.44707188,0.0005559474,0.0002238502,0.00025368092,0.0016278392,0.0008158543,0.005075949],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9979747,0.00066308887,0.00031517437,0.0004953908,0.0003288757,0.00022280068],"domain_scores_gemma":[0.99732757,0.0011495968,0.00023765684,0.00068653806,0.0004577165,0.00014084623],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019692397,0.0007290944,0.0010210542,0.0012138451,0.0012381934,0.004448407,0.00195809,0.0014824733,0.0063828565],"category_scores_gemma":[0.00773271,0.0007779533,0.0024197048,0.00124382,0.0033497564,0.013472344,0.0038834822,0.0030596633,0.0013241804],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000103438935,0.000060333557,0.0006864372,0.00016394684,0.000027345668,0.000068376394,0.0008618387,0.0046918103,0.002329798,0.9286314,0.0022546265,0.06012063],"study_design_scores_gemma":[0.000015320236,0.000020044643,0.00025112368,0.000026845702,0.000020624304,0.000042242922,0.000136226,0.029200722,0.001068207,0.96544003,0.0037595215,0.000019029538],"about_ca_topic_score_codex":0.0045039267,"about_ca_topic_score_gemma":0.005583641,"teacher_disagreement_score":0.0063828565,"about_ca_system_score_codex":0.0012837045,"about_ca_system_score_gemma":0.0026189,"threshold_uncertainty_score":0.021352768},"labels":[],"label_agreement":null},{"id":"W4408803666","doi":"10.63485/fckzx-qt804","title":"John Hoey and Richard Smith on OA","year":2007,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Management; Economics; Law and economics","score_opus":0.017981914093297056,"score_gpt":0.28985458535930275,"score_spread":0.27187267126600567,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4408803666","genre_codex":"commentary","genre_gemma":"commentary","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"commentary","genre_consensus":"commentary","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0020166636,0.05488891,0.0054189297,0.54233354,0.042934302,0.000033084903,0.00024070687,0.00026706755,0.35186687],"genre_scores_gemma":[0.030190028,0.029407322,0.002519161,0.066360526,0.016314708,0.000033042936,0.00020852912,0.0006977318,0.85426897],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9970009,0.00072486425,0.00010743507,0.0003757336,0.0013500204,0.00044105883],"domain_scores_gemma":[0.99433774,0.002135919,0.00016930616,0.00045083338,0.0016590352,0.0012472073],"candidate_categories":["scholarly_communication","open_science"],"consensus_categories":[],"category_scores_codex":[0.003054327,0.00053786277,0.00053937244,0.0015562782,0.005671212,0.0071931034,0.0009359773,0.003069179,0.06513618],"category_scores_gemma":[0.011990589,0.0003065007,0.0005263642,0.00219032,0.0039881924,0.010329035,0.0032270474,0.005343081,0.012082063],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000020559537,0.000009900766,0.00014028592,0.000057569712,0.0000039906454,0.00007214011,0.00082779396,0.000052024883,0.00014279383,0.099567175,0.8732303,0.025875404],"study_design_scores_gemma":[0.000002074092,0.0000030481974,0.000063876876,0.000047899368,0.0000010596832,0.000024496563,0.00036694057,0.000021277077,0.000047532507,0.006370429,0.99304634,0.000004972391],"about_ca_topic_score_codex":0.035288025,"about_ca_topic_score_gemma":0.079284154,"teacher_disagreement_score":0.999064,"about_ca_system_score_codex":0.00517062,"about_ca_system_score_gemma":0.0034557185,"threshold_uncertainty_score":0.21790224},"labels":[],"label_agreement":null},{"id":"W4408820614","doi":"10.1038/s41597-025-04771-w","title":"The Eye Movement Database of Passage Reading in Vertically Written Traditional Mongolian","year":2025,"lang":"en","type":"article","venue":"Scientific Data","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"Social Sciences and Humanities Research Council of Canada; Canada Research Chairs; Government of Canada","keywords":"Reading (process); Movement (music); Computer science; Database; Information retrieval; Linguistics; Art","score_opus":0.03549533225539785,"score_gpt":0.3043132777584626,"score_spread":0.26881794550306476,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4408820614","genre_codex":"empirical","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9062384,0.0007663311,0.0034119708,0.00008133133,0.00003136993,0.00031480592,0.0805408,0.0004931781,0.008121852],"genre_scores_gemma":[0.81433976,0.00035367245,0.009814444,0.00005240411,0.0000276253,0.00091945723,0.16802739,0.000115483206,0.0063497513],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9996767,0.000063247295,0.000055207533,0.00011023132,0.000060467948,0.000034263972],"domain_scores_gemma":[0.99822384,0.00050962373,0.00024960912,0.00033152205,0.0005765631,0.00010883441],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00041743633,0.0002834982,0.0002819061,0.0022871972,0.00041806753,0.00044544917,0.00030341538,0.0003677359,0.0038027633],"category_scores_gemma":[0.0021490904,0.00010826181,0.00013537664,0.002062087,0.00025880884,0.00037881243,0.000602921,0.00016899435,0.0012410105],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020491926,0.0005919591,0.25592044,0.0051713493,0.00025777204,0.003108496,0.023810823,0.0019393808,0.11948972,0.0022732897,0.049244054,0.53614354],"study_design_scores_gemma":[0.000041345585,0.00019029001,0.941327,0.000093944975,0.00004481695,0.0010257185,0.0035977038,0.0013627307,0.0074715298,0.00033276033,0.044466108,0.000046112415],"about_ca_topic_score_codex":0.01562839,"about_ca_topic_score_gemma":0.025335232,"teacher_disagreement_score":0.01562839,"about_ca_system_score_codex":0.0004223812,"about_ca_system_score_gemma":0.0006535131,"threshold_uncertainty_score":0.031074882},"labels":[],"label_agreement":null},{"id":"W4408861326","doi":"10.1109/icce63647.2025.10929816","title":"FLODA: Harnessing Vision-Language Models for Deepfake Assessment","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada); University of Toronto","funders":"National Research Foundation of Korea","keywords":"Computer science; Artificial intelligence; Natural language processing","score_opus":0.012149839901385451,"score_gpt":0.3430470502657695,"score_spread":0.330897210364384,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4408861326","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024769548,0.0009506038,0.9390413,0.00055762695,0.00023446573,0.00019114483,0.0005995703,0.028952047,0.0047037546],"genre_scores_gemma":[0.4765796,0.0003619536,0.50708336,0.00092517387,0.00009779451,0.00031860304,0.0022875858,0.0016731555,0.010672863],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99895227,0.00023537505,0.00005377996,0.00034960717,0.00029309918,0.0001158303],"domain_scores_gemma":[0.9977496,0.0009872569,0.00017987171,0.00044864631,0.00050498697,0.00012963373],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017429887,0.0016511776,0.0007672657,0.0015310287,0.00060147635,0.0020988788,0.0031733857,0.0018934447,0.0051289066],"category_scores_gemma":[0.008470837,0.0006231103,0.0011154908,0.00046738682,0.0009436268,0.0031358358,0.0024987273,0.0028299678,0.0025879215],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00038647885,0.00024655028,0.0025393744,0.0002969158,0.0001930796,0.000306285,0.0002584262,0.23727658,0.027748572,0.015859207,0.021525199,0.69336337],"study_design_scores_gemma":[0.000014633655,0.000049571492,0.00015356451,0.000017301514,0.000012861245,0.000054142438,0.000023750072,0.9764173,0.00944288,0.010049937,0.0037470108,0.000016988632],"about_ca_topic_score_codex":0.0078514,"about_ca_topic_score_gemma":0.014579906,"teacher_disagreement_score":0.0078514,"about_ca_system_score_codex":0.0018170322,"about_ca_system_score_gemma":0.0016004766,"threshold_uncertainty_score":0.017157972},"labels":[],"label_agreement":null},{"id":"W4408967777","doi":"10.22364/bjmc.2025.13.1.13","title":"Applying Word Embeddings for Lithuanian Morphology: The Case of Adjectival Participles","year":2025,"lang":"en","type":"article","venue":"Baltic Journal of Modern Computing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Leibniz-Gemeinschaft; Atomic Energy of Canada Limited; University of Galway","keywords":"Lithuanian; Morphology (biology); Linguistics; Word (group theory); Word formation; Computer science; Natural language processing; Artificial intelligence; Philosophy; Biology; Zoology","score_opus":0.022118085991670047,"score_gpt":0.32303248215102787,"score_spread":0.30091439615935783,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4408967777","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.57460815,0.0015686813,0.40414533,0.0008725583,0.00031727226,0.00022180067,0.0011088876,0.0014989432,0.01565833],"genre_scores_gemma":[0.83858734,0.0006399454,0.15578422,0.00007013765,0.000038517974,0.00008787698,0.00082073547,0.00024169892,0.0037296268],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.998963,0.00042815617,0.0001281207,0.00024236848,0.00017306698,0.00006527441],"domain_scores_gemma":[0.99739504,0.0013671225,0.00023863,0.00044991644,0.0005035722,0.00004578948],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010293247,0.000674179,0.00032091726,0.0019788023,0.00073916855,0.0022603816,0.00032267423,0.0005632038,0.0032389513],"category_scores_gemma":[0.005231697,0.0002740298,0.0005711428,0.0021701553,0.00089627673,0.0036608237,0.001882271,0.0007511159,0.0013036764],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004264529,0.00013928577,0.02796982,0.0010509235,0.00016301915,0.0022064678,0.0152868675,0.009627309,0.04953998,0.026924247,0.0037474586,0.8629182],"study_design_scores_gemma":[0.00008734588,0.00095567346,0.08655397,0.0011033933,0.0004895914,0.011005884,0.04756755,0.35869563,0.15160124,0.14726569,0.19426033,0.00041373403],"about_ca_topic_score_codex":0.0014687624,"about_ca_topic_score_gemma":0.0021376915,"teacher_disagreement_score":0.0032389513,"about_ca_system_score_codex":0.0003447033,"about_ca_system_score_gemma":0.0006562271,"threshold_uncertainty_score":0.01083535},"labels":[],"label_agreement":null},{"id":"W4408987812","doi":"10.1016/b978-0-323-95504-1.00504-4","title":"Systemic Functional Grammar","year":2025,"lang":"en","type":"book-chapter","venue":"Elsevier eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Trois-Rivières","funders":"","keywords":"Systemic functional grammar; Grammar; Linguistics; Computer science; Philosophy","score_opus":0.011811565384639625,"score_gpt":0.23228609125261987,"score_spread":0.22047452586798025,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4408987812","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.002294017,0.0046668607,0.060943056,0.0013809582,0.00037768745,0.00002262596,0.00042228092,0.00078742055,0.92910504],"genre_scores_gemma":[0.09368648,0.007534366,0.03295935,0.0006901541,0.00060925924,0.000089054825,0.0019273052,0.001182955,0.86132115],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9998323,0.000035932666,0.000009882057,0.00004474546,0.00006373971,0.000013321142],"domain_scores_gemma":[0.9998623,0.000059425805,0.0000048069137,0.000038077334,0.000027631138,0.00000779671],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002599271,0.00070371595,0.0005735111,0.001099518,0.000862713,0.0021850367,0.0005598214,0.0006628853,0.059757188],"category_scores_gemma":[0.0005981435,0.00045440576,0.00044101576,0.0016991356,0.0017186563,0.0027683447,0.0009354616,0.0013800188,0.017866526],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000004993559,0.000010844672,0.000060031925,0.00009965882,0.0000043241084,0.000049062888,0.0002672954,0.000406443,0.00056000054,0.8250221,0.058401335,0.11511391],"study_design_scores_gemma":[0.0000037366126,0.0000060500893,0.00015836532,0.000065131804,0.0000051737757,0.00016581484,0.00006450315,0.000750265,0.00030680822,0.6554075,0.343061,0.000005792635],"about_ca_topic_score_codex":0.0017928964,"about_ca_topic_score_gemma":0.002641526,"teacher_disagreement_score":0.059757188,"about_ca_system_score_codex":0.0011380961,"about_ca_system_score_gemma":0.0007125465,"threshold_uncertainty_score":0.19990772},"labels":[],"label_agreement":null},{"id":"W4409058646","doi":"10.1109/spicscon64195.2024.10941231","title":"BDNEWS: Advance Bengali Sensitivity Corpus and Classifiers","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Regina","funders":"","keywords":"Bengali; Sensitivity (control systems); Computer science; Artificial intelligence; Natural language processing; Speech recognition; Engineering","score_opus":0.010495918152729412,"score_gpt":0.2684747367076321,"score_spread":0.2579788185549027,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409058646","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11759722,0.0069396705,0.03916562,0.0018959866,0.0018877271,0.0018474443,0.71374285,0.029536784,0.0873867],"genre_scores_gemma":[0.11515281,0.0015755738,0.030506523,0.00044066628,0.0003211237,0.0018997347,0.82054114,0.0019755932,0.027586816],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99805933,0.00039215555,0.00027744917,0.000488122,0.0005377262,0.0002451906],"domain_scores_gemma":[0.99674356,0.00062283775,0.00016540279,0.00095978123,0.0013027902,0.00020567089],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011621558,0.0018646456,0.0009837014,0.007456626,0.002068788,0.002080925,0.0022728636,0.0013187709,0.021633526],"category_scores_gemma":[0.005724571,0.00060719566,0.0007692622,0.0059297085,0.00093069335,0.0022268211,0.0019970406,0.0014270445,0.030215172],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00080199144,0.00049235765,0.013206432,0.0024440414,0.00020196925,0.0015420949,0.0015960977,0.004763287,0.037113033,0.009013318,0.68430936,0.244516],"study_design_scores_gemma":[0.00010137942,0.00015642539,0.035535317,0.00024534744,0.00014506138,0.0013048854,0.0014499887,0.012308342,0.036796566,0.003520517,0.9082448,0.00019136182],"about_ca_topic_score_codex":0.054053295,"about_ca_topic_score_gemma":0.061269548,"teacher_disagreement_score":0.054053295,"about_ca_system_score_codex":0.002211761,"about_ca_system_score_gemma":0.0025425244,"threshold_uncertainty_score":0.10747737},"labels":[],"label_agreement":null},{"id":"W4409071264","doi":"10.1075/ml.24034.der","title":"Finnish noun inflections and the FLH from two perspectives","year":2024,"lang":"en","type":"article","venue":"The Mental Lexicon","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Windsor; Concordia University; University of Alberta","funders":"","keywords":"Noun; Linguistics; Natural language processing; Computer science; Philosophy","score_opus":0.00828755562613475,"score_gpt":0.28103343779446915,"score_spread":0.2727458821683344,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409071264","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9520415,0.0030748306,0.014714558,0.0011180082,0.00004533278,0.00002210751,0.00023342714,0.00006645407,0.028683905],"genre_scores_gemma":[0.99599826,0.000282433,0.002908434,0.00006709364,0.000024933717,0.0000123515865,0.00009290667,0.000010227958,0.00060340064],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.999316,0.00021616017,0.000051963507,0.00016642253,0.00018700147,0.00006247847],"domain_scores_gemma":[0.99503905,0.0035804254,0.00052456046,0.00038413072,0.00032693564,0.00014494058],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010661407,0.00023918963,0.0002664817,0.0021300977,0.0007054644,0.0026444711,0.00044741694,0.00084455626,0.0036826634],"category_scores_gemma":[0.006182091,0.0003373448,0.00022465487,0.0012381903,0.0031763662,0.0039290255,0.0018517694,0.00075836136,0.000234961],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001583388,0.00037054336,0.1692097,0.001411765,0.00014846046,0.0072039,0.087168366,0.001215046,0.18196802,0.28400975,0.0018955227,0.26381558],"study_design_scores_gemma":[0.00006285243,0.00048552433,0.6606836,0.00027391553,0.000120677534,0.007995872,0.03910161,0.006344631,0.034298215,0.20433997,0.046078164,0.0002149755],"about_ca_topic_score_codex":0.00089214236,"about_ca_topic_score_gemma":0.0013539747,"teacher_disagreement_score":0.0036826634,"about_ca_system_score_codex":0.0005622526,"about_ca_system_score_gemma":0.00042711693,"threshold_uncertainty_score":0.012319744},"labels":[],"label_agreement":null},{"id":"W4409156994","doi":"10.1109/ieeeconf60004.2024.10942686","title":"Parameter Efficient Fine-tuning of Transformer-Based Language Models Using Dataset Pruning","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Transformer; Language model; Pruning; Artificial intelligence; Engineering; Voltage; Electrical engineering","score_opus":0.033176155495351516,"score_gpt":0.3133406116317856,"score_spread":0.2801644561364341,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409156994","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10919515,0.0012939862,0.8373753,0.00031246842,0.00028762367,0.00039042314,0.0015066162,0.044889033,0.0047493903],"genre_scores_gemma":[0.5884806,0.00047463315,0.3956313,0.0007270805,0.00008476875,0.0006155092,0.0072598825,0.0023924722,0.004333753],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9987993,0.00028312168,0.00012412178,0.00038467807,0.0002852361,0.0001235019],"domain_scores_gemma":[0.9975968,0.0010283164,0.00011098067,0.0007933553,0.00039732072,0.00007320503],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012479911,0.0014530877,0.0015242608,0.0010540024,0.0004596285,0.0014133063,0.0025894938,0.0008791902,0.0033908982],"category_scores_gemma":[0.0076455157,0.0006799456,0.0014698788,0.00095427193,0.00045795937,0.0027997673,0.0016573471,0.0017795515,0.002096657],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007876273,0.00047255968,0.004485643,0.00034123653,0.00026252615,0.000455343,0.0003011281,0.23012128,0.056792814,0.0032364998,0.01732187,0.6854214],"study_design_scores_gemma":[0.00008269153,0.000115880626,0.00086203543,0.00001827585,0.000063058505,0.00015783017,0.00008480086,0.97126454,0.018334119,0.0041779955,0.0048025027,0.000036217018],"about_ca_topic_score_codex":0.008601866,"about_ca_topic_score_gemma":0.015811237,"teacher_disagreement_score":0.008601866,"about_ca_system_score_codex":0.00086031924,"about_ca_system_score_gemma":0.0015507053,"threshold_uncertainty_score":0.017103612},"labels":[],"label_agreement":null},{"id":"W4409164491","doi":"10.1016/j.eswa.2025.127421","title":"How do LLMs perform on Turkish? A multi-faceted multi-prompt evaluation","year":2025,"lang":"en","type":"article","venue":"Expert Systems with Applications","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Zoo","funders":"Boğaziçi Üniversitesi","keywords":"Turkish; Computer science; Business; Risk analysis (engineering)","score_opus":0.02679430898198465,"score_gpt":0.3229170655409644,"score_spread":0.29612275655897974,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409164491","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.97520965,0.00063195167,0.00877607,0.00047207615,0.0001775415,0.00029421854,0.0035553167,0.0025744338,0.008308785],"genre_scores_gemma":[0.9771184,0.00019080791,0.010559531,0.00016491849,0.00003190436,0.0002574943,0.0077748573,0.000432841,0.0034692762],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99307764,0.00420296,0.0005945786,0.0006981787,0.000909845,0.0005168265],"domain_scores_gemma":[0.9774374,0.012655114,0.0011656124,0.0021253792,0.0054795123,0.0011369362],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00517498,0.0013287447,0.0011088785,0.0017289525,0.0011627725,0.001954598,0.0010622006,0.0018298059,0.006590028],"category_scores_gemma":[0.03417983,0.00028736817,0.00078574306,0.0013243493,0.0005789503,0.004099123,0.0025278209,0.0011147243,0.005375652],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.034060992,0.0065785805,0.14628392,0.0065158447,0.00095216034,0.004037867,0.020139443,0.04393933,0.060342,0.003712528,0.088892065,0.5845453],"study_design_scores_gemma":[0.0026639625,0.015601075,0.39490324,0.0013147487,0.0016025711,0.003977785,0.048880484,0.2784905,0.10266734,0.008799523,0.13998084,0.0011179909],"about_ca_topic_score_codex":0.0064967894,"about_ca_topic_score_gemma":0.008302894,"teacher_disagreement_score":0.006590028,"about_ca_system_score_codex":0.0012100172,"about_ca_system_score_gemma":0.0013262464,"threshold_uncertainty_score":0.027368248},"labels":[],"label_agreement":null},{"id":"W4409166972","doi":"10.3389/flang.2025.1413119","title":"The acquisition of object clitic pronouns in Heritage Romanian","year":2025,"lang":"en","type":"article","venue":"Frontiers in Language Sciences","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Clitic; Romanian; Linguistics; Object (grammar); Object pronoun; Computer science; Psychology; Geography; Personal pronoun; Philosophy","score_opus":0.004728209661695602,"score_gpt":0.2663765647334201,"score_spread":0.26164835507172446,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409166972","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99935275,0.00004607122,0.000022797081,0.0000066741495,3.921691e-7,0.0000012443893,0.000014531365,0.0000025905365,0.0005529894],"genre_scores_gemma":[0.99920195,0.00008891941,0.000121539364,0.0000058155742,4.146684e-7,0.0000020917566,0.000035170742,0.0000026588998,0.00054143777],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99962926,0.000058613034,0.000027913356,0.00007074493,0.000093684976,0.000119831886],"domain_scores_gemma":[0.99925715,0.00012902496,0.00031264967,0.00006075348,0.0001494552,0.00009089197],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00048018346,0.0002823397,0.00034183855,0.0007440389,0.0006407648,0.00092177646,0.0003086619,0.00020101374,0.0019279519],"category_scores_gemma":[0.001307276,0.00022475942,0.00017558863,0.00052617607,0.00080062886,0.00041716543,0.00083866884,0.0003969489,0.00035829298],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016504436,0.00012819866,0.8080511,0.00014495288,0.000020827787,0.005284009,0.11687716,0.00009541419,0.029390967,0.00061286223,0.00025882744,0.038970582],"study_design_scores_gemma":[0.000004010269,0.00023118246,0.9526617,0.000021745054,0.00001566702,0.004399237,0.03683055,0.00011300519,0.0034933505,0.00007054454,0.0021442624,0.000014848323],"about_ca_topic_score_codex":0.040649764,"about_ca_topic_score_gemma":0.09221517,"teacher_disagreement_score":0.040649764,"about_ca_system_score_codex":0.00070032425,"about_ca_system_score_gemma":0.00089624774,"threshold_uncertainty_score":0.08082634},"labels":[],"label_agreement":null},{"id":"W4409299747","doi":"10.1016/j.mlwa.2025.100649","title":"Optimizing translation for low-resource languages: Efficient fine-tuning with custom prompt engineering in large language models","year":2025,"lang":"en","type":"article","venue":"Machine Learning with Applications","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Impact","funders":"University of Pretoria","keywords":"Computer science; Translation (biology); Resource (disambiguation); Chemistry","score_opus":0.005241564749280432,"score_gpt":0.24987995040214414,"score_spread":0.24463838565286372,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409299747","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11709872,0.0019257631,0.8226693,0.0009380328,0.0005334026,0.00038419184,0.001019815,0.05088485,0.004545831],"genre_scores_gemma":[0.59966063,0.00061797135,0.384933,0.0013435027,0.00018601782,0.0007225327,0.004271935,0.0030415673,0.0052228076],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988111,0.00049412367,0.00007903234,0.00037047738,0.00014069755,0.00010445281],"domain_scores_gemma":[0.99765223,0.001483002,0.000094254414,0.0003590799,0.00031106017,0.00010030028],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022539063,0.0021464955,0.0011895094,0.00065006054,0.000574099,0.0014430074,0.0017867933,0.0014266478,0.0034370357],"category_scores_gemma":[0.009686245,0.00075667,0.0013456971,0.00072850706,0.0006565089,0.0030213832,0.0016805233,0.0033036955,0.0035333063],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006374363,0.00045397028,0.004980737,0.00049399235,0.00029107384,0.00043624328,0.00052113534,0.52304214,0.024747072,0.0040112627,0.017906971,0.42247802],"study_design_scores_gemma":[0.000070251386,0.000118696,0.00032862247,0.000021754046,0.000059789778,0.00008409579,0.000101771424,0.9847898,0.0062737665,0.0041156923,0.004009444,0.000026275171],"about_ca_topic_score_codex":0.005731475,"about_ca_topic_score_gemma":0.011898915,"teacher_disagreement_score":0.005731475,"about_ca_system_score_codex":0.00075791206,"about_ca_system_score_gemma":0.001849217,"threshold_uncertainty_score":0.011919916},"labels":[],"label_agreement":null},{"id":"W4409328024","doi":"10.29140/vli.v14n1.2097","title":"Metrics for investigations into L2 knowledge of derivational affixes","year":2025,"lang":"en","type":"article","venue":"Vocabulary Learning and Instruction","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Natural language processing; Linguistics; Philosophy","score_opus":0.010330319861456098,"score_gpt":0.28368405070485536,"score_spread":0.27335373084339926,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409328024","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.60410607,0.005890677,0.28856534,0.0013271726,0.00024451874,0.0012756679,0.03502372,0.0040561045,0.059510738],"genre_scores_gemma":[0.7710347,0.00085391867,0.20632963,0.000109650086,0.000058922295,0.0021854532,0.015376284,0.0010225392,0.0030289134],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99307704,0.0025309527,0.0008542324,0.00095584866,0.0022850463,0.00029681576],"domain_scores_gemma":[0.9177635,0.059299268,0.0074963006,0.007727487,0.006405536,0.0013079254],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008165122,0.001038048,0.00094404863,0.009019487,0.0011284063,0.0030610773,0.0010117413,0.00089493964,0.00796966],"category_scores_gemma":[0.07812022,0.00032484223,0.0009354665,0.013957003,0.0014391046,0.0047101965,0.0030636685,0.0019720392,0.0018208728],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00057896524,0.00050829217,0.2903005,0.0019305211,0.00048955274,0.0002839191,0.012801383,0.008608235,0.017851127,0.071315974,0.020207709,0.5751239],"study_design_scores_gemma":[0.00006524168,0.0015173911,0.6604934,0.0008373758,0.00014231907,0.0012476917,0.010719105,0.034163468,0.020318111,0.10158216,0.16857041,0.00034331504],"about_ca_topic_score_codex":0.005324838,"about_ca_topic_score_gemma":0.004831035,"teacher_disagreement_score":0.009019487,"about_ca_system_score_codex":0.0014476418,"about_ca_system_score_gemma":0.0012283658,"threshold_uncertainty_score":0.043181777},"labels":[],"label_agreement":null},{"id":"W4409348132","doi":"10.1609/aaai.v39i23.34639","title":"Prompt Compression with Context-Aware Sentence Encoding for Fast and Improved LLM Inference","year":2025,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Encoding (memory); Sentence; Inference; Computer science; Context (archaeology); Compression (physics); Natural language processing; Artificial intelligence; Speech recognition; History","score_opus":0.03933237982885971,"score_gpt":0.3119017386325069,"score_spread":0.2725693588036472,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409348132","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.052535955,0.0026474455,0.90831244,0.0013194922,0.00058191165,0.00034791915,0.0043792822,0.026382832,0.0034927023],"genre_scores_gemma":[0.344532,0.0010385388,0.63144696,0.0011117052,0.0004763296,0.00060243165,0.0137281995,0.0010553437,0.0060084183],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99926513,0.00026847274,0.00005609379,0.00020762891,0.00014842219,0.0000542295],"domain_scores_gemma":[0.9978569,0.0011892699,0.00010326512,0.0003940733,0.00038546036,0.00007105766],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011320319,0.001232145,0.00071275426,0.001027822,0.00038548585,0.00077687314,0.0011430483,0.001040201,0.0070593716],"category_scores_gemma":[0.0083423285,0.00030746887,0.0007010418,0.0007237176,0.00043130617,0.0024763872,0.0013916291,0.0024939158,0.0033547524],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005536999,0.00024693727,0.0020908753,0.0005092771,0.00006098711,0.00029767223,0.00040512698,0.028154802,0.04157463,0.008111493,0.0336311,0.88436335],"study_design_scores_gemma":[0.0001719509,0.0004047826,0.0019137203,0.00011201613,0.00009637655,0.00041470272,0.0002923441,0.8776525,0.060260314,0.035001878,0.023610475,0.00006884844],"about_ca_topic_score_codex":0.002423014,"about_ca_topic_score_gemma":0.004939731,"teacher_disagreement_score":0.0070593716,"about_ca_system_score_codex":0.00070631,"about_ca_system_score_gemma":0.0012352661,"threshold_uncertainty_score":0.023615956},"labels":[],"label_agreement":null},{"id":"W4409362513","doi":"10.1609/aaai.v39i24.34737","title":"EBBS: An Ensemble with Bi-Level Beam Search for Zero-Shot Machine Translation","year":2025,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Northern Alberta Institute of Technology; University of Alberta","funders":"Alliance de recherche numérique du Canada; Alberta Innovates; Natural Sciences and Engineering Research Council of Canada; Mitacs; Canadian Institute for Advanced Research","keywords":"Zero (linguistics); Shot (pellet); Translation (biology); Machine translation; Beam (structure); Computer science; Physics; Artificial intelligence; Optics; Materials science; Linguistics","score_opus":0.13573189638109215,"score_gpt":0.3552919551217023,"score_spread":0.21956005874061016,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409362513","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0170069,0.0007760199,0.976846,0.0002259555,0.00010401723,0.0000644906,0.00014117634,0.0036317024,0.0012038387],"genre_scores_gemma":[0.40900925,0.0005339719,0.5789361,0.0011458719,0.00025971944,0.0003488605,0.0022899339,0.0010431955,0.0064331437],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985623,0.00057500065,0.000072812356,0.00033596248,0.00031051956,0.00014339661],"domain_scores_gemma":[0.9980719,0.00097592355,0.0000985527,0.0003580448,0.0003765146,0.000119054916],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024870252,0.0018665796,0.0025698694,0.0012977393,0.001032382,0.0011897853,0.0037401167,0.0022516383,0.0033102618],"category_scores_gemma":[0.005419554,0.0011809545,0.0013725851,0.0017718441,0.0011346758,0.0031915924,0.0028263787,0.002870491,0.0022182486],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004465949,0.0002593142,0.0021286323,0.0001485565,0.00044580194,0.00015955884,0.0002873229,0.47687304,0.0075874124,0.01158737,0.011638342,0.48843804],"study_design_scores_gemma":[0.00001901013,0.000040297018,0.00007814843,0.000007240287,0.000020319369,0.000017237986,0.0000124390945,0.9927907,0.0009845691,0.005382832,0.00063891534,0.000008214865],"about_ca_topic_score_codex":0.00893619,"about_ca_topic_score_gemma":0.016727833,"teacher_disagreement_score":0.00893619,"about_ca_system_score_codex":0.0007399251,"about_ca_system_score_gemma":0.00195588,"threshold_uncertainty_score":0.017768383},"labels":[],"label_agreement":null},{"id":"W4409364798","doi":"10.1609/aaai.v39i12.33365","title":"AlphaForge: A Framework to Mine and Dynamically Combine Formulaic Alpha Factors","year":2025,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; York University","funders":"Fundamental Research Funds for the Central Universities; National Natural Science Foundation of China","keywords":"Alpha (finance); Computer science; Data science; Human–computer interaction; Psychology; Developmental psychology","score_opus":0.030045716920110585,"score_gpt":0.3118845256309136,"score_spread":0.281838808710803,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409364798","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0050189923,0.0002613019,0.9908223,0.00012180742,0.000017890203,0.00007668201,0.00014824748,0.0027606906,0.00077196694],"genre_scores_gemma":[0.14638469,0.00031396936,0.85001856,0.00018958979,0.000048270365,0.00023670285,0.0008020127,0.00036210995,0.0016440427],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99909997,0.00021670599,0.000063368585,0.00025795435,0.00027129508,0.000090761445],"domain_scores_gemma":[0.9980913,0.0010161691,0.0002318742,0.00024818344,0.00032215365,0.00009023565],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026965186,0.0015524153,0.0011315116,0.0025768285,0.00053868524,0.0019490172,0.0025478012,0.0011030261,0.0025544115],"category_scores_gemma":[0.00849791,0.0008057338,0.0014598018,0.0015050691,0.0009926482,0.002519464,0.0017926686,0.001811247,0.00088109396],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022643483,0.00014263915,0.008987346,0.00020510577,0.0002424291,0.0002987324,0.0002798107,0.32768658,0.004922417,0.021823,0.0071768877,0.6280085],"study_design_scores_gemma":[0.000017096532,0.000037709506,0.0003464753,0.000022325998,0.000020502774,0.00008581681,0.000021249398,0.9730992,0.0018646991,0.020934293,0.0035352746,0.000015418438],"about_ca_topic_score_codex":0.0044927867,"about_ca_topic_score_gemma":0.009111359,"teacher_disagreement_score":0.0044927867,"about_ca_system_score_codex":0.00086506974,"about_ca_system_score_gemma":0.0018795971,"threshold_uncertainty_score":0.014260709},"labels":[],"label_agreement":null},{"id":"W4409407626","doi":"10.1007/978-3-031-88036-0_8","title":"GERA: A Corpus of Russian School Texts Annotated for Grammatical Error Correction","year":2025,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Error analysis; Linguistics; Mathematics; Philosophy","score_opus":0.011770450229882962,"score_gpt":0.2758623785684261,"score_spread":0.2640919283385431,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409407626","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.254973,0.005541711,0.023770541,0.0012363618,0.0011480727,0.0006732851,0.6339568,0.016841406,0.06185888],"genre_scores_gemma":[0.21249434,0.0014310207,0.041862432,0.00022463176,0.00023576728,0.000512771,0.7206328,0.0037384292,0.01886793],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9978448,0.0007577615,0.0002541882,0.0006086729,0.00041561082,0.00011896575],"domain_scores_gemma":[0.99514246,0.0023420283,0.00036746688,0.0008917859,0.0010903977,0.00016578754],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001518273,0.0011581326,0.00085431826,0.0055306074,0.0017907064,0.0011482489,0.0009965346,0.0011710637,0.01622104],"category_scores_gemma":[0.0053988104,0.00065545796,0.0005040997,0.0042557972,0.0009178835,0.0014244885,0.001531601,0.0009750792,0.015496394],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015566107,0.0006807077,0.015213613,0.008754805,0.00029749644,0.003258895,0.015355512,0.003891631,0.08316336,0.009983946,0.5781515,0.27969185],"study_design_scores_gemma":[0.00026330948,0.00023117887,0.08876015,0.00070022006,0.00026665427,0.0026557518,0.003907891,0.0051838285,0.026698412,0.0029124462,0.8682514,0.00016874097],"about_ca_topic_score_codex":0.005445157,"about_ca_topic_score_gemma":0.010840602,"teacher_disagreement_score":0.01622104,"about_ca_system_score_codex":0.0007442715,"about_ca_system_score_gemma":0.0017833564,"threshold_uncertainty_score":0.054264784},"labels":[],"label_agreement":null},{"id":"W4409526824","doi":"10.5430/wjel.v15n5p362","title":"A Linguistics-based deep learning Approach to ETL for Automated Translation of English Language Data","year":2025,"lang":"en","type":"article","venue":"World Journal of English Language","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Natural language processing; Translation (biology); Artificial intelligence; Machine translation; Linguistics; Information retrieval; Philosophy; Chemistry","score_opus":0.01717591592644528,"score_gpt":0.3053799950883426,"score_spread":0.28820407916189733,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409526824","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00592581,0.00021392804,0.98856014,0.00050397974,0.0001021001,0.00008033127,0.00028043796,0.0027637286,0.0015694709],"genre_scores_gemma":[0.18366341,0.0005793261,0.8014056,0.00072841806,0.00018433845,0.0003207267,0.0023755855,0.00050894904,0.010233703],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988695,0.0003512742,0.00014452578,0.00029364115,0.0002457128,0.00009519695],"domain_scores_gemma":[0.9983133,0.0006567881,0.0001544835,0.00028345338,0.00053454685,0.000057371177],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017459458,0.0008122071,0.0005846469,0.0011218784,0.00064357027,0.0015253013,0.0012525026,0.0010864343,0.004526444],"category_scores_gemma":[0.0044964585,0.0004162611,0.0010322925,0.0015079388,0.0008122168,0.0022408639,0.0021185814,0.0028119457,0.0024779744],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018861669,0.00026862192,0.0010592133,0.00028707433,0.000110280496,0.00031316932,0.00039350704,0.09402957,0.020922743,0.029833427,0.015056453,0.8375374],"study_design_scores_gemma":[0.000017936112,0.00006186291,0.0003083718,0.000041755204,0.000025780015,0.00010562419,0.00008329942,0.95499086,0.010501727,0.025482472,0.008357833,0.00002251156],"about_ca_topic_score_codex":0.004194937,"about_ca_topic_score_gemma":0.0070714448,"teacher_disagreement_score":0.004526444,"about_ca_system_score_codex":0.0011506873,"about_ca_system_score_gemma":0.0022361404,"threshold_uncertainty_score":0.015142441},"labels":[],"label_agreement":null},{"id":"W4409537504","doi":"10.1145/3727200.3727220","title":"Towards Sustainable Large Language Model Serving","year":2024,"lang":"en","type":"article","venue":"ACM SIGEnergy Energy Informatics Review","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Business","score_opus":0.012000158081089993,"score_gpt":0.29057773019446503,"score_spread":0.278577572113375,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409537504","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012105591,0.0011490838,0.9678994,0.004794863,0.00013242791,0.00008785014,0.00073239417,0.0042779394,0.008820311],"genre_scores_gemma":[0.2620654,0.0019079897,0.7141131,0.002040323,0.00028449466,0.00038464018,0.0031927484,0.0026083093,0.013402979],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.997968,0.00091157155,0.00009557649,0.00024555615,0.0006411016,0.00013820306],"domain_scores_gemma":[0.99492043,0.0032036807,0.00018944767,0.0009138547,0.00066132814,0.00011129897],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031220734,0.00067825645,0.00083718373,0.0007720403,0.0006849624,0.0030732625,0.0026279406,0.0016279094,0.0050330204],"category_scores_gemma":[0.009593454,0.00052874896,0.0014976534,0.001212853,0.001530858,0.0068774596,0.0027772896,0.0030395628,0.0021254264],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000089556655,0.00013798688,0.00094203535,0.0004982433,0.00010084387,0.0003111027,0.0007370001,0.35985938,0.0053599547,0.49035305,0.031195184,0.11041561],"study_design_scores_gemma":[0.000010056765,0.000012900267,0.000062066894,0.000022740667,0.000009710683,0.000056761983,0.00013130944,0.7871527,0.001513606,0.19015616,0.020859536,0.000012374173],"about_ca_topic_score_codex":0.007562078,"about_ca_topic_score_gemma":0.0119020445,"teacher_disagreement_score":0.007562078,"about_ca_system_score_codex":0.0025708827,"about_ca_system_score_gemma":0.002589793,"threshold_uncertainty_score":0.018653154},"labels":[],"label_agreement":null},{"id":"W4409540887","doi":"10.32388/y43w9w","title":"Review of: \"Enhancing Code LLMs with Reinforcement Learning in Code Generation: A Survey\"","year":2025,"lang":"en","type":"peer-review","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ontario Tech University","funders":"","keywords":"Code (set theory); Reinforcement learning; Reinforcement; Computer science; Psychology; Programming language; Artificial intelligence; Social psychology","score_opus":0.039969855940232224,"score_gpt":0.3369271904147481,"score_spread":0.2969573344745159,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409540887","genre_codex":"review","genre_gemma":"commentary","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"commentary","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0037848593,0.8490828,0.054143682,0.050691307,0.012320231,0.0002999925,0.0008125477,0.0021031448,0.026761372],"genre_scores_gemma":[0.029092677,0.8733307,0.030721694,0.016122624,0.00914395,0.0003327079,0.0024516473,0.0013735415,0.037430353],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.997029,0.0007631836,0.00029831313,0.00034771246,0.001410014,0.0001517797],"domain_scores_gemma":[0.9622125,0.0154401325,0.001445147,0.0015536238,0.018220583,0.0011280063],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0051191445,0.00072587584,0.0008245686,0.0037423738,0.0006097595,0.0018077923,0.0018893795,0.0014889239,0.013816848],"category_scores_gemma":[0.030066095,0.0004713754,0.0005706286,0.0050381236,0.0010732225,0.0044062543,0.0013359281,0.0014282361,0.007171974],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000058517635,0.00004826253,0.00033795703,0.0074766693,0.00004603214,0.00004247042,0.0001239317,0.0013763172,0.0007480973,0.003980011,0.23214261,0.7536191],"study_design_scores_gemma":[0.000014715212,0.000118644944,0.0010311636,0.0040161647,0.00006788728,0.00018565891,0.00012825617,0.0013860163,0.0015167237,0.0017727215,0.9897307,0.00003139949],"about_ca_topic_score_codex":0.005223876,"about_ca_topic_score_gemma":0.0053821267,"teacher_disagreement_score":0.013816848,"about_ca_system_score_codex":0.001588715,"about_ca_system_score_gemma":0.0045650736,"threshold_uncertainty_score":0.04622203},"labels":[],"label_agreement":null},{"id":"W4409576386","doi":"10.61091/jcmcc127a-136","title":"Analyzing Semantic Alignment Mechanisms and Translation Accuracy in English-Chinese Translation Using Support Vector Machines","year":2025,"lang":"en","type":"article","venue":"Journal of Combinatorial Mathematics and Combinatorial Computing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Translation (biology); Computer science; Natural language processing; Artificial intelligence; Machine translation; Support vector machine; Biology","score_opus":0.013830815796082908,"score_gpt":0.2899171666701963,"score_spread":0.2760863508741134,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409576386","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.60403407,0.0013871082,0.3895207,0.0004587268,0.00013366464,0.00006412296,0.00019189842,0.0012157236,0.0029940058],"genre_scores_gemma":[0.97206736,0.00023545338,0.02590786,0.000058970665,0.000027385146,0.000037686696,0.0004897443,0.0000750896,0.0011005048],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99880207,0.00044252194,0.0001097528,0.0002817977,0.0002413253,0.00012244382],"domain_scores_gemma":[0.99781966,0.0010888132,0.000219516,0.0003042532,0.0005095909,0.000058288446],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028966984,0.00094419345,0.0006165554,0.00091719156,0.0005282175,0.0009177594,0.0006610438,0.000749345,0.0009696725],"category_scores_gemma":[0.009403311,0.00025432874,0.00072327326,0.0010363606,0.0007019917,0.002497552,0.0007215972,0.0009719506,0.0005902098],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006555184,0.00026605232,0.031200632,0.00033703187,0.00029672752,0.00037460114,0.0007265535,0.395621,0.03139244,0.01236102,0.0044032848,0.5223651],"study_design_scores_gemma":[0.000013025151,0.00011846869,0.0041135885,0.000013442318,0.000048561575,0.00006173196,0.0001298672,0.9796444,0.011761018,0.003335603,0.00074187736,0.00001836394],"about_ca_topic_score_codex":0.004870721,"about_ca_topic_score_gemma":0.0034734593,"teacher_disagreement_score":0.004870721,"about_ca_system_score_codex":0.00057923095,"about_ca_system_score_gemma":0.0008466806,"threshold_uncertainty_score":0.015319407},"labels":[],"label_agreement":null},{"id":"W4409576553","doi":"10.61091/jcmcc127a-131","title":"Inference of Discourse Hierarchical Features in English Corpus Based on Multilayer Bayesian Modeling","year":2025,"lang":"en","type":"article","venue":"Journal of Combinatorial Mathematics and Combinatorial Computing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Inference; Computer science; Artificial intelligence; Natural language processing; Bayesian inference; Bayesian probability; Linguistics","score_opus":0.010008845663375152,"score_gpt":0.2906160256711784,"score_spread":0.2806071800078032,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409576553","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04761271,0.0004193604,0.94739014,0.00038036914,0.000027945778,0.00015050542,0.0011801731,0.00044687328,0.00239189],"genre_scores_gemma":[0.6090999,0.0007722842,0.37927544,0.0002097919,0.00011556788,0.00084572955,0.0045059617,0.00018615129,0.0049891244],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99659985,0.0012995092,0.00026086433,0.00097643165,0.0006549953,0.0002083834],"domain_scores_gemma":[0.9931288,0.004884529,0.0005613845,0.00043692865,0.00086351635,0.00012487345],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0040770886,0.00091420766,0.0011553873,0.0037544365,0.0011246749,0.0020550545,0.0015910228,0.0010057938,0.0026839543],"category_scores_gemma":[0.019365842,0.0009872401,0.0014421111,0.0025242225,0.0009518811,0.0046185073,0.0020120973,0.0020642115,0.0006959912],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006973388,0.0005113417,0.064150184,0.0008497152,0.00060230563,0.0011400017,0.008645433,0.22779568,0.013571322,0.1896372,0.009982241,0.48241726],"study_design_scores_gemma":[0.00002926249,0.000042311378,0.01061841,0.00009279365,0.00013400814,0.0001255995,0.00042337336,0.90707755,0.0022260656,0.073703215,0.0054565896,0.00007076355],"about_ca_topic_score_codex":0.023469431,"about_ca_topic_score_gemma":0.029653149,"teacher_disagreement_score":0.023469431,"about_ca_system_score_codex":0.0015095033,"about_ca_system_score_gemma":0.0018309184,"threshold_uncertainty_score":0.04666561},"labels":[],"label_agreement":null},{"id":"W4409581635","doi":"10.31235/osf.io/wg82k_v2","title":"Updating “The Future of Coding”: Qualitative Coding with Generative Large Language Models","year":2025,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Canadian Institute for Advanced Research; Yale University; City University of New York; Sage Foundation","keywords":"Generative grammar; Coding (social sciences); Computer science; Natural language processing; Artificial intelligence; Linguistics; Mathematics; Statistics; Philosophy","score_opus":0.023302360580045527,"score_gpt":0.34020316410229956,"score_spread":0.316900803522254,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409581635","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010166634,0.00037013058,0.96973115,0.009501516,0.00030904895,0.0002276725,0.0005837569,0.000924988,0.008185076],"genre_scores_gemma":[0.29632467,0.00049625593,0.6936782,0.0028579014,0.00016886345,0.0012887155,0.00082927477,0.0013862334,0.0029699367],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"qualitative","domain_scores_codex":[0.90777344,0.079039976,0.0018528617,0.004684538,0.0059080557,0.0007411404],"domain_scores_gemma":[0.663354,0.25472212,0.008702617,0.052572962,0.018889436,0.0017588724],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.06779722,0.0012469906,0.00075096346,0.0030295698,0.0033634764,0.010315595,0.0041428073,0.002496336,0.010227415],"category_scores_gemma":[0.32012838,0.001303073,0.0014307976,0.0029531969,0.020850176,0.021812303,0.008550578,0.0061433213,0.0026648687],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015537064,0.000055989254,0.0034609938,0.00070440903,0.00006524481,0.0001724834,0.0513682,0.008678955,0.0018268533,0.8304624,0.010216925,0.09283217],"study_design_scores_gemma":[0.000045544806,0.000028203058,0.000518962,0.00063411606,0.000022539067,0.00013764638,0.006435598,0.03848104,0.0028807519,0.9051442,0.04557294,0.00009841563],"about_ca_topic_score_codex":0.0056882636,"about_ca_topic_score_gemma":0.0065409313,"teacher_disagreement_score":0.93220276,"about_ca_system_score_codex":0.0071238196,"about_ca_system_score_gemma":0.009165776,"threshold_uncertainty_score":0.35855025},"labels":[],"label_agreement":null},{"id":"W4409602165","doi":"10.61091/jcmcc127b-044","title":"A Language Model-based Approach to Context Analysis in Business English Translation","year":2025,"lang":"en","type":"article","venue":"Journal of Combinatorial Mathematics and Combinatorial Computing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Translation (biology); Computer science; Linguistics; Context (archaeology); Business English; Natural language processing; History; Philosophy; Archaeology","score_opus":0.012103283140480895,"score_gpt":0.2661340313487471,"score_spread":0.2540307482082662,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409602165","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006070841,0.000643512,0.9896947,0.00031638646,0.000103058876,0.00007582023,0.00019060439,0.0011954383,0.0017096067],"genre_scores_gemma":[0.23488021,0.0009871566,0.7577024,0.0005193167,0.00023135482,0.0002850953,0.0011971624,0.00048481105,0.00371252],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985636,0.0006160519,0.000115895265,0.0003832118,0.00023965556,0.00008158031],"domain_scores_gemma":[0.9987431,0.00049906055,0.000100224075,0.00025607616,0.00035641974,0.000045079163],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010352284,0.0010031483,0.00079408084,0.0015321422,0.0007665477,0.001644178,0.00095227157,0.0008698252,0.002927063],"category_scores_gemma":[0.0039932034,0.00046991758,0.0015252401,0.0014165672,0.0007670602,0.0027491527,0.0016116942,0.0021812932,0.0015892728],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00034412893,0.00024584687,0.0025005455,0.0009190539,0.0002445796,0.0006801801,0.0013908383,0.09900225,0.061037708,0.10610793,0.009069292,0.71845764],"study_design_scores_gemma":[0.000045908535,0.00019376607,0.001036323,0.000097039636,0.00016576536,0.00042312473,0.0003778576,0.86641866,0.024885235,0.081331246,0.02492347,0.00010168755],"about_ca_topic_score_codex":0.005311704,"about_ca_topic_score_gemma":0.006269712,"teacher_disagreement_score":0.005311704,"about_ca_system_score_codex":0.0008256185,"about_ca_system_score_gemma":0.001429467,"threshold_uncertainty_score":0.010561585},"labels":[],"label_agreement":null},{"id":"W4409603734","doi":"10.61091/jcmcc127b-203","title":"Context-Aware Translation Accuracy Improvement Strategies Based on Deep Reinforcement Learning in English Translation","year":2025,"lang":"en","type":"article","venue":"Journal of Combinatorial Mathematics and Combinatorial Computing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Translation (biology); Reinforcement learning; Computer science; Artificial intelligence; Context (archaeology); Natural language processing; History; Biology","score_opus":0.012860370487113847,"score_gpt":0.27477542401613353,"score_spread":0.26191505352901967,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409603734","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17441624,0.0021073495,0.8143694,0.00053278095,0.00022052902,0.00007681046,0.00008606442,0.0031662434,0.005024598],"genre_scores_gemma":[0.92997366,0.00032763582,0.06684056,0.00020307375,0.000043412867,0.00004827428,0.00014573497,0.00010462137,0.0023130102],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.999432,0.00016426445,0.000055025685,0.0001565713,0.00011950732,0.00007251133],"domain_scores_gemma":[0.9995571,0.00014635923,0.00005303956,0.000061822764,0.00015282145,0.00002883488],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008920144,0.00079321046,0.00090353895,0.00035089624,0.00042817192,0.00056209636,0.00077664887,0.00059285713,0.0011270333],"category_scores_gemma":[0.0022865955,0.00027697242,0.00054824585,0.00039557347,0.00039065804,0.0010558944,0.0007680996,0.0008432616,0.00038612416],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00031756455,0.0002310286,0.0035941615,0.0001804813,0.00013380918,0.00038306328,0.00027388046,0.52796966,0.0278266,0.006253432,0.0037842703,0.429052],"study_design_scores_gemma":[0.000022222963,0.00005775561,0.0003289531,0.0000068160393,0.000029298497,0.000049619648,0.000014105865,0.9925991,0.0045935605,0.0016046929,0.00068412715,0.000009757307],"about_ca_topic_score_codex":0.005947482,"about_ca_topic_score_gemma":0.0054813772,"teacher_disagreement_score":0.005947482,"about_ca_system_score_codex":0.0006679756,"about_ca_system_score_gemma":0.0010207673,"threshold_uncertainty_score":0.011825681},"labels":[],"label_agreement":null},{"id":"W4409603837","doi":"10.61091/jcmcc127b-164","title":"A Computational Approach to the Classification of Chinese Syntactic Structures Using a Large-Scale Corpus","year":2025,"lang":"en","type":"article","venue":"Journal of Combinatorial Mathematics and Combinatorial Computing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Natural language processing; Computer science; Artificial intelligence; Scale (ratio); Cartography; Geography","score_opus":0.013692508274592502,"score_gpt":0.28896891387485657,"score_spread":0.27527640560026406,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409603837","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022731347,0.00018283291,0.97131395,0.0003571705,0.000053620577,0.0002045385,0.00077046186,0.002857473,0.0015286005],"genre_scores_gemma":[0.17235646,0.00026850062,0.8199803,0.0001403576,0.00007319525,0.0008756862,0.0039369897,0.00021906337,0.0021495018],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.998938,0.00027226162,0.00008811275,0.000410159,0.00023260136,0.00005890092],"domain_scores_gemma":[0.99790156,0.0010791719,0.00013333926,0.00041675096,0.00042473504,0.00004448698],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001404447,0.0011782572,0.0008169977,0.0028973098,0.0012903726,0.0010194047,0.00173357,0.00063904404,0.0030788512],"category_scores_gemma":[0.005482105,0.00067247206,0.0013197226,0.0030579823,0.0009419364,0.0028934856,0.001329358,0.0014388977,0.0008894146],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016136555,0.00019229105,0.005283314,0.00060797,0.00017595054,0.00048738072,0.00084286724,0.12653077,0.024801917,0.05339619,0.016291298,0.7712287],"study_design_scores_gemma":[0.000022884147,0.000036092366,0.0016906194,0.000020279722,0.000043898548,0.00010594852,0.00014275352,0.97233117,0.004079848,0.017906219,0.003588678,0.000031647865],"about_ca_topic_score_codex":0.022872726,"about_ca_topic_score_gemma":0.028723195,"teacher_disagreement_score":0.022872726,"about_ca_system_score_codex":0.001452116,"about_ca_system_score_gemma":0.0038925903,"threshold_uncertainty_score":0.04547918},"labels":[],"label_agreement":null},{"id":"W4409604062","doi":"10.63485/r0tft-wq383","title":"Willinsky on OA in Vancouver","year":2005,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Geography","score_opus":0.013315314061339246,"score_gpt":0.2777989826792172,"score_spread":0.264483668617878,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409604062","genre_codex":"other","genre_gemma":"commentary","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"commentary","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00589462,0.013556643,0.00041625195,0.069758065,0.01504507,0.000051463798,0.0011724959,0.00022771732,0.8938776],"genre_scores_gemma":[0.008625489,0.0016676101,0.000112080364,0.0021663713,0.0002928612,0.000007796243,0.00016485214,0.000085292195,0.9868776],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9986797,0.00009933147,0.000020131225,0.00016515136,0.00060583523,0.0004298494],"domain_scores_gemma":[0.9984819,0.00007638735,0.000025367934,0.00004636371,0.00050342333,0.00086663646],"candidate_categories":["scholarly_communication","open_science"],"consensus_categories":[],"category_scores_codex":[0.0006827843,0.00057137694,0.0005725343,0.0015062798,0.01780073,0.007881053,0.0006983262,0.0020479632,0.17649482],"category_scores_gemma":[0.0020699617,0.00036489102,0.0004066948,0.0023823937,0.0021054607,0.0020094193,0.00350915,0.003751484,0.028060077],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":true,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000035553407,0.000024562389,0.0008377728,0.000034853376,0.0000056860285,0.0002611315,0.00080350705,0.00005800772,0.00019689558,0.015205735,0.95368624,0.028849917],"study_design_scores_gemma":[0.0000014549945,0.000003293988,0.0005041402,0.000029595185,0.0000010962094,0.000015484107,0.0007042571,0.000011334855,0.000045912206,0.0005714714,0.99810743,0.000004565],"about_ca_topic_score_codex":0.65755624,"about_ca_topic_score_gemma":0.9051581,"teacher_disagreement_score":0.9993017,"about_ca_system_score_codex":0.01480811,"about_ca_system_score_gemma":0.016171146,"threshold_uncertainty_score":0.688921},"labels":[],"label_agreement":null},{"id":"W4409613891","doi":"10.61091/jcmcc127b-008","title":"A Study of Word Vector Computation and Multilayer Network Representation in English Corpus","year":2025,"lang":"en","type":"article","venue":"Journal of Combinatorial Mathematics and Combinatorial Computing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Word (group theory); Computer science; Representation (politics); Natural language processing; Computation; Artificial intelligence; Linguistics; Algorithm","score_opus":0.01470057720781689,"score_gpt":0.2977669803224671,"score_spread":0.28306640311465026,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409613891","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19267653,0.0010296962,0.8022729,0.0005817495,0.00008440387,0.000105719584,0.00033649788,0.0007932085,0.0021192578],"genre_scores_gemma":[0.7579482,0.0007669845,0.23681346,0.00009974809,0.0000828074,0.00019795001,0.0011000626,0.00007641711,0.0029143759],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990332,0.00031783103,0.0000817993,0.0003216367,0.00016734548,0.0000781988],"domain_scores_gemma":[0.9973666,0.0015402816,0.00022687454,0.0002460376,0.00055498804,0.00006519266],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001664898,0.0006544932,0.000678365,0.0017576923,0.00051475246,0.0010082844,0.00095951924,0.0006581173,0.0010580275],"category_scores_gemma":[0.009553081,0.00031522376,0.00066497375,0.0026018275,0.0005072578,0.0037246589,0.00066213775,0.00093452795,0.00019668872],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028550642,0.00020579733,0.009969896,0.00030241106,0.00017855785,0.00024038921,0.0005800533,0.48556998,0.0066491496,0.029941274,0.003516293,0.4625607],"study_design_scores_gemma":[0.0000029552186,0.000015573836,0.0006406478,0.0000048559846,0.000011368133,0.00001549721,0.000023872653,0.99480987,0.00055047375,0.0035879281,0.00033169083,0.0000051927673],"about_ca_topic_score_codex":0.01947186,"about_ca_topic_score_gemma":0.014015066,"teacher_disagreement_score":0.01947186,"about_ca_system_score_codex":0.001294939,"about_ca_system_score_gemma":0.0007711308,"threshold_uncertainty_score":0.03871703},"labels":[],"label_agreement":null},{"id":"W4409678925","doi":"10.22148/001c.128010","title":"The “Mapping German fiction in translation” dataset: Data collection, scope, and data quality","year":2025,"lang":"en","type":"article","venue":"Journal of Cultural Analytics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Scope (computer science); German; Computer science; Translation (biology); Data quality; Quality (philosophy); Data collection; Information retrieval; Data science; History; Statistics; Engineering; Operations management; Mathematics","score_opus":0.10180379937609223,"score_gpt":0.4145446143603592,"score_spread":0.312740814984267,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409678925","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021463046,0.00089463644,0.001808335,0.0011756853,0.00015966942,0.00029503962,0.96743983,0.0006533304,0.006110449],"genre_scores_gemma":[0.014469065,0.00025385295,0.004029494,0.00017853684,0.000054157277,0.00068047264,0.97839934,0.00013069512,0.0018044106],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99627614,0.001058518,0.00065424695,0.00057527964,0.0011581426,0.00027762252],"domain_scores_gemma":[0.98876905,0.0054648663,0.0009401199,0.0020070635,0.0023349496,0.00048393247],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037122313,0.000682471,0.0006639403,0.008186221,0.0011328231,0.002323367,0.0011089224,0.0012996967,0.009980333],"category_scores_gemma":[0.023218317,0.00021940857,0.00052095007,0.010678519,0.00088640436,0.0013209091,0.0023437333,0.0010528524,0.010587164],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022919214,0.0001413227,0.019570732,0.002584262,0.00010085411,0.00042485024,0.0018653023,0.001170133,0.0017856457,0.0057777427,0.91743124,0.048918575],"study_design_scores_gemma":[0.000111400885,0.000049649334,0.041325383,0.0006239642,0.000049842278,0.00035151246,0.0020260552,0.0011081671,0.0020144894,0.0024838226,0.9497887,0.00006708778],"about_ca_topic_score_codex":0.00934694,"about_ca_topic_score_gemma":0.021793785,"teacher_disagreement_score":0.009980333,"about_ca_system_score_codex":0.0015385388,"about_ca_system_score_gemma":0.0026234528,"threshold_uncertainty_score":0.033387482},"labels":[],"label_agreement":null},{"id":"W4409740531","doi":"10.1007/s10994-025-06767-4","title":"Developing safe and responsible large language model: can we balance bias reduction and language understanding?","year":2025,"lang":"en","type":"article","venue":"Machine Learning","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; Vector Institute","funders":"","keywords":"Reduction (mathematics); Balance (ability); Computer science; Cognitive psychology; Linguistics; Psychology; Mathematics; Philosophy","score_opus":0.024262651807243713,"score_gpt":0.30208355966579215,"score_spread":0.2778209078585484,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409740531","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018654944,0.00028005577,0.9720076,0.0044598905,0.00009024163,0.00008322762,0.00017467357,0.0028329305,0.0014164828],"genre_scores_gemma":[0.49063092,0.00069107127,0.49898177,0.002382956,0.0002823063,0.00029001513,0.00088415155,0.0024484347,0.0034083668],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99587953,0.0020837183,0.00022376128,0.0006588035,0.0008329463,0.00032119243],"domain_scores_gemma":[0.9679933,0.017299227,0.0013739691,0.008687766,0.0036748936,0.00097075687],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009263448,0.0010751954,0.001673326,0.00078218634,0.00091733667,0.0033426955,0.0035285603,0.0022224595,0.0040070224],"category_scores_gemma":[0.056290504,0.0011028306,0.0011131676,0.00052488566,0.0025670354,0.015886765,0.0053209024,0.006344907,0.0025105523],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010635057,0.00047879072,0.012815839,0.00073246495,0.0005327993,0.00046091824,0.0019270643,0.165463,0.027431795,0.2581331,0.022723656,0.5082371],"study_design_scores_gemma":[0.00007123703,0.000061914056,0.00039301955,0.000054024862,0.00009104424,0.00012356936,0.00024824942,0.6054031,0.011899557,0.3764156,0.0052002724,0.00003841707],"about_ca_topic_score_codex":0.0043692407,"about_ca_topic_score_gemma":0.008204532,"teacher_disagreement_score":0.009263448,"about_ca_system_score_codex":0.0014274748,"about_ca_system_score_gemma":0.0055312444,"threshold_uncertainty_score":0.04899037},"labels":[],"label_agreement":null},{"id":"W4409759276","doi":"10.1021/acs.jproteome.6c00336","title":"Label-Free Quantification in the Crux Toolkit","year":2025,"lang":"en","type":"preprint","venue":"Journal of Proteome Research","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"National Science Foundation Graduate Research Fellowship Program; Division of Mathematical Sciences; National Research University Higher School of Economics; National Science Foundation","keywords":"Computer science","score_opus":0.1513608603829877,"score_gpt":0.45681963144160886,"score_spread":0.3054587710586212,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409759276","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006522853,0.00044445225,0.6895235,0.0002952578,0.00017007263,0.00013447939,0.004774843,0.29250905,0.0056254296],"genre_scores_gemma":[0.06956398,0.00057411875,0.8385191,0.000911734,0.000110055065,0.0009214991,0.016764795,0.0607336,0.011901108],"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99671006,0.00050785195,0.0002885231,0.000856626,0.001345124,0.00029181055],"domain_scores_gemma":[0.9964095,0.0010415481,0.00022607934,0.0013157991,0.0009032882,0.00010381401],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032556555,0.0020012895,0.0011140815,0.0016725431,0.00069660036,0.0023879397,0.002701202,0.0013116322,0.01942893],"category_scores_gemma":[0.0068046725,0.0013567972,0.0014148557,0.0010023104,0.0010165428,0.0036489987,0.0033692038,0.0034156085,0.016928762],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020144985,0.0002108483,0.0034188733,0.002491819,0.00051021186,0.00075171934,0.0013124804,0.02044876,0.3628087,0.06388848,0.27612385,0.2660198],"study_design_scores_gemma":[0.0002279604,0.00014647823,0.0025438084,0.00034406024,0.000104257546,0.0008211698,0.0001727521,0.23793338,0.3805193,0.04610615,0.3306228,0.00045789356],"about_ca_topic_score_codex":0.0032665115,"about_ca_topic_score_gemma":0.002723928,"teacher_disagreement_score":0.01942893,"about_ca_system_score_codex":0.0010439234,"about_ca_system_score_gemma":0.0016028029,"threshold_uncertainty_score":0.0649963},"labels":[],"label_agreement":null},{"id":"W4409763820","doi":"10.32628/ijsrst251222653","title":"Fusion of Fast-text and Indo-Wordnet for Disambiguation of Word Sense in the Marathi Language","year":2025,"lang":"en","type":"article","venue":"International Journal of Scientific Research in Science and Technology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Horizon College and Seminary","funders":"","keywords":"WordNet; Marathi; Natural language processing; Word-sense disambiguation; Computer science; Artificial intelligence; Word (group theory); Linguistics; Philosophy","score_opus":0.03322278057973219,"score_gpt":0.41547804448658787,"score_spread":0.38225526390685566,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409763820","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2772177,0.0087908115,0.64298093,0.002958637,0.0022693216,0.0013084367,0.011565771,0.014166714,0.03874172],"genre_scores_gemma":[0.58910006,0.0027858922,0.365736,0.00076392945,0.00038238626,0.00051756296,0.02862214,0.0005977675,0.011494244],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99798214,0.00051536044,0.00021737935,0.0006090372,0.0005099617,0.00016611401],"domain_scores_gemma":[0.9979913,0.00061392406,0.00015879648,0.00034444043,0.0007694375,0.00012196594],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022685726,0.0016886105,0.0014286832,0.009104685,0.001302694,0.002541505,0.0012605464,0.0010967682,0.003091793],"category_scores_gemma":[0.005161939,0.00040087182,0.0012648044,0.0051383898,0.000746158,0.008528414,0.0028750561,0.0013603133,0.0038560415],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00088567,0.0008753779,0.014769102,0.0016596062,0.00046271557,0.0012561933,0.0016636605,0.01695096,0.022731435,0.015295748,0.018233351,0.90521616],"study_design_scores_gemma":[0.00016108497,0.001371117,0.026562482,0.00094510446,0.00074624864,0.0032511135,0.008833361,0.6641412,0.049813006,0.07703205,0.16676176,0.00038143268],"about_ca_topic_score_codex":0.0056118136,"about_ca_topic_score_gemma":0.014898426,"teacher_disagreement_score":0.009104685,"about_ca_system_score_codex":0.0008189724,"about_ca_system_score_gemma":0.0021723981,"threshold_uncertainty_score":0.011997461},"labels":[],"label_agreement":null},{"id":"W4409787596","doi":"10.61091/jcmcc127a-333","title":"Using Machine Learning Techniques to Improve the Accuracy of Computer Translation in the English Translation of Specialized Terms","year":2025,"lang":"en","type":"article","venue":"Journal of Combinatorial Mathematics and Combinatorial Computing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Translation (biology); Computer science; Machine translation; Natural language processing; Artificial intelligence; Example-based machine translation; Computer-assisted translation; Machine learning; Chemistry","score_opus":0.018315891337762364,"score_gpt":0.29902304069785524,"score_spread":0.2807071493600929,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409787596","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08138445,0.0037021188,0.90036607,0.00078799325,0.00024471333,0.0001009635,0.00021327284,0.00382702,0.009373506],"genre_scores_gemma":[0.63990873,0.0022868393,0.34988987,0.00035260047,0.00018115697,0.00013709572,0.0009531115,0.00048311154,0.005807489],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99842864,0.0006064805,0.00016470566,0.00037050928,0.00033039154,0.00009919623],"domain_scores_gemma":[0.9977106,0.0012421006,0.00019572841,0.00038556062,0.0004388959,0.00002715079],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027293987,0.0011954991,0.00092421786,0.0015502528,0.00056711136,0.0012528147,0.00085392443,0.00086325244,0.00195493],"category_scores_gemma":[0.009613084,0.00029281396,0.0007736931,0.0018093667,0.00050204375,0.0024338935,0.0008205555,0.0010954192,0.0017812711],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019255152,0.00017542836,0.005219531,0.00050169625,0.00019969312,0.00020246925,0.00029782575,0.15217136,0.04055742,0.010485452,0.005273568,0.784723],"study_design_scores_gemma":[0.000027175265,0.00016846019,0.0027079934,0.00007058966,0.00012281937,0.00021199642,0.00009720929,0.93422353,0.041717328,0.012073048,0.008539766,0.00004006294],"about_ca_topic_score_codex":0.0036712864,"about_ca_topic_score_gemma":0.0045083864,"teacher_disagreement_score":0.0036712864,"about_ca_system_score_codex":0.0008951319,"about_ca_system_score_gemma":0.0012289662,"threshold_uncertainty_score":0.014434636},"labels":[],"label_agreement":null},{"id":"W4409787646","doi":"10.61091/jcmcc127a-266","title":"Research on Syntactic Optimization and Semantic Reconstruction Strategies for English Translation Based on Machine Learning","year":2025,"lang":"en","type":"article","venue":"Journal of Combinatorial Mathematics and Combinatorial Computing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Translation (biology); Machine translation; Chemistry","score_opus":0.023560382792047026,"score_gpt":0.31565830878664125,"score_spread":0.2920979259945942,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409787646","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03098638,0.001938931,0.96199006,0.00045515285,0.00008771363,0.000049107402,0.00004537519,0.0005628857,0.0038843357],"genre_scores_gemma":[0.64940757,0.0035472969,0.33861288,0.00035179395,0.0001761848,0.00017249672,0.00050182006,0.00030802787,0.0069219046],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9992112,0.00028622538,0.000071226204,0.00019693699,0.00015655941,0.00007781269],"domain_scores_gemma":[0.999212,0.0003986892,0.000075738586,0.000097025346,0.00019187327,0.000024549263],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011450055,0.0009692102,0.0009197123,0.00080980186,0.0006007906,0.00086867926,0.0008895646,0.0006484786,0.0015324937],"category_scores_gemma":[0.0031092758,0.00034847134,0.0011660252,0.0011719643,0.00074113853,0.0027584932,0.00072155526,0.0011102812,0.00045819805],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012561811,0.00017906078,0.0028689222,0.0005989691,0.00018675305,0.00020609966,0.00042125484,0.27295578,0.017458206,0.04778571,0.0030570128,0.6541567],"study_design_scores_gemma":[0.000022198503,0.0000913463,0.0007278436,0.000024582192,0.00008109227,0.00012397565,0.00009170827,0.96698695,0.009855366,0.01905645,0.0029103528,0.000028118224],"about_ca_topic_score_codex":0.0048642033,"about_ca_topic_score_gemma":0.0034107945,"teacher_disagreement_score":0.0048642033,"about_ca_system_score_codex":0.00081294664,"about_ca_system_score_gemma":0.0014409078,"threshold_uncertainty_score":0.009671807},"labels":[],"label_agreement":null},{"id":"W4409808981","doi":"10.1002/ca.24284","title":"Pulmonalis or Pulmonaris? It's Elementarius, My Dear Watson","year":2025,"lang":"en","type":"article","venue":"Clinical Anatomy","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Saskatchewan; McMaster University; University of Ottawa; Dalhousie University","funders":"","keywords":"Suffix; Linguistics; Medicine; Watson; Limiting; Proper noun; Artificial intelligence; Computer science; Philosophy","score_opus":0.03353194534376824,"score_gpt":0.4030202892912881,"score_spread":0.36948834394751984,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409808981","genre_codex":"commentary","genre_gemma":"commentary","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"commentary","genre_consensus":"commentary","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06551601,0.14915967,0.013414085,0.53627455,0.03766596,0.00006411929,0.0008348679,0.0005219925,0.19654875],"genre_scores_gemma":[0.42328602,0.09715055,0.014981009,0.16328159,0.011348654,0.000090130896,0.000319791,0.0009647086,0.2885775],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9998167,0.000055546523,0.000014807681,0.000047597314,0.000051650924,0.000013820374],"domain_scores_gemma":[0.99964786,0.00018522801,0.00004355948,0.000018717375,0.000063326355,0.000041302414],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003802376,0.00020269665,0.00019046488,0.00035231785,0.00070831843,0.0011602178,0.00023501636,0.0007455617,0.009881226],"category_scores_gemma":[0.0024070218,0.00011621961,0.00012150051,0.00040838713,0.0016471501,0.0029727847,0.0004642332,0.0015818976,0.0035947997],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013985048,0.000025705262,0.007941263,0.00032532605,0.000016762822,0.0020302604,0.0066936743,0.00006624321,0.0019095645,0.07310611,0.6539901,0.2537551],"study_design_scores_gemma":[0.0000052919136,0.000017317461,0.0027343351,0.00017605655,0.000009721871,0.0064900494,0.0042819744,0.000097292024,0.0007026795,0.013208933,0.9722601,0.000016191401],"about_ca_topic_score_codex":0.0030468253,"about_ca_topic_score_gemma":0.007478928,"teacher_disagreement_score":0.009881226,"about_ca_system_score_codex":0.0005074121,"about_ca_system_score_gemma":0.0004241675,"threshold_uncertainty_score":0.03305602},"labels":[],"label_agreement":null},{"id":"W4409916786","doi":"10.1109/tse.2025.3565387","title":"Question Selection for Multimodal Code Search Synthesis Using Probabilistic Version Spaces","year":2025,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"National Natural Science Foundation of China","keywords":"Computer science; Selection (genetic algorithm); Probabilistic logic; Modal; Programming language; Code (set theory); Theoretical computer science; Artificial intelligence; Set (abstract data type)","score_opus":0.01307841608488208,"score_gpt":0.2768731439809722,"score_spread":0.2637947278960901,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409916786","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010780149,0.00013201161,0.97758853,0.00018867693,0.00003236389,0.00017144534,0.0002073731,0.008015975,0.0028835987],"genre_scores_gemma":[0.30966115,0.00012976937,0.6815495,0.00019194059,0.00004816418,0.00059341313,0.0008937071,0.0014922147,0.005440101],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9939803,0.0026439042,0.00043420153,0.0013280434,0.0013415072,0.000272014],"domain_scores_gemma":[0.99019367,0.0074127596,0.00039685096,0.0010203886,0.0007538295,0.00022245852],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00431377,0.001236836,0.0012918463,0.002160809,0.0009557842,0.0027350802,0.0022634175,0.0019001684,0.019643089],"category_scores_gemma":[0.023985554,0.00080375647,0.0018171681,0.00088781555,0.0015101079,0.0043212483,0.004903532,0.0012323944,0.0031172985],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013125453,0.0002567356,0.0023565057,0.0008732995,0.00012934112,0.00072827825,0.0027930194,0.12976028,0.030256303,0.17278086,0.0090611065,0.64969176],"study_design_scores_gemma":[0.0000873324,0.00014137648,0.00027886813,0.00006746634,0.000042581214,0.00020511368,0.00027433297,0.88889617,0.016344894,0.08181558,0.011789792,0.000056557677],"about_ca_topic_score_codex":0.0016606309,"about_ca_topic_score_gemma":0.0018439629,"teacher_disagreement_score":0.019643089,"about_ca_system_score_codex":0.0012506358,"about_ca_system_score_gemma":0.0011594495,"threshold_uncertainty_score":0.06571269},"labels":[],"label_agreement":null},{"id":"W4410087602","doi":"10.1109/wi-iat62293.2024.00029","title":"Unveiling the Source: Differentiating Human and Machine-Generated Texts in a Multilingual Setting","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Wilfrid Laurier University","funders":"","keywords":"Computer science; Natural language processing; Machine translation; Artificial intelligence","score_opus":0.013913648384467282,"score_gpt":0.29174558848264687,"score_spread":0.2778319400981796,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410087602","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5669844,0.0025434147,0.4058344,0.001683795,0.00044984175,0.00023659418,0.0035495525,0.0059026266,0.012815368],"genre_scores_gemma":[0.92613393,0.00035533056,0.06786768,0.00013415159,0.0001816236,0.00006822925,0.002687449,0.00030145788,0.0022702843],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99563557,0.0022371407,0.00031252412,0.0009743197,0.0006328951,0.00020765056],"domain_scores_gemma":[0.9811529,0.010203722,0.002330102,0.0041818023,0.0017324701,0.00039902388],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037868335,0.00083375257,0.0005007331,0.0047069583,0.000894732,0.0028574413,0.00070345873,0.0010949194,0.0015875659],"category_scores_gemma":[0.025029661,0.00024287852,0.00054633815,0.0022099456,0.0011621292,0.004263592,0.0029547585,0.0011493827,0.0020136568],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011795912,0.0003731571,0.081866875,0.00081987964,0.00020960902,0.002279998,0.005285757,0.020412749,0.027179038,0.013407789,0.013503692,0.83348185],"study_design_scores_gemma":[0.000085809974,0.00030056864,0.04697964,0.00033112412,0.00017467361,0.0036400994,0.0044325828,0.7797598,0.06491234,0.044828627,0.054333307,0.00022143275],"about_ca_topic_score_codex":0.001475924,"about_ca_topic_score_gemma":0.0027307111,"teacher_disagreement_score":0.0047069583,"about_ca_system_score_codex":0.0005770051,"about_ca_system_score_gemma":0.0007714443,"threshold_uncertainty_score":0.020026922},"labels":[],"label_agreement":null},{"id":"W4410216840","doi":"10.3765/plsa.v10i1.5926","title":"Access to contextually-determined states in the interpretation of English stative participles","year":2025,"lang":"en","type":"article","venue":"Proceedings of the Linguistic Society of America","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"","keywords":"Interpretation (philosophy); Linguistics; Psychology; Natural language processing; Computer science; Philosophy","score_opus":0.014281392209521007,"score_gpt":0.3134465945750987,"score_spread":0.2991652023655777,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410216840","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7219704,0.0010128386,0.13586111,0.0027589418,0.00022591228,0.00007986293,0.0005645616,0.0010475209,0.13647883],"genre_scores_gemma":[0.99208516,0.0001683825,0.0051272702,0.00013752104,0.000055017677,0.000030887797,0.00014546854,0.00023001885,0.0020202145],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"qualitative","domain_scores_codex":[0.996894,0.0017153766,0.0002830628,0.00044805967,0.00035980754,0.0002996421],"domain_scores_gemma":[0.9960544,0.0022230444,0.00031017393,0.0006255903,0.00066997187,0.000116832525],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003179455,0.00064134,0.0005729915,0.0009774286,0.0021872679,0.006950009,0.0011786906,0.0017432385,0.0046102023],"category_scores_gemma":[0.004686364,0.0013275634,0.000800941,0.0008963865,0.0069691725,0.007839041,0.0032599275,0.0025785337,0.0008613565],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00027470605,0.000057663336,0.0026451298,0.0002093931,0.000031561605,0.0009537795,0.059645656,0.00082228257,0.012677167,0.9105209,0.0011627066,0.010999066],"study_design_scores_gemma":[0.00014187291,0.00017129941,0.008986496,0.00034219146,0.00013254995,0.0010972287,0.029911013,0.009421491,0.01300882,0.89678985,0.03975745,0.00023973853],"about_ca_topic_score_codex":0.0026420325,"about_ca_topic_score_gemma":0.003165496,"teacher_disagreement_score":0.006950009,"about_ca_system_score_codex":0.0017785677,"about_ca_system_score_gemma":0.00086679246,"threshold_uncertainty_score":0.016814768},"labels":[],"label_agreement":null},{"id":"W4410353531","doi":"10.2169/internalmedicine.4834-24","title":"Arrhythmias in Cronkhite-Canada Syndrome","year":2025,"lang":"en","type":"article","venue":"Internal Medicine","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Medicine; Dermatology; General surgery","score_opus":0.00666745626989386,"score_gpt":0.2680242100054716,"score_spread":0.26135675373557776,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410353531","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99120516,0.002277395,0.00062970736,0.0009993264,0.000120265446,0.0000544392,0.00013768475,0.00005115172,0.0045248093],"genre_scores_gemma":[0.9986706,0.00048040843,0.00018821786,0.00019873209,0.000123078,0.000004917057,0.00005988249,0.000003904426,0.00027025345],"study_design_codex":"case_report","study_design_gemma":"case_report","domain_scores_codex":[0.9998305,0.000016631197,0.000017138193,0.00003757729,0.00003629907,0.000061755214],"domain_scores_gemma":[0.9995479,0.00011099388,0.000085548374,0.000015943408,0.00003749963,0.0002020947],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00008612228,0.0006981842,0.0003458236,0.0008064512,0.0009860543,0.00044082684,0.00032440524,0.0015983591,0.0014968279],"category_scores_gemma":[0.0012498291,0.00021073855,0.00022982467,0.00082324335,0.000577385,0.00038419923,0.00039478517,0.00096802344,0.0002465956],"study_design_candidate":"case_report","study_design_consensus":"case_report","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023681702,0.000051912222,0.11485802,0.000046480727,0.000025877947,0.87422675,0.00025953315,0.00010713005,0.0047143563,0.00029370017,0.0007827109,0.004396751],"study_design_scores_gemma":[0.00010134646,0.00015916431,0.14866796,0.000032899115,0.000042448086,0.84750783,0.00038557808,0.0006172866,0.0005999888,0.00040581787,0.001451195,0.000028548784],"about_ca_topic_score_codex":0.012480698,"about_ca_topic_score_gemma":0.017045774,"teacher_disagreement_score":0.012480698,"about_ca_system_score_codex":0.0005807893,"about_ca_system_score_gemma":0.0005408775,"threshold_uncertainty_score":0.024816096},"labels":[],"label_agreement":null},{"id":"W4410512285","doi":"10.1007/978-3-031-51447-0_155-1","title":"Wordlists and Data-Driven Learning","year":2025,"lang":"en","type":"book-chapter","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Computer science","score_opus":0.023025976093062454,"score_gpt":0.2905915697451161,"score_spread":0.2675655936520536,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410512285","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0020963815,0.00539048,0.9646497,0.0009550877,0.00054000656,0.000064907515,0.0013282241,0.0051863575,0.019788954],"genre_scores_gemma":[0.051971078,0.0077345422,0.85233104,0.0008080977,0.00047145816,0.000315282,0.008423957,0.002389572,0.07555506],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9993284,0.00013843672,0.000056626784,0.00017006329,0.0002751277,0.000031494867],"domain_scores_gemma":[0.9979183,0.0014200283,0.000053116622,0.00024300053,0.00032634244,0.000039129118],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010839165,0.0008028552,0.0007103447,0.001863479,0.00044011365,0.0022899953,0.001816714,0.0008168587,0.021225026],"category_scores_gemma":[0.0050618066,0.0005311559,0.0007590327,0.0035622637,0.0009327254,0.0055522597,0.0011053454,0.0016800889,0.011375654],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000041010564,0.000041743635,0.00017554073,0.00030633653,0.000018390485,0.000034830213,0.00006585396,0.008849214,0.0012425649,0.10537278,0.041028205,0.8428235],"study_design_scores_gemma":[0.000015236813,0.000027251994,0.00023654893,0.0001823854,0.000018625375,0.00014970398,0.00007219879,0.11981154,0.0064488407,0.6966094,0.17639752,0.000030728144],"about_ca_topic_score_codex":0.0017626676,"about_ca_topic_score_gemma":0.0025155845,"teacher_disagreement_score":0.021225026,"about_ca_system_score_codex":0.0010010746,"about_ca_system_score_gemma":0.000908131,"threshold_uncertainty_score":0.07100481},"labels":[],"label_agreement":null},{"id":"W4410537502","doi":"10.1145/3736407","title":"CodeUltraFeedback: An LLM-as-a-Judge Dataset for Aligning Large Language Models to Coding Preferences","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Coding (social sciences); Natural language processing; Artificial intelligence; Programming language; Statistics; Mathematics","score_opus":0.0823360169353762,"score_gpt":0.36758500400909394,"score_spread":0.2852489870737177,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410537502","genre_codex":"empirical","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.407223,0.004532687,0.14539485,0.0040328857,0.0023728414,0.003202879,0.2473659,0.15245856,0.03341642],"genre_scores_gemma":[0.33622497,0.0004527857,0.12730294,0.002258604,0.00021620486,0.0033737037,0.51091605,0.006567552,0.012687123],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9922357,0.003972553,0.0005622721,0.0014568007,0.0014009301,0.0003716477],"domain_scores_gemma":[0.9853184,0.008112648,0.0006775443,0.0024428351,0.002658139,0.0007904847],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004622301,0.0032753432,0.0008615781,0.0019060731,0.0012079363,0.0018025776,0.0024735653,0.0029161354,0.0074150073],"category_scores_gemma":[0.030730693,0.0005495879,0.0013105334,0.0013932728,0.0011109551,0.0027889672,0.002731407,0.0035098784,0.008638643],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002893943,0.0018646268,0.026920449,0.0032144694,0.00051393756,0.0011260307,0.00279384,0.041786443,0.03892606,0.005330113,0.5992308,0.27539936],"study_design_scores_gemma":[0.001701753,0.0025344295,0.041017644,0.0005626281,0.00024969495,0.0014762162,0.0028367285,0.57031256,0.05678223,0.015271726,0.30654952,0.00070484425],"about_ca_topic_score_codex":0.010693455,"about_ca_topic_score_gemma":0.029158758,"teacher_disagreement_score":0.010693455,"about_ca_system_score_codex":0.0016142774,"about_ca_system_score_gemma":0.0020728053,"threshold_uncertainty_score":0.024805605},"labels":[],"label_agreement":null},{"id":"W4410552311","doi":"10.1016/j.jss.2025.112493","title":"Variational Prefix Tuning for diverse and accurate code summarization using pre-trained language models","year":2025,"lang":"en","type":"article","venue":"Journal of Systems and Software","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; University of New Brunswick","funders":"Natural Sciences and Engineering Research Council of Canada; Connaught Fund; University of Toronto","keywords":"Automatic summarization; Prefix; Code (set theory); Computer science; Natural language processing; Programming language; Algorithm; Artificial intelligence; Linguistics","score_opus":0.02418294193712048,"score_gpt":0.2990349434856573,"score_spread":0.2748520015485368,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410552311","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.044349868,0.0012841696,0.9387566,0.00049385644,0.00028856972,0.0001269263,0.00083919737,0.011365742,0.0024950537],"genre_scores_gemma":[0.44992375,0.0005153604,0.5300462,0.0006844359,0.00027628045,0.00040615056,0.0076644816,0.0024522983,0.008031059],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99906725,0.00022819614,0.0000569715,0.00033243216,0.00018050509,0.00013464909],"domain_scores_gemma":[0.9976936,0.0014342794,0.000085512314,0.00026699202,0.00042501866,0.00009455921],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013055017,0.0013177678,0.0014062243,0.001316867,0.00088092854,0.0013189894,0.0019078354,0.0018350378,0.004610502],"category_scores_gemma":[0.006056296,0.00076810113,0.0010897845,0.0012482491,0.0007100973,0.0029899778,0.0016227924,0.0029573995,0.0024442459],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006185037,0.00026656606,0.0014063767,0.0003050395,0.00018435724,0.00023484832,0.00026570619,0.3221436,0.031744976,0.008652174,0.020582682,0.6135952],"study_design_scores_gemma":[0.000021175347,0.000038902617,0.0001416382,0.00000837698,0.000018795417,0.00002190855,0.000032960852,0.9906527,0.0033653865,0.0045281476,0.001159398,0.000010615219],"about_ca_topic_score_codex":0.01005664,"about_ca_topic_score_gemma":0.022888795,"teacher_disagreement_score":0.01005664,"about_ca_system_score_codex":0.00090939034,"about_ca_system_score_gemma":0.002362501,"threshold_uncertainty_score":0.019996226},"labels":[],"label_agreement":null},{"id":"W4410552853","doi":"10.1109/saner64311.2025.00070","title":"Exploring the Potential of Llama Models in Automated Code Refinement: A Replication Study","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; Concordia University","funders":"","keywords":"Computer science; Replication (statistics); Code (set theory); Programming language; Mathematics; Statistics","score_opus":0.05302030036310057,"score_gpt":0.3186383970244456,"score_spread":0.265618096661345,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410552853","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"reproducibility","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"reproducibility","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8347199,0.0012939296,0.13839823,0.0019002428,0.00033159045,0.0017450319,0.0012949153,0.012729171,0.0075870845],"genre_scores_gemma":[0.8511473,0.00034020163,0.13852005,0.0007579267,0.0000749827,0.0018320614,0.0021345671,0.002361097,0.0028319224],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9783393,0.014377239,0.0012805505,0.0024480952,0.0030412113,0.0005135976],"domain_scores_gemma":[0.7874101,0.13354085,0.0076236366,0.054681353,0.014984083,0.0017599039],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.023226611,0.0016369645,0.0011470169,0.0013632008,0.0011287996,0.002853413,0.0037128117,0.0021157425,0.003164992],"category_scores_gemma":[0.14980687,0.0010234475,0.0023826407,0.0011478509,0.0022049882,0.0072921,0.0035299538,0.0041550794,0.0017379078],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0126608545,0.011991673,0.11100943,0.005604014,0.0015782529,0.0023153948,0.023057537,0.19701965,0.044710558,0.031792816,0.034676462,0.5235833],"study_design_scores_gemma":[0.0018536481,0.007025648,0.022517689,0.00070222485,0.000906432,0.0010027577,0.003910379,0.84738153,0.027891204,0.023581387,0.06269466,0.00053250755],"about_ca_topic_score_codex":0.0086741485,"about_ca_topic_score_gemma":0.008449577,"teacher_disagreement_score":0.9767734,"about_ca_system_score_codex":0.0018816896,"about_ca_system_score_gemma":0.0029567457,"threshold_uncertainty_score":0.12283552},"labels":[{"model":"gemma","categories":[],"domain":null,"study_design":"simulation_or_modeling","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"low"},{"model":"gpt","categories":[],"domain":null,"study_design":"design_other","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"high"}],"label_agreement":"split"},{"id":"W4410552935","doi":"10.1109/saner64311.2025.00059","title":"AdvFusion: Adapter-based Knowledge Transfer for Code Summarization on Code Language Models","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia, Okanagan Campus; University of British Columbia","funders":"","keywords":"Computer science; Automatic summarization; Adapter (computing); Programming language; Code (set theory); Code generation; Natural language processing; Operating system","score_opus":0.02334172306772232,"score_gpt":0.3071305704208185,"score_spread":0.2837888473530962,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410552935","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02922298,0.0004432369,0.8539271,0.00048763826,0.00013514108,0.00035764504,0.0016254561,0.11055343,0.003247383],"genre_scores_gemma":[0.28825295,0.00043014184,0.6845557,0.0008949861,0.00012078555,0.0007862542,0.012940733,0.003130239,0.008888245],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9981421,0.00047689493,0.00013017579,0.00066405575,0.00043609896,0.00015060617],"domain_scores_gemma":[0.9952402,0.0017409374,0.00020728582,0.0017690008,0.00087894266,0.00016365663],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00326111,0.0022020056,0.0011044729,0.0020107687,0.0006237371,0.0014073806,0.003770463,0.0020999617,0.00690411],"category_scores_gemma":[0.011841403,0.00078132807,0.0014213333,0.0017552703,0.00084390794,0.007112716,0.0053138104,0.003036472,0.003991165],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003794209,0.0004747064,0.0024523295,0.00026221407,0.00017502205,0.00017422086,0.00032061987,0.07018561,0.0141640045,0.0038213953,0.029594805,0.8779957],"study_design_scores_gemma":[0.000052304003,0.00012085526,0.0006823631,0.000019218569,0.00003590173,0.00007263091,0.0000827933,0.9621942,0.018727252,0.010359982,0.0076195896,0.000033012395],"about_ca_topic_score_codex":0.007917233,"about_ca_topic_score_gemma":0.009120499,"teacher_disagreement_score":0.007917233,"about_ca_system_score_codex":0.0013560518,"about_ca_system_score_gemma":0.0017641373,"threshold_uncertainty_score":0.023096502},"labels":[],"label_agreement":null},{"id":"W4410553371","doi":"10.1109/saner64311.2025.00036","title":"Preprocessing is All You Need: Boosting the Performance of Log Parsers with a General Preprocessing Framework","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Preprocessor; Boosting (machine learning); Computer science; Parsing; Artificial intelligence; Data pre-processing; Natural language processing; Data mining","score_opus":0.011891278941740182,"score_gpt":0.27762287998126356,"score_spread":0.26573160103952337,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410553371","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.085986584,0.0019503739,0.5534464,0.003221719,0.00041946807,0.0006915457,0.004488226,0.34261268,0.0071830726],"genre_scores_gemma":[0.2949514,0.00077897764,0.66367584,0.002581176,0.00018383627,0.0005553975,0.016749224,0.0155215375,0.0050026453],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99328387,0.0021345126,0.0006100654,0.0015449506,0.0018337105,0.0005929598],"domain_scores_gemma":[0.9807478,0.009303783,0.0010325585,0.0055029676,0.0029954591,0.0004173561],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007262018,0.0027305733,0.0015229182,0.0030198048,0.0014728486,0.0035610993,0.004379265,0.002064173,0.003838231],"category_scores_gemma":[0.028740948,0.0014709599,0.00209579,0.0029870595,0.0018371932,0.0109801255,0.0041223355,0.0041167876,0.0037172858],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016352987,0.0014963257,0.018951157,0.0015306367,0.00027449624,0.0010164244,0.0014852888,0.05003772,0.059125405,0.01828813,0.10932554,0.7368336],"study_design_scores_gemma":[0.00042614102,0.0009305406,0.010777935,0.00021178198,0.0004142075,0.0011907308,0.0007478708,0.7383482,0.11825396,0.042916253,0.085416995,0.00036531064],"about_ca_topic_score_codex":0.008305973,"about_ca_topic_score_gemma":0.011046898,"teacher_disagreement_score":0.008305973,"about_ca_system_score_codex":0.0017766487,"about_ca_system_score_gemma":0.0066559296,"threshold_uncertainty_score":0.038405716},"labels":[],"label_agreement":null},{"id":"W4410573952","doi":"10.4018/jdm.377526","title":"FADE","year":2025,"lang":"en","type":"article","venue":"Journal of Database Management","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"IBM (Canada); University of Alberta","funders":"","keywords":"Computer science","score_opus":0.008869374136667394,"score_gpt":0.2861501527559221,"score_spread":0.2772807786192547,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410573952","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.092201635,0.0057723955,0.6900308,0.0036939685,0.0025621532,0.0024689925,0.018101124,0.1465922,0.03857676],"genre_scores_gemma":[0.54670197,0.002235885,0.37161937,0.003718959,0.0004729499,0.0010405369,0.018112335,0.0060529048,0.05004501],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9952744,0.00066698994,0.0004909922,0.0010957518,0.0021254872,0.00034632732],"domain_scores_gemma":[0.9727092,0.009976002,0.0017197619,0.008088013,0.007001805,0.00050531235],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.006411219,0.0015424854,0.0013332138,0.0038394094,0.0010488998,0.003045975,0.0023868198,0.0012679821,0.016191483],"category_scores_gemma":[0.038240105,0.00064134854,0.0010796511,0.001977598,0.0009206021,0.0056093764,0.0030940264,0.0015655435,0.008691366],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018016761,0.00030267966,0.022514924,0.0012956077,0.00025786992,0.0005654706,0.0009788782,0.0069698226,0.012662842,0.012458257,0.10660057,0.8335914],"study_design_scores_gemma":[0.00029682313,0.0013837293,0.02642122,0.0009555661,0.0004254797,0.0032188813,0.0020980185,0.21476012,0.121329136,0.05855445,0.570119,0.00043758683],"about_ca_topic_score_codex":0.003390866,"about_ca_topic_score_gemma":0.0042408723,"teacher_disagreement_score":0.9838085,"about_ca_system_score_codex":0.00099715,"about_ca_system_score_gemma":0.0013947603,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W4410672227","doi":"10.1075/term.00084.vez","title":"When LMF and TMF meet","year":2025,"lang":"en","type":"article","venue":"Terminology International Journal of Theoretical and Applied Issues in Specialized Communication","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"Fundação para a Ciência e a Tecnologia; HORIZON EUROPE Framework Programme; CHIST-ERA; European Commission; Universidade Nova de Lisboa; Università degli Studi di Padova","keywords":"Computer science","score_opus":0.008692772929978398,"score_gpt":0.31466907023021146,"score_spread":0.30597629730023307,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410672227","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20402119,0.0017948602,0.20771188,0.027599752,0.0010384615,0.00048988604,0.0015190623,0.0014436968,0.55438125],"genre_scores_gemma":[0.9461541,0.0003115673,0.037727512,0.0013333526,0.0003084798,0.0003183781,0.0011804862,0.00035804027,0.012308017],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.986949,0.004492746,0.00093569735,0.002343752,0.002362187,0.002916589],"domain_scores_gemma":[0.96915543,0.016477069,0.0022840535,0.0036844881,0.0061646863,0.0022342307],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010743527,0.00044603256,0.0010933111,0.0032694004,0.0041799806,0.011488334,0.0019337917,0.0058019133,0.033738993],"category_scores_gemma":[0.08216742,0.00044129163,0.0008456732,0.0027038185,0.0059796073,0.028956199,0.0070070126,0.0032857365,0.006822317],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003092363,0.000110158486,0.008533567,0.00021242759,0.000023392264,0.0010047258,0.004715252,0.002255109,0.0021602805,0.89568484,0.016465751,0.06852514],"study_design_scores_gemma":[0.00004249297,0.00008618045,0.0054418817,0.00018085496,0.000018292565,0.0013268193,0.012746993,0.016684044,0.0016165179,0.8997738,0.062020186,0.000062001745],"about_ca_topic_score_codex":0.0069218623,"about_ca_topic_score_gemma":0.003485625,"teacher_disagreement_score":0.033738993,"about_ca_system_score_codex":0.004529876,"about_ca_system_score_gemma":0.0042000473,"threshold_uncertainty_score":0.11286819},"labels":[],"label_agreement":null},{"id":"W4410705334","doi":"10.29173/cais2020","title":"Applying LLMs and Semantic Technologies for Data Extraction in Literature Reviews","year":2025,"lang":"fr","type":"article","venue":"Proceedings of the Annual Conference of CAIS / Actes du congrès annuel de l ACSI","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Semantic technology; Data extraction; Information retrieval; Data science; Natural language processing; Semantic Web; Semantic computing; Political science; MEDLINE","score_opus":0.0450675097573206,"score_gpt":0.3213783706396935,"score_spread":0.2763108608823729,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410705334","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14522287,0.012007393,0.7154189,0.01157188,0.0006186148,0.02369237,0.046454392,0.022737436,0.022276198],"genre_scores_gemma":[0.14304172,0.0024193355,0.82706577,0.0010365307,0.000071520066,0.011493729,0.01296523,0.00044703155,0.0014591811],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.95343995,0.027960861,0.008804012,0.0031750745,0.006044308,0.00057577586],"domain_scores_gemma":[0.77673924,0.18043543,0.0119256275,0.015733266,0.014171903,0.0009946054],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.05936283,0.0013762658,0.0014789574,0.018390207,0.0012058951,0.0067898678,0.0017960191,0.0013261378,0.0044835806],"category_scores_gemma":[0.1977131,0.0011653362,0.0042602997,0.013958421,0.0010252085,0.008322875,0.0054034526,0.00124007,0.0016495829],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002913289,0.00046688676,0.036866896,0.05212862,0.002553411,0.00067008234,0.02083955,0.010627926,0.019437753,0.02271518,0.020158134,0.8106224],"study_design_scores_gemma":[0.0023482174,0.002528246,0.079735786,0.031193122,0.009259191,0.0018084348,0.026126962,0.1620875,0.080744445,0.07870643,0.52449065,0.00097099127],"about_ca_topic_score_codex":0.0067891222,"about_ca_topic_score_gemma":0.013515456,"teacher_disagreement_score":0.9406372,"about_ca_system_score_codex":0.0044483542,"about_ca_system_score_gemma":0.012417124,"threshold_uncertainty_score":0.3139444},"labels":[],"label_agreement":null},{"id":"W4410730166","doi":"10.1007/978-3-031-91585-7_24","title":"Embedding Geometries of Contrastive Language-Image Pre-training","year":2025,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Glycemic Index Laboratories","funders":"","keywords":"Computer science; Embedding; Training (meteorology); Artificial intelligence; Natural language processing; Image (mathematics)","score_opus":0.010926765436282486,"score_gpt":0.28872016188290783,"score_spread":0.2777933964466254,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410730166","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.052516248,0.0002074344,0.9226054,0.00030350988,0.00011627241,0.00005027899,0.00038389128,0.0011520237,0.02266506],"genre_scores_gemma":[0.6472198,0.00035483876,0.31346548,0.00017294104,0.00012521925,0.000102019454,0.00089538685,0.0009973705,0.03666689],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99972194,0.000080551064,0.000016214795,0.0000735866,0.00006776297,0.00003982501],"domain_scores_gemma":[0.99927324,0.00030399457,0.000047693815,0.00016881712,0.00014324249,0.00006312369],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00033694124,0.00048841967,0.0003301263,0.00035037947,0.00025323918,0.00088529824,0.00078340335,0.00056778576,0.010964018],"category_scores_gemma":[0.0017557223,0.0003183071,0.00033838968,0.00021034507,0.000764614,0.0011896946,0.001158377,0.0011799638,0.0028910777],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00044261038,0.000100696874,0.00052005704,0.0002518523,0.0000221557,0.00029837887,0.0003914387,0.07368283,0.058985475,0.54902875,0.01098046,0.3052952],"study_design_scores_gemma":[0.000035382167,0.00034350954,0.00078406936,0.000061441584,0.000024151033,0.000532476,0.00021372199,0.54793876,0.061096773,0.3663176,0.022600787,0.000051203908],"about_ca_topic_score_codex":0.0007209191,"about_ca_topic_score_gemma":0.0011711495,"teacher_disagreement_score":0.010964018,"about_ca_system_score_codex":0.00044268643,"about_ca_system_score_gemma":0.00026596745,"threshold_uncertainty_score":0.036678255},"labels":[],"label_agreement":null},{"id":"W4410818076","doi":"10.36227/techrxiv.174495034.42657551/v2","title":"LLM-in-the-Loop: Replicating Human Insight with LLMs for Better Machine Learning Applications","year":2025,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Loop (graph theory); Computer science; Human-in-the-loop; Psychology; Artificial intelligence; Mathematics","score_opus":0.01782143569512096,"score_gpt":0.3059178948561969,"score_spread":0.28809645916107596,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410818076","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007690917,0.0026577653,0.9581407,0.01163609,0.0004568034,0.0002849221,0.0003166165,0.0045756646,0.014240516],"genre_scores_gemma":[0.2956374,0.001962453,0.68653756,0.005239757,0.0009349976,0.0009336183,0.00076860574,0.0016197888,0.006365859],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9811025,0.011829378,0.00074385677,0.002785825,0.0029941427,0.0005443514],"domain_scores_gemma":[0.948137,0.02615787,0.002685112,0.018378835,0.0034658276,0.0011753408],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.020141799,0.0013389706,0.0014759599,0.0028770098,0.0022574058,0.008902257,0.0056798663,0.004828516,0.008861592],"category_scores_gemma":[0.07798557,0.0008484057,0.0015528891,0.0029523952,0.009077863,0.01864617,0.015231954,0.0056144833,0.004677589],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00031380943,0.0003148085,0.0035296092,0.0019015099,0.00022691625,0.00022556462,0.004081588,0.030165346,0.006846438,0.5335232,0.026959628,0.39191163],"study_design_scores_gemma":[0.0000670923,0.00012349957,0.0005584835,0.0003089429,0.000049947514,0.00013497204,0.0006686952,0.11770987,0.0042419187,0.8110439,0.06501558,0.00007711693],"about_ca_topic_score_codex":0.0022548125,"about_ca_topic_score_gemma":0.0028660644,"teacher_disagreement_score":0.020141799,"about_ca_system_score_codex":0.0031938264,"about_ca_system_score_gemma":0.0049250317,"threshold_uncertainty_score":0.10652125},"labels":[],"label_agreement":null},{"id":"W4410822982","doi":"10.1609/aaaiss.v5i1.35613","title":"Dialectic Preference Bias in Large Language Models","year":2025,"lang":"en","type":"article","venue":"Proceedings of the AAAI Symposium Series","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute; York University","funders":"Natural Sciences and Engineering Research Council of Canada; Canada First Research Excellence Fund","keywords":"Preference; Dialectic; Linguistics; Philosophy; Mathematics; Epistemology; Statistics","score_opus":0.018722294054562987,"score_gpt":0.25396809364420436,"score_spread":0.23524579958964137,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410822982","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5864554,0.0005517785,0.40176314,0.0013320956,0.000040424857,0.00013309115,0.0004838794,0.0007843296,0.00845588],"genre_scores_gemma":[0.96843463,0.00009196889,0.029950745,0.00016876448,0.000019509655,0.00008510649,0.00027007997,0.000082817845,0.00089637714],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9925817,0.00539997,0.00032169058,0.0008409763,0.0006401359,0.00021547744],"domain_scores_gemma":[0.97192603,0.022970332,0.0016245772,0.0019702401,0.0010872945,0.00042147771],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009378625,0.0004919954,0.00068329915,0.00093944784,0.0006882446,0.002965601,0.00087905617,0.00066711573,0.003200111],"category_scores_gemma":[0.041803785,0.00037302956,0.0008499038,0.00075712235,0.0011912885,0.004105503,0.0020213232,0.0014269678,0.0006232552],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0023134155,0.00048078017,0.1247967,0.000806557,0.0010072113,0.0009118786,0.012055458,0.2800471,0.017595528,0.31712273,0.0055678133,0.23729478],"study_design_scores_gemma":[0.000052949592,0.00012277477,0.0055821775,0.0000508478,0.000067395355,0.00017732197,0.0009929313,0.72223634,0.0027792822,0.2647264,0.0031622325,0.00004938599],"about_ca_topic_score_codex":0.002219075,"about_ca_topic_score_gemma":0.0033955793,"teacher_disagreement_score":0.009378625,"about_ca_system_score_codex":0.0011539832,"about_ca_system_score_gemma":0.00070035394,"threshold_uncertainty_score":0.04959953},"labels":[],"label_agreement":null},{"id":"W4411112917","doi":"10.18653/v1/2025.findings-naacl.35","title":"SIMPLOT: Enhancing Chart Question Answering by Distilling Essentials","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"Institute for Information and Communications Technology Promotion; Ministry of Science and ICT, South Korea; National Research Foundation of Korea; National Research Foundation","keywords":"Computer science; Chart; Question answering; Information retrieval; Statistics; Mathematics","score_opus":0.004315419184712681,"score_gpt":0.276673815474003,"score_spread":0.2723583962892903,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411112917","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01353011,0.0005233867,0.94088006,0.0006578001,0.00016263954,0.00042582417,0.0050752647,0.034759156,0.0039857714],"genre_scores_gemma":[0.11105808,0.0004345774,0.86976105,0.00043886364,0.0000943221,0.00023105976,0.013385862,0.000868513,0.00372764],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9982332,0.00034512472,0.00010951866,0.0006462455,0.0005651369,0.00010077975],"domain_scores_gemma":[0.9970024,0.001597264,0.00021242906,0.0005040693,0.0005368562,0.00014694463],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015227586,0.0020302122,0.000863851,0.0029870763,0.0008065185,0.0025565445,0.0023771431,0.0015212195,0.010331945],"category_scores_gemma":[0.009589007,0.00047480446,0.002184916,0.0015763832,0.0008982402,0.0061009843,0.002673259,0.0018279447,0.002875455],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000628007,0.00050189294,0.0057234345,0.0017326357,0.00014832836,0.0007271817,0.001548813,0.03816783,0.043470718,0.04168985,0.072070986,0.79359037],"study_design_scores_gemma":[0.0000874001,0.00021423717,0.0019315098,0.00017127421,0.00011156276,0.0004734809,0.0006032487,0.8224932,0.038890522,0.069444664,0.06548601,0.000093011724],"about_ca_topic_score_codex":0.010385908,"about_ca_topic_score_gemma":0.017054707,"teacher_disagreement_score":0.010385908,"about_ca_system_score_codex":0.0012128968,"about_ca_system_score_gemma":0.002241233,"threshold_uncertainty_score":0.03456384},"labels":[],"label_agreement":null},{"id":"W4411113084","doi":"10.18653/v1/2025.findings-naacl.168","title":"Language Modeling with Editable External Knowledge","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; National Science Foundation","keywords":"Computer science; Human–computer interaction","score_opus":0.008067877120099843,"score_gpt":0.279647331919547,"score_spread":0.2715794547994472,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411113084","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01195714,0.00021047938,0.9751771,0.0004480256,0.00008282054,0.00011069779,0.0015196819,0.005167793,0.0053261747],"genre_scores_gemma":[0.44796503,0.00060422276,0.5242463,0.00046905098,0.00018547107,0.00049731694,0.0063221785,0.0014532497,0.018257216],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9980963,0.00071624847,0.00012559468,0.00050004874,0.000453523,0.000108212094],"domain_scores_gemma":[0.9954432,0.0023348269,0.0003460872,0.001248435,0.0005155288,0.00011194288],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017818632,0.00092937646,0.0006457606,0.001213021,0.00047916314,0.0027749569,0.0025271801,0.0012710777,0.005324288],"category_scores_gemma":[0.008658877,0.0006454459,0.0020108444,0.00097231835,0.0008625502,0.0032873861,0.0017060805,0.001981124,0.0021771456],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014696406,0.0001489289,0.0021172315,0.0001798016,0.00016429665,0.00054257596,0.00044159434,0.815808,0.0029326347,0.05998463,0.00825992,0.109273426],"study_design_scores_gemma":[0.000011842978,0.000014799008,0.00012880398,0.000010114844,0.000017559363,0.00005368468,0.000020445063,0.96574867,0.0010299563,0.026857726,0.0060966993,0.000009695075],"about_ca_topic_score_codex":0.010767677,"about_ca_topic_score_gemma":0.016983343,"teacher_disagreement_score":0.010767677,"about_ca_system_score_codex":0.0010165733,"about_ca_system_score_gemma":0.0010860132,"threshold_uncertainty_score":0.021409988},"labels":[],"label_agreement":null},{"id":"W4411113116","doi":"10.18653/v1/2025.findings-naacl.191","title":"Multilingual Blending: Large Language Model Safety Alignment Evaluation with Language Mixture","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Japan Society for the Promotion of Science; Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Natural language processing; Language model; Artificial intelligence; Programming language","score_opus":0.009740026035793837,"score_gpt":0.3144537245929391,"score_spread":0.30471369855714525,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411113116","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7734669,0.0014341857,0.19205207,0.0008793448,0.00047520775,0.001270518,0.0035716728,0.020686325,0.0061638276],"genre_scores_gemma":[0.834795,0.00025811963,0.15028681,0.00044933442,0.00007919924,0.00066683575,0.010216353,0.0012961031,0.0019521714],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.992442,0.0046133557,0.00067228166,0.0012135619,0.0008057795,0.00025300897],"domain_scores_gemma":[0.9850176,0.010000837,0.0006354382,0.002115601,0.0016780337,0.00055251276],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010543936,0.0019821113,0.0007116063,0.0012925601,0.00091592356,0.0018948258,0.0019581125,0.0015594118,0.0044400967],"category_scores_gemma":[0.026768548,0.00062841194,0.0013923959,0.0008438126,0.0011588251,0.005118812,0.0038684004,0.0027833183,0.0020303666],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0074037216,0.0036482643,0.03010137,0.0017755337,0.0013734216,0.0009784709,0.0034153245,0.3880173,0.037304014,0.009949275,0.023195453,0.49283785],"study_design_scores_gemma":[0.0005465093,0.0020900054,0.0036982354,0.000069461945,0.00024985502,0.00033716488,0.001165728,0.9540205,0.024992766,0.0061920295,0.006508609,0.00012910552],"about_ca_topic_score_codex":0.0067303614,"about_ca_topic_score_gemma":0.005689167,"teacher_disagreement_score":0.010543936,"about_ca_system_score_codex":0.0012097948,"about_ca_system_score_gemma":0.0017924545,"threshold_uncertainty_score":0.05576235},"labels":[],"label_agreement":null},{"id":"W4411113549","doi":"10.18653/v1/2024.inlg-main.46","title":"Generating Attractive Ad Text by Facilitating the Reuse of Landing Page Expressions","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Atomic Energy of Canada Limited","keywords":"Reuse; Computer science; World Wide Web; Information retrieval; Human–computer interaction; Engineering","score_opus":0.020773276630857613,"score_gpt":0.30336415052900195,"score_spread":0.28259087389814436,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411113549","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.34410343,0.00071637705,0.601897,0.00071458926,0.00043132438,0.0010743678,0.0012514156,0.019975618,0.029835917],"genre_scores_gemma":[0.73063225,0.00023558295,0.25242254,0.00023900595,0.00010570987,0.00028384067,0.0016241813,0.0013568711,0.013099964],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991429,0.00036607572,0.000056848596,0.00016036879,0.00023478096,0.000039091145],"domain_scores_gemma":[0.9944226,0.0038091328,0.00031230325,0.00053125084,0.00080102804,0.00012372517],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009683768,0.00091148703,0.00041163282,0.00058907375,0.00020922534,0.00096775824,0.0006875195,0.000511735,0.006377283],"category_scores_gemma":[0.007772449,0.00020775133,0.00036680314,0.0003214534,0.000320084,0.0013186153,0.0006071557,0.0006727268,0.0030972296],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00090231764,0.0009183115,0.0045585465,0.0010026421,0.000066834225,0.0008969704,0.00083613937,0.03187637,0.14619437,0.010444254,0.024010606,0.77829254],"study_design_scores_gemma":[0.00021050933,0.0006214487,0.0033007523,0.0000685646,0.000116645766,0.0008550333,0.0002997533,0.802782,0.14769803,0.012448578,0.03151221,0.00008643283],"about_ca_topic_score_codex":0.00055680063,"about_ca_topic_score_gemma":0.000790986,"teacher_disagreement_score":0.006377283,"about_ca_system_score_codex":0.00029280642,"about_ca_system_score_gemma":0.00032614492,"threshold_uncertainty_score":0.021334112},"labels":[],"label_agreement":null},{"id":"W4411113567","doi":"10.18653/v1/2024.inlg-genchal.5","title":"pyrealb at the GEM’24 Data-to-text Task: Symbolic English Text Generation from RDF Triples","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Discovery Centre","funders":"","keywords":"RDF; Computer science; Task (project management); Natural language processing; Text generation; Information retrieval; Artificial intelligence; Semantic Web","score_opus":0.040794522273139136,"score_gpt":0.2924567835231798,"score_spread":0.25166226125004065,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411113567","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01996657,0.00042056752,0.48386702,0.0033325201,0.0012981544,0.0015190204,0.07084227,0.3778529,0.040900964],"genre_scores_gemma":[0.13218445,0.0003098581,0.5705219,0.0012429609,0.0002521609,0.0017572646,0.21263468,0.05271912,0.028377628],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99682176,0.0011187517,0.00022319941,0.00071180105,0.00085476425,0.00026973127],"domain_scores_gemma":[0.995876,0.0015357267,0.00011907688,0.0013479886,0.00071924547,0.00040198228],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037350832,0.0018149293,0.0011037125,0.0015634148,0.0016921713,0.0021912751,0.002147165,0.0016998709,0.0696078],"category_scores_gemma":[0.012920509,0.00082668883,0.0012346538,0.0011335986,0.0010138453,0.0043491847,0.0070954603,0.002535432,0.0405192],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00078016776,0.00040169805,0.0014718306,0.0012202965,0.000105174804,0.000889493,0.0013765062,0.00459971,0.021790523,0.018707607,0.6961623,0.25249472],"study_design_scores_gemma":[0.0006453831,0.00028463537,0.002704111,0.00031648346,0.00007688641,0.0011571313,0.0015994343,0.12037997,0.07333107,0.07153686,0.72775245,0.00021561702],"about_ca_topic_score_codex":0.004168217,"about_ca_topic_score_gemma":0.0052787317,"teacher_disagreement_score":0.0696078,"about_ca_system_score_codex":0.0009996092,"about_ca_system_score_gemma":0.0024164496,"threshold_uncertainty_score":0.23286128},"labels":[],"label_agreement":null},{"id":"W4411119057","doi":"10.18653/v1/2025.repl4nlp-1","title":"Proceedings of the 10th Workshop on Representation Learning for NLP (RepL4NLP-2025)","year":2025,"lang":"en","type":"paratext","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Bundesministerium für Bildung und Forschung; European Commission; Azrieli Foundation; Open Philanthropy Project; UK Research and Innovation; National Science Foundation","keywords":"Natural language processing; Computer science; Artificial intelligence; Representation (politics); Political science","score_opus":0.024223393349253896,"score_gpt":0.33140671212995704,"score_spread":0.30718331878070315,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411119057","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011256829,0.031555854,0.7417165,0.031228064,0.020368112,0.0015748993,0.033900145,0.072988994,0.055410627],"genre_scores_gemma":[0.040621825,0.016492032,0.54555833,0.009913193,0.0045956518,0.0032990288,0.24289326,0.021642104,0.1149846],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9897893,0.0048622447,0.00077856804,0.0021880164,0.0017367258,0.0006452057],"domain_scores_gemma":[0.9862855,0.0064482405,0.0002794724,0.003882837,0.0021043553,0.0009995752],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012281197,0.0041577183,0.0036372617,0.0029709365,0.0014424771,0.008281058,0.008022809,0.004995269,0.09328277],"category_scores_gemma":[0.024614278,0.0018339667,0.0037853965,0.0027895907,0.0020822713,0.014367196,0.009532961,0.008808618,0.05463525],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00048382083,0.0004266938,0.00040039315,0.0009384493,0.00026047,0.0003102449,0.0003862839,0.0051615555,0.0022536363,0.01307924,0.71002203,0.26627722],"study_design_scores_gemma":[0.00021836351,0.0002353097,0.001006854,0.00080914993,0.00014294621,0.00038519475,0.0003017346,0.055892594,0.004015812,0.05906381,0.87783545,0.000092686234],"about_ca_topic_score_codex":0.009184893,"about_ca_topic_score_gemma":0.012261574,"teacher_disagreement_score":0.09328277,"about_ca_system_score_codex":0.0033580528,"about_ca_system_score_gemma":0.0044744443,"threshold_uncertainty_score":0.3120619},"labels":[],"label_agreement":null},{"id":"W4411119481","doi":"10.18653/v1/2025.naacl-long.338","title":"Pula: Training Large Language Models for Setswana","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Impact","funders":"Clemson University; National Science Foundation","keywords":"Computer science; Training (meteorology); Natural language processing","score_opus":0.022048698075191973,"score_gpt":0.31352790643042633,"score_spread":0.29147920835523433,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411119481","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.062141187,0.0081582405,0.6483354,0.0035664241,0.0033594673,0.00089616165,0.029466625,0.22841208,0.015664425],"genre_scores_gemma":[0.2882746,0.0022591571,0.5303735,0.0026258219,0.0005703479,0.0027361312,0.1401965,0.0108350385,0.022128925],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99836713,0.00065985793,0.000103915656,0.0005719113,0.00019007732,0.000107181484],"domain_scores_gemma":[0.99685067,0.0021080233,0.00005612875,0.0005489463,0.0003100444,0.00012618134],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00328663,0.0035265973,0.0016335852,0.0017692342,0.00180501,0.0029487466,0.004550315,0.002800811,0.017002285],"category_scores_gemma":[0.008283102,0.0023532787,0.0036252476,0.0014541293,0.0007320791,0.0053566964,0.0040044943,0.007017066,0.016248524],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015721318,0.00090484775,0.0041642133,0.0008163388,0.001985316,0.0005861406,0.00041258396,0.14354277,0.0052954657,0.00692681,0.2901396,0.5436538],"study_design_scores_gemma":[0.00035224893,0.00017921189,0.0006972771,0.00008738722,0.00020566427,0.000121571386,0.00017862748,0.9528336,0.004408299,0.018973866,0.02190273,0.000059475835],"about_ca_topic_score_codex":0.013432349,"about_ca_topic_score_gemma":0.030504676,"teacher_disagreement_score":0.017002285,"about_ca_system_score_codex":0.0015569897,"about_ca_system_score_gemma":0.0024092991,"threshold_uncertainty_score":0.05687833},"labels":[],"label_agreement":null},{"id":"W4411119714","doi":"10.18653/v1/2025.naacl-long.613","title":"JAWAHER: A Multidialectal Dataset of Arabic Proverbs for LLM Benchmarking","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Benchmarking; Arabic; Computer science; Natural language processing; Artificial intelligence; Information retrieval; Linguistics; Philosophy","score_opus":0.008947348003498994,"score_gpt":0.2973077930139572,"score_spread":0.2883604450104582,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411119714","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0691063,0.0056494623,0.012590366,0.0018789789,0.0012600295,0.00093968335,0.8337311,0.041274708,0.03356937],"genre_scores_gemma":[0.04258007,0.00050421804,0.018316686,0.0003453661,0.00012694263,0.0005928377,0.93030316,0.0010606899,0.0061699273],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99840957,0.0003950059,0.0002163337,0.00031404497,0.00051605806,0.00014890678],"domain_scores_gemma":[0.9965437,0.00091266667,0.00022013302,0.00090410636,0.0010484748,0.00037093303],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013759263,0.0020258885,0.0010472499,0.0071880347,0.0018947781,0.001968449,0.002600086,0.0026907953,0.018978644],"category_scores_gemma":[0.0072945966,0.00051085284,0.0010205825,0.003645954,0.00073227787,0.0032421565,0.0037819385,0.0017357243,0.027542721],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011097332,0.0006542907,0.0065231137,0.0028555607,0.00019973321,0.0010738142,0.0009381283,0.0021865712,0.012944466,0.004243645,0.8510517,0.11621917],"study_design_scores_gemma":[0.0008103573,0.0002864527,0.028673291,0.0005712769,0.00015154511,0.0015848271,0.002210645,0.02503155,0.018821448,0.0043607014,0.91726583,0.00023209909],"about_ca_topic_score_codex":0.0097838715,"about_ca_topic_score_gemma":0.017027277,"teacher_disagreement_score":0.018978644,"about_ca_system_score_codex":0.0012211069,"about_ca_system_score_gemma":0.0015766751,"threshold_uncertainty_score":0.063489914},"labels":[],"label_agreement":null},{"id":"W4411120121","doi":"10.18653/v1/2025.naacl-long.427","title":"THREAD: Thinking Deeper with Recursive Spawning","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Thread (computing); Computer science; Parallel computing; Programming language","score_opus":0.0056948474130138795,"score_gpt":0.2551516932529223,"score_spread":0.24945684583990843,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411120121","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008066592,0.0013390396,0.9604847,0.004580123,0.0008062088,0.000100090074,0.00016580829,0.005546787,0.018910704],"genre_scores_gemma":[0.23491521,0.0021408133,0.7217086,0.003295816,0.0011248474,0.00050288084,0.00058042665,0.006449425,0.029282004],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99657094,0.0015040674,0.00020345113,0.0006717576,0.00066477695,0.00038488908],"domain_scores_gemma":[0.99056655,0.005641201,0.00030418878,0.0023811099,0.0007590236,0.00034784057],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006210697,0.0012234219,0.0011824927,0.0013577705,0.0029295846,0.0059663886,0.002937625,0.0023737326,0.02404447],"category_scores_gemma":[0.021319551,0.0015621128,0.0028990612,0.0013674059,0.0064591058,0.03000622,0.0058518825,0.005847784,0.0045173923],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001587924,0.000053786825,0.00064492784,0.00020290304,0.000052589236,0.0001366183,0.00092983665,0.007040711,0.0009894483,0.92870647,0.019151265,0.041932635],"study_design_scores_gemma":[0.00004662183,0.00001917011,0.00004921294,0.000055204935,0.00004215757,0.000067131405,0.000130535,0.03454519,0.0012662546,0.93147296,0.032280415,0.000025003514],"about_ca_topic_score_codex":0.004403477,"about_ca_topic_score_gemma":0.0064533064,"teacher_disagreement_score":0.02404447,"about_ca_system_score_codex":0.0017888459,"about_ca_system_score_gemma":0.002092425,"threshold_uncertainty_score":0.080436826},"labels":[{"model":"gemma","categories":[],"domain":null,"study_design":"theoretical_or_conceptual","genre":"methods","about_ca_system":false,"about_ca_topic":false,"confidence":"low"},{"model":"gpt","categories":[],"domain":null,"study_design":"theoretical_or_conceptual","genre":"methods","about_ca_system":false,"about_ca_topic":false,"confidence":"low"}],"label_agreement":"agree"},{"id":"W4411120475","doi":"10.18653/v1/2025.findings-naacl.238","title":"Effective Self-Mining of In-Context Examples for Unsupervised Machine Translation with LLMs","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Computer science; Context (archaeology); Machine translation; Translation (biology); Artificial intelligence; Machine learning; Natural language processing; History; Chemistry","score_opus":0.010366034599708732,"score_gpt":0.26542093057702726,"score_spread":0.2550548959773185,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411120475","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06631993,0.0012048207,0.9095337,0.0005749934,0.00016552712,0.00023539498,0.0012297663,0.01839332,0.0023426684],"genre_scores_gemma":[0.33117872,0.0002700228,0.6561981,0.000629255,0.0001348492,0.0005540862,0.007686356,0.0010889347,0.0022596845],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9978829,0.0009508184,0.00019519587,0.00055733434,0.00031931303,0.000094468356],"domain_scores_gemma":[0.9954674,0.0023964213,0.00027684853,0.0010303272,0.0007328319,0.00009616578],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018390863,0.0014395982,0.0014605013,0.0014158181,0.0009133936,0.0012247559,0.0019853786,0.0015180901,0.0023332813],"category_scores_gemma":[0.0111604305,0.000607686,0.001326506,0.001438438,0.0008647158,0.0027615894,0.0022763817,0.0018827342,0.003122605],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006105434,0.00041871975,0.006413708,0.0006558733,0.0002224336,0.00055807445,0.00075821724,0.09864751,0.030492462,0.0068839956,0.020338703,0.83399975],"study_design_scores_gemma":[0.00010796695,0.00013156974,0.00091709616,0.00005730249,0.00005808233,0.00031802387,0.00022221664,0.94979775,0.021476608,0.018487135,0.008389148,0.00003707823],"about_ca_topic_score_codex":0.0018580217,"about_ca_topic_score_gemma":0.005236685,"teacher_disagreement_score":0.0023332813,"about_ca_system_score_codex":0.00051857874,"about_ca_system_score_gemma":0.001278439,"threshold_uncertainty_score":0.009726107},"labels":[],"label_agreement":null},{"id":"W4411157337","doi":"10.46586/tches.v2025.i3.493-515","title":"Accelerating EdDSA Signature Verification with Faster Scalar Size Halving","year":2025,"lang":"en","type":"article","venue":"IACR Transactions on Cryptographic Hardware and Embedded Systems","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Signature (topology); Scalar (mathematics); Computer science; Mathematics; Physics; Algorithm; Geometry","score_opus":0.009931151980072331,"score_gpt":0.24352553186925682,"score_spread":0.2335943798891845,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411157337","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09417346,0.0004210814,0.8790126,0.00037514133,0.00023910128,0.00019022608,0.00032416213,0.019154923,0.006109323],"genre_scores_gemma":[0.476537,0.00008450228,0.51578134,0.00019371636,0.00004978508,0.0001316064,0.00053165044,0.0006537916,0.006036656],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9968708,0.0006757008,0.00032723713,0.0006094015,0.0012468699,0.00027002612],"domain_scores_gemma":[0.99549085,0.0014013587,0.00041890863,0.0018032808,0.0007700178,0.000115534895],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00174062,0.00060409645,0.0006778635,0.000863229,0.00053391594,0.001420219,0.0013749301,0.0005752397,0.007877545],"category_scores_gemma":[0.0064459248,0.00034762308,0.00083683676,0.00057136087,0.0008480799,0.0025535903,0.0017928478,0.0011119319,0.0034230312],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0026432688,0.00044090557,0.0053372122,0.0003909714,0.00012260044,0.00037368422,0.00048234873,0.05193214,0.2089134,0.05947566,0.0134632345,0.6564246],"study_design_scores_gemma":[0.0004117318,0.00068959536,0.0016142128,0.00006549671,0.000053788644,0.00074230315,0.00015988169,0.53359383,0.40254462,0.027495947,0.032485858,0.00014272192],"about_ca_topic_score_codex":0.0008858032,"about_ca_topic_score_gemma":0.0010337988,"teacher_disagreement_score":0.007877545,"about_ca_system_score_codex":0.0011057525,"about_ca_system_score_gemma":0.0017546612,"threshold_uncertainty_score":0.026353061},"labels":[],"label_agreement":null},{"id":"W4411172844","doi":"10.1109/iccit64611.2024.11021891","title":"BDeedNet: A Deep Learning Framework for Bengali Deed Summarization","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Regina","funders":"","keywords":"Bengali; Deed; Automatic summarization; Computer science; Artificial intelligence; Natural language processing","score_opus":0.012285842455555427,"score_gpt":0.29428792222126626,"score_spread":0.28200207976571084,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411172844","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.052129388,0.0045868317,0.86785126,0.001748293,0.0006442649,0.00042812602,0.016149756,0.036747236,0.019714806],"genre_scores_gemma":[0.4733783,0.0025429204,0.42034295,0.00076150015,0.00027744425,0.0005239919,0.04848588,0.0013572385,0.052329753],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99974996,0.000057925838,0.000023068813,0.000085077125,0.00004832842,0.000035498146],"domain_scores_gemma":[0.99969375,0.000075532545,0.00003371317,0.000049105965,0.00012795423,0.000019941237],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000498357,0.0011614187,0.00046076137,0.0013757147,0.0005155317,0.0010737599,0.0015426456,0.00084373384,0.0048102206],"category_scores_gemma":[0.0017747519,0.00022354166,0.0005681631,0.0011035217,0.00027233397,0.0016227444,0.000738979,0.0012697369,0.0024049021],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00033673577,0.0001759061,0.0022853822,0.0004084385,0.00022513348,0.0003226839,0.0004912203,0.117103174,0.015602816,0.01419661,0.060297936,0.78855395],"study_design_scores_gemma":[0.000028142735,0.000114690854,0.0013196866,0.000060536368,0.00009476106,0.000108321714,0.0002292716,0.9238973,0.014094321,0.0124759255,0.047538597,0.000038499293],"about_ca_topic_score_codex":0.030413818,"about_ca_topic_score_gemma":0.05666448,"teacher_disagreement_score":0.030413818,"about_ca_system_score_codex":0.0018244097,"about_ca_system_score_gemma":0.0011565523,"threshold_uncertainty_score":0.06047356},"labels":[],"label_agreement":null},{"id":"W4411203676","doi":"10.1109/deeptest66595.2025.00012","title":"On the Effectiveness of LLMs for Manual Test Verifications","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Test (biology); Computer science; Geology","score_opus":0.00941145104263115,"score_gpt":0.30171288416617614,"score_spread":0.292301433123545,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411203676","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7770298,0.0013996281,0.18837142,0.0012392388,0.0002266615,0.0017531128,0.001045026,0.018079132,0.01085597],"genre_scores_gemma":[0.8230185,0.0002091378,0.17133516,0.00031005754,0.000046004614,0.00055086863,0.0014439002,0.001803798,0.0012825296],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9008168,0.060882386,0.006776771,0.007847327,0.02229526,0.0013814148],"domain_scores_gemma":[0.43905732,0.43145528,0.028069846,0.063234426,0.03587637,0.002306773],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.06774879,0.001987015,0.00085355964,0.0037594426,0.0011141287,0.0052004894,0.0035967864,0.0020945899,0.004618067],"category_scores_gemma":[0.3897926,0.0011419491,0.0019480212,0.001127494,0.0021862409,0.00638273,0.0055236174,0.0020800906,0.0023061025],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.006397269,0.0023899514,0.0915815,0.0037062992,0.00072813395,0.0011573522,0.027474895,0.03322448,0.063643225,0.0047021164,0.008639576,0.75635517],"study_design_scores_gemma":[0.0017475524,0.018030327,0.16736819,0.004725536,0.0017309922,0.004498529,0.015345221,0.5058537,0.21295609,0.014016087,0.052257594,0.0014701791],"about_ca_topic_score_codex":0.0038635067,"about_ca_topic_score_gemma":0.00384445,"teacher_disagreement_score":0.06774879,"about_ca_system_score_codex":0.0022396818,"about_ca_system_score_gemma":0.0029000787,"threshold_uncertainty_score":0.35829413},"labels":[],"label_agreement":null},{"id":"W4411206941","doi":"10.1109/icsses64899.2025.11009852","title":"Cross-Lingual Summarization for Overseas Applications Using Multilingual Pre-Trained Models and Knowledge Distillation","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Automatic summarization; Distillation; Computer science; Natural language processing; Artificial intelligence; Information retrieval; Chromatography; Chemistry","score_opus":0.027414681455291898,"score_gpt":0.38216958920731287,"score_spread":0.35475490775202095,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411206941","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04110764,0.0021855088,0.89860463,0.0008065135,0.00042030256,0.00026839503,0.0035076807,0.04823531,0.0048640473],"genre_scores_gemma":[0.38917512,0.0014364674,0.55933523,0.00059609406,0.00040982856,0.0004199861,0.03183506,0.003070874,0.013721327],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987627,0.00036050094,0.00011781909,0.00043688164,0.00021704394,0.00010494608],"domain_scores_gemma":[0.99803394,0.00065737596,0.00014993953,0.00047582292,0.0005928968,0.0000901142],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017458808,0.0026031507,0.0012646285,0.002448619,0.001016125,0.002495054,0.0018179892,0.0013281072,0.0046403],"category_scores_gemma":[0.0058116666,0.00054183294,0.0016385791,0.0023819266,0.0005550761,0.0039710077,0.0029207713,0.0020912096,0.004985741],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00042064965,0.00024706984,0.0018124062,0.0006447016,0.00030036154,0.00061506615,0.0009190497,0.0880722,0.026993832,0.0047810306,0.026740069,0.8484536],"study_design_scores_gemma":[0.0000719624,0.0003255843,0.0011209495,0.00007775871,0.0002496296,0.0002392708,0.0008683887,0.9154043,0.032590326,0.014164246,0.034780398,0.00010713585],"about_ca_topic_score_codex":0.010035575,"about_ca_topic_score_gemma":0.018535448,"teacher_disagreement_score":0.010035575,"about_ca_system_score_codex":0.00091457966,"about_ca_system_score_gemma":0.0020174973,"threshold_uncertainty_score":0.019954324},"labels":[],"label_agreement":null},{"id":"W4411272189","doi":"10.1109/icse-companion66252.2025.00075","title":"Revisiting SWE-Bench: On the Importance of Data Quality for LLM-Based Code Models","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Computer science; Code (set theory); Quality (philosophy); Programming language; Software engineering; Physics","score_opus":0.14261135126178118,"score_gpt":0.40459487372469816,"score_spread":0.261983522462917,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411272189","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.65435296,0.015042793,0.2231017,0.027405122,0.0024019347,0.0010071773,0.032259554,0.03182192,0.012606879],"genre_scores_gemma":[0.75102687,0.00238499,0.18201748,0.0047514155,0.00048768023,0.0006045588,0.051645577,0.0050026784,0.002078834],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9714271,0.012352563,0.0025341646,0.0044354578,0.0084851505,0.0007656102],"domain_scores_gemma":[0.80349225,0.13212232,0.0058402023,0.038911864,0.017617041,0.0020162875],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03167545,0.0017178254,0.0010932044,0.0032221992,0.0014402927,0.006022414,0.004378398,0.00199154,0.0018218663],"category_scores_gemma":[0.22209382,0.00084093184,0.0016082757,0.0030433408,0.0028399955,0.008654105,0.004636491,0.006013649,0.0012178128],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0024248583,0.0017079785,0.17650504,0.005622926,0.0012068858,0.0010536327,0.0052010263,0.1399554,0.021657886,0.03110524,0.18003525,0.43352392],"study_design_scores_gemma":[0.00047572245,0.0014056797,0.048818607,0.0023870976,0.00033818552,0.0014110281,0.003469158,0.76860213,0.024988625,0.052353058,0.09542724,0.0003235531],"about_ca_topic_score_codex":0.014042961,"about_ca_topic_score_gemma":0.019527586,"teacher_disagreement_score":0.03167545,"about_ca_system_score_codex":0.0021637958,"about_ca_system_score_gemma":0.004283668,"threshold_uncertainty_score":0.16751778},"labels":[],"label_agreement":null},{"id":"W4411328365","doi":"10.1080/15228886.2025.2518369","title":"Using AI Tools for Slavic Transliteration to Support Cataloging Workflows: Potential Use Cases and Limitations","year":2025,"lang":"en","type":"article","venue":"Slavic & East European Information Resources","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Transliteration; Cataloging; Workflow; Slavic languages; Computer science; World Wide Web; Information retrieval; Natural language processing; Linguistics; Database","score_opus":0.04921929063986303,"score_gpt":0.297691071386127,"score_spread":0.24847178074626397,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411328365","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03729643,0.0011761701,0.8621454,0.005570605,0.0003291489,0.0008309998,0.00090727565,0.06389135,0.027852504],"genre_scores_gemma":[0.14367491,0.00083076244,0.83692235,0.0010173945,0.000094657706,0.0005489991,0.0018061518,0.007739198,0.007365615],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9857028,0.007491432,0.0017276694,0.0016448474,0.0028919335,0.0005412629],"domain_scores_gemma":[0.8786921,0.07917803,0.0031915323,0.023360174,0.0144074345,0.0011708018],"candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.022994332,0.001253249,0.00093922456,0.003423392,0.0021266625,0.008228463,0.0040359227,0.0012087759,0.0081635965],"category_scores_gemma":[0.07429572,0.0010244402,0.0010594962,0.0042180866,0.0034096632,0.009689012,0.0039362875,0.0029424268,0.0063337004],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005949265,0.0003063123,0.008039662,0.003170024,0.00016405067,0.001240831,0.023457715,0.010396675,0.028382357,0.07376701,0.031356055,0.8191244],"study_design_scores_gemma":[0.0001931479,0.00035674282,0.003946338,0.0027781634,0.00017757715,0.0024660742,0.008937758,0.16894116,0.09225658,0.10531024,0.6141687,0.00046753327],"about_ca_topic_score_codex":0.010536984,"about_ca_topic_score_gemma":0.009072413,"teacher_disagreement_score":0.9917715,"about_ca_system_score_codex":0.003167984,"about_ca_system_score_gemma":0.0061116978,"threshold_uncertainty_score":0.121607065},"labels":[],"label_agreement":null},{"id":"W4411344985","doi":"10.33767/osf.io/t3b62_v1","title":"Measuring Lexical Distance between Parallel Corpora: The Case of AI-Generated News Translation","year":2025,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Institut de Valorisation des Données; Canada First Research Excellence Fund","keywords":"Translation (biology); Natural language processing; Computer science; Artificial intelligence; Parallel corpora; Machine translation; Linguistics; Information retrieval; Biology; Philosophy; Genetics","score_opus":0.07683607276715325,"score_gpt":0.31526203693399385,"score_spread":0.2384259641668406,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411344985","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5908662,0.0033156583,0.38001066,0.0016931487,0.00029135318,0.0005672488,0.0012541908,0.00094907405,0.021052497],"genre_scores_gemma":[0.6796799,0.00054298283,0.31489825,0.00010753197,0.00011314947,0.00036457652,0.0018087848,0.0003153936,0.0021695402],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9784557,0.011574523,0.0019005075,0.0027712465,0.0048318766,0.00046611528],"domain_scores_gemma":[0.9096325,0.06254276,0.005698944,0.010689129,0.0108166095,0.0006200741],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011960935,0.0007111155,0.0011831924,0.0063384534,0.0025407353,0.0045880973,0.0018633726,0.001885156,0.0017188061],"category_scores_gemma":[0.093958445,0.0006930425,0.00064300315,0.013205052,0.003215417,0.006694297,0.0032592723,0.0015360746,0.0009832291],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00200304,0.0007661678,0.058440346,0.0024509248,0.00056750333,0.0065415166,0.025923846,0.070398636,0.025873763,0.13701779,0.007189194,0.6628273],"study_design_scores_gemma":[0.00029830143,0.000846491,0.054287758,0.0005312015,0.00036485135,0.005876095,0.016435029,0.5737076,0.056738146,0.23364303,0.056906443,0.0003650971],"about_ca_topic_score_codex":0.0040970175,"about_ca_topic_score_gemma":0.004149984,"teacher_disagreement_score":0.011960935,"about_ca_system_score_codex":0.0019491406,"about_ca_system_score_gemma":0.0013351917,"threshold_uncertainty_score":0.063256264},"labels":[],"label_agreement":null},{"id":"W4411360040","doi":"10.1109/icpc66645.2025.00068","title":"Code Review Comprehension: Reviewing Strategies Seen Through Code Comprehension Theories","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"Schweizerischer Nationalfonds zur Förderung der Wissenschaftlichen Forschung","keywords":"Program comprehension; Comprehension; Computer science; Code (set theory); Programming language; Software; Software system","score_opus":0.027671463183987645,"score_gpt":0.33997917969916386,"score_spread":0.3123077165151762,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411360040","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.65592265,0.0032090638,0.30529258,0.0058153346,0.0002100514,0.0014937998,0.0002565545,0.0013117176,0.026488231],"genre_scores_gemma":[0.94091994,0.0011016115,0.05363887,0.0008124207,0.000080979175,0.00077701616,0.00023809892,0.00030609174,0.0021250283],"study_design_codex":"qualitative","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9266391,0.04753996,0.0049609197,0.004506544,0.014884527,0.0014689282],"domain_scores_gemma":[0.504236,0.38808203,0.035587084,0.014489748,0.05538454,0.0022207263],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.052641854,0.00074546726,0.0005887451,0.008753779,0.0026915735,0.0066980426,0.0021819824,0.0018592533,0.001519616],"category_scores_gemma":[0.34310704,0.0006717287,0.00066164444,0.0034013789,0.0055357157,0.008762643,0.003944927,0.0023056935,0.00058070105],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000284393,0.00023190393,0.056618847,0.0015191382,0.00010177354,0.0012305824,0.70448565,0.0015390438,0.014771463,0.020665608,0.004919068,0.19363256],"study_design_scores_gemma":[0.00027231796,0.0015270505,0.116080776,0.0062970323,0.00048729053,0.0076321946,0.4869698,0.066802606,0.03751004,0.12545389,0.15009096,0.0008761013],"about_ca_topic_score_codex":0.0025135954,"about_ca_topic_score_gemma":0.00268554,"teacher_disagreement_score":0.052641854,"about_ca_system_score_codex":0.003163769,"about_ca_system_score_gemma":0.0058449768,"threshold_uncertainty_score":0.27840006},"labels":[],"label_agreement":null},{"id":"W4411413826","doi":"10.33767/osf.io/t3b62_v2","title":"Measuring Lexical Distance between Parallel Corpora: The Case of AI-Generated News Translation","year":2025,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Institut de Valorisation des Données; Canada First Research Excellence Fund","keywords":"Computer science; Jaccard index; Source text; Natural language processing; Compiler; Artificial intelligence; Translation (biology); Strengths and weaknesses; Heuristic; Machine translation; Value (mathematics); Information retrieval; Linguistics; Programming language; Cluster analysis","score_opus":0.07683607276715325,"score_gpt":0.31526203693399385,"score_spread":0.2384259641668406,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411413826","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5908662,0.0033156583,0.38001066,0.0016931487,0.00029135318,0.0005672488,0.0012541908,0.00094907405,0.021052497],"genre_scores_gemma":[0.6796799,0.00054298283,0.31489825,0.00010753197,0.00011314947,0.00036457652,0.0018087848,0.0003153936,0.0021695402],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9784557,0.011574523,0.0019005075,0.0027712465,0.0048318766,0.00046611528],"domain_scores_gemma":[0.9096325,0.06254276,0.005698944,0.010689129,0.0108166095,0.0006200741],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011960935,0.0007111155,0.0011831924,0.0063384534,0.0025407353,0.0045880973,0.0018633726,0.001885156,0.0017188061],"category_scores_gemma":[0.093958445,0.0006930425,0.00064300315,0.013205052,0.003215417,0.006694297,0.0032592723,0.0015360746,0.0009832291],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00200304,0.0007661678,0.058440346,0.0024509248,0.00056750333,0.0065415166,0.025923846,0.070398636,0.025873763,0.13701779,0.007189194,0.6628273],"study_design_scores_gemma":[0.00029830143,0.000846491,0.054287758,0.0005312015,0.00036485135,0.005876095,0.016435029,0.5737076,0.056738146,0.23364303,0.056906443,0.0003650971],"about_ca_topic_score_codex":0.0040970175,"about_ca_topic_score_gemma":0.004149984,"teacher_disagreement_score":0.011960935,"about_ca_system_score_codex":0.0019491406,"about_ca_system_score_gemma":0.0013351917,"threshold_uncertainty_score":0.063256264},"labels":[],"label_agreement":null},{"id":"W4411549495","doi":"10.1145/3701716.3715311","title":"SynDL: A Large-Scale Synthetic Test Collection for Passage Retrieval","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Microsoft (Canada)","funders":"Engineering and Physical Sciences Research Council; Universitas Brawijaya","keywords":"Computer science; Scale (ratio); Test (biology); Artificial intelligence; Information retrieval; Geology; Cartography; Geography","score_opus":0.007113296214746126,"score_gpt":0.2683040401957007,"score_spread":0.2611907439809546,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411549495","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5600635,0.0024853942,0.10922214,0.0022794236,0.0021538422,0.0060818084,0.2524364,0.036391787,0.02888575],"genre_scores_gemma":[0.32063752,0.0004796821,0.09173924,0.00096466934,0.0002986546,0.0036651525,0.5684138,0.0018538809,0.011947357],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9956512,0.002063768,0.00046787455,0.00056617765,0.001031122,0.00021992328],"domain_scores_gemma":[0.98325783,0.0063896673,0.00073528383,0.0039547947,0.004348043,0.0013143665],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0040795403,0.0013743732,0.0010085189,0.0032961564,0.0015678619,0.0016390746,0.0027044679,0.0017304256,0.007365456],"category_scores_gemma":[0.018070169,0.0004916063,0.0010853915,0.003012952,0.0012882204,0.00211243,0.0022079695,0.0019362064,0.0077884067],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002891805,0.0053117317,0.03815646,0.004229195,0.00070312055,0.0030383873,0.0027972874,0.033714224,0.049648374,0.0055426555,0.61514366,0.23882303],"study_design_scores_gemma":[0.002369855,0.006677653,0.085184194,0.0003980067,0.00047780917,0.0047135972,0.004542576,0.22364232,0.09559187,0.008879029,0.56671005,0.00081300095],"about_ca_topic_score_codex":0.011290905,"about_ca_topic_score_gemma":0.020383516,"teacher_disagreement_score":0.011290905,"about_ca_system_score_codex":0.0014535777,"about_ca_system_score_gemma":0.001878017,"threshold_uncertainty_score":0.024639964},"labels":[],"label_agreement":null},{"id":"W4411552547","doi":"10.1109/icse55347.2025.00236","title":"INTERTRANS: Leveraging Transitive Intermediate Translations to Enhance LLM-Based Code Translation","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada); University of Waterloo; Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Transitive relation; Translation (biology); Programming language; Code (set theory); Natural language processing; Mathematics","score_opus":0.014500625365265996,"score_gpt":0.31733938168597736,"score_spread":0.3028387563207114,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411552547","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04738467,0.00068586454,0.8739771,0.00044369037,0.00024804336,0.0002801392,0.0009893634,0.06915658,0.0068345973],"genre_scores_gemma":[0.22192885,0.00029460413,0.75958306,0.00043666814,0.000071738585,0.00028452498,0.005311606,0.007552785,0.004536153],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99725914,0.0007610395,0.0002518782,0.00065826124,0.00083993044,0.00022972429],"domain_scores_gemma":[0.9955556,0.0014799411,0.00038200608,0.0015321627,0.000941401,0.000108957516],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015666425,0.0021787207,0.0009103679,0.001983356,0.0007803348,0.0018700581,0.0019302756,0.0012633837,0.003987427],"category_scores_gemma":[0.010175064,0.0006799631,0.0013127283,0.0013760384,0.0012936705,0.0031121387,0.0030068422,0.0021259966,0.0041095275],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00063146703,0.0004133471,0.005815325,0.0011924307,0.00018729278,0.0012941814,0.001657128,0.07123306,0.079311155,0.02041048,0.03328515,0.784569],"study_design_scores_gemma":[0.00014574923,0.00042790512,0.0012565437,0.0001369967,0.00011765675,0.0008119048,0.00053157576,0.8128113,0.11072523,0.03229517,0.04063881,0.00010120499],"about_ca_topic_score_codex":0.0034850435,"about_ca_topic_score_gemma":0.0062442557,"teacher_disagreement_score":0.003987427,"about_ca_system_score_codex":0.0007741049,"about_ca_system_score_gemma":0.0025068687,"threshold_uncertainty_score":0.013339281},"labels":[],"label_agreement":null},{"id":"W4411556646","doi":"10.7202/1118381ar","title":"Error annotation and analysis of a (semi-)specialised English-French learner translation corpus","year":2024,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Annotation; Natural language processing; Translation (biology); Computer science; Artificial intelligence; Error analysis; Linguistics; Mathematics; Biology; Philosophy","score_opus":0.03405760421058877,"score_gpt":0.2937376729954785,"score_spread":0.2596800687848897,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411556646","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.92133594,0.000740402,0.060000297,0.0004650587,0.00019875949,0.00071446015,0.007692768,0.0025877145,0.006264563],"genre_scores_gemma":[0.8566837,0.0003742432,0.09990873,0.00019191482,0.000092089285,0.0016728614,0.033535063,0.0010811203,0.0064603426],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9901239,0.00429401,0.0012584074,0.0015628987,0.0024957017,0.0002650588],"domain_scores_gemma":[0.9330976,0.04114522,0.003836589,0.004734581,0.016568707,0.00061733735],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0054853633,0.00090234436,0.0008553426,0.004795984,0.0015674453,0.001759679,0.00095804676,0.001285669,0.0043761577],"category_scores_gemma":[0.034450915,0.00034545685,0.00045188147,0.0037712262,0.001438067,0.0012697835,0.00225855,0.001120586,0.0022051139],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002724384,0.0017189981,0.07274947,0.0070843417,0.0003549568,0.01123089,0.06461579,0.013067121,0.15092002,0.0047172448,0.025150461,0.64566636],"study_design_scores_gemma":[0.000576427,0.0019832044,0.24573661,0.0014563916,0.00063727447,0.015538881,0.038137406,0.115309514,0.30542192,0.0044444297,0.2701493,0.00060870574],"about_ca_topic_score_codex":0.004580108,"about_ca_topic_score_gemma":0.0051979,"teacher_disagreement_score":0.0054853633,"about_ca_system_score_codex":0.0009943092,"about_ca_system_score_gemma":0.0019337669,"threshold_uncertainty_score":0.02900976},"labels":[],"label_agreement":null},{"id":"W4411558583","doi":"10.1017/nlp.2024.6","title":"Editors’ foreword","year":2025,"lang":"en","type":"article","venue":"Natural language processing.","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Psychology; Computer science; Cognitive science","score_opus":0.004900954798458003,"score_gpt":0.2795578525872755,"score_spread":0.2746568977888175,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411558583","genre_codex":"editorial","genre_gemma":"editorial","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"editorial","genre_consensus":"editorial","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.000118496886,0.005257195,0.0004327724,0.046707407,0.9267867,0.00005229886,0.00048488277,0.00020460313,0.019955518],"genre_scores_gemma":[0.0021541708,0.009427588,0.00064561766,0.049242195,0.77226096,0.00013148141,0.0008199013,0.00042809398,0.16489],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99771786,0.00028266927,0.00026210462,0.00045704184,0.0010613465,0.00021894669],"domain_scores_gemma":[0.9877302,0.0026601767,0.0007120003,0.00048654372,0.0064814864,0.001929603],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028243286,0.0015565968,0.001462624,0.0024137353,0.0014563841,0.006337774,0.0021898106,0.0042922595,0.15813571],"category_scores_gemma":[0.01572473,0.00047741132,0.001162962,0.0014528637,0.0007724263,0.003780216,0.0018210055,0.006871327,0.12448617],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000012191943,0.0000057525285,0.000019307083,0.000074203,0.000002310415,0.000018226781,0.0000047683498,0.000017231197,0.000038818147,0.00026242048,0.9894014,0.010143355],"study_design_scores_gemma":[0.000010122971,0.000016221153,0.0001405164,0.00016311693,0.0000049027503,0.000069667214,0.000014683003,0.000050912735,0.00010332523,0.00046641155,0.998953,0.000007099789],"about_ca_topic_score_codex":0.000908301,"about_ca_topic_score_gemma":0.0016978274,"teacher_disagreement_score":0.15813571,"about_ca_system_score_codex":0.0014495213,"about_ca_system_score_gemma":0.0019168485,"threshold_uncertainty_score":0.5290167},"labels":[],"label_agreement":null},{"id":"W4411638674","doi":"10.18653/v1/2024.eacl-short.6","title":"Multilingual Gradient Word-Order Typology from Universal Dependencies","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"McGill University","keywords":"Word order; Typology; Computer science; Natural language processing; Word (group theory); Artificial intelligence; Linguistics; Order (exchange); History; Philosophy","score_opus":0.01316816496745179,"score_gpt":0.27465383259308696,"score_spread":0.2614856676256352,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411638674","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.43495512,0.0012766963,0.08960342,0.0007410278,0.0005398747,0.00060989935,0.43243504,0.005690276,0.034148633],"genre_scores_gemma":[0.28906402,0.00036786252,0.050483212,0.00016258424,0.000095086034,0.0010186996,0.6529288,0.0007151906,0.0051646163],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99866223,0.00028856538,0.00016334823,0.00047613567,0.00025533172,0.0001543893],"domain_scores_gemma":[0.99661344,0.0011497594,0.00031552193,0.0010431545,0.0006455723,0.00023247986],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009923663,0.0005951849,0.00054815755,0.0048691244,0.0011824669,0.001709574,0.00097230915,0.0007845983,0.0097755315],"category_scores_gemma":[0.0060436768,0.00032095518,0.0006015913,0.0050002416,0.0007655356,0.002291848,0.002692239,0.0013237699,0.0051553445],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017129821,0.0008230598,0.10244773,0.00313723,0.00027569474,0.0030865185,0.0054615675,0.019927444,0.019266658,0.06864734,0.37094724,0.40426654],"study_design_scores_gemma":[0.0003235184,0.00025816343,0.104989305,0.00059766957,0.00010681566,0.002649181,0.00632051,0.06701364,0.013306027,0.121841945,0.68238735,0.00020582328],"about_ca_topic_score_codex":0.006382775,"about_ca_topic_score_gemma":0.013909637,"teacher_disagreement_score":0.0097755315,"about_ca_system_score_codex":0.0007514474,"about_ca_system_score_gemma":0.001453467,"threshold_uncertainty_score":0.032702386},"labels":[],"label_agreement":null},{"id":"W4411638707","doi":"10.18653/v1/2024.eacl-tutorials","title":"Proceedings of the 18th Conference of the European Chapter of the Association for Computational Linguistics: Tutorial Abstracts","year":2024,"lang":"en","type":"paratext","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Biological and Environmental Research; Atomic Energy of Canada Limited; Qatar National Research Fund; University of Notre Dame; Office of Science; Qatar Foundation; Fonds National de la Recherche Luxembourg; U.S. Department of Energy; York University; Institute for Catastrophic Loss Reduction; National Science Foundation","keywords":"Computer science; Computational linguistics; Association (psychology); Linguistics; Library science; Natural language processing; Philosophy; Epistemology","score_opus":0.01946760611778048,"score_gpt":0.2695734943444074,"score_spread":0.25010588822662694,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411638707","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015742421,0.09249878,0.14847909,0.05354377,0.24692154,0.0010574068,0.013913617,0.008782408,0.41906095],"genre_scores_gemma":[0.013321786,0.025376877,0.023142725,0.002612708,0.019538725,0.0006751088,0.013399754,0.0035922753,0.89834005],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9983292,0.0004946293,0.00012062721,0.00032476019,0.0005519434,0.00017870373],"domain_scores_gemma":[0.99342793,0.0017385529,0.00025192718,0.00051226915,0.0024500804,0.0016192328],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0043350263,0.0016676759,0.0029857233,0.003376025,0.0015514178,0.008061576,0.0016399121,0.0018995525,0.2043964],"category_scores_gemma":[0.0058855475,0.00055612467,0.0010259765,0.0030217364,0.0014612804,0.004780741,0.0034348508,0.0031266047,0.13699733],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012862524,0.00011922967,0.00024477806,0.00031009284,0.000029535991,0.00007057047,0.00013125085,0.00015162799,0.0008962999,0.0028118433,0.90097624,0.0941298],"study_design_scores_gemma":[0.000036238715,0.000058964688,0.00091421674,0.00024902172,0.000048218284,0.00013552525,0.00014329246,0.0012180555,0.0005043221,0.0038191816,0.99285346,0.000019444398],"about_ca_topic_score_codex":0.0036041185,"about_ca_topic_score_gemma":0.0054474687,"teacher_disagreement_score":0.2043964,"about_ca_system_score_codex":0.0016134306,"about_ca_system_score_gemma":0.0038954348,"threshold_uncertainty_score":0.68377405},"labels":[],"label_agreement":null},{"id":"W4411811241","doi":"10.1007/978-3-031-97141-9_10","title":"Context is the Key for LLM-Based Text Segmentation","year":2025,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ontario Tech University","funders":"","keywords":"Computer science; Key (lock); Context (archaeology); Segmentation; Artificial intelligence; Information retrieval; Data science; Natural language processing; Computer security; History","score_opus":0.015158586573179362,"score_gpt":0.27685971944323856,"score_spread":0.2617011328700592,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411811241","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017022196,0.007859404,0.9147504,0.0022243508,0.0015376421,0.00024234595,0.0019628443,0.029483424,0.02491747],"genre_scores_gemma":[0.2665697,0.0033580086,0.69608665,0.0014228666,0.0015145171,0.00024769932,0.004178575,0.004587254,0.022034785],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992083,0.00013855139,0.00008001963,0.00027617847,0.00021693295,0.00008005612],"domain_scores_gemma":[0.998287,0.00068844616,0.00014538706,0.00034776374,0.00039125865,0.00013998924],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00060181186,0.0010247314,0.0009541666,0.0019864966,0.0011302389,0.0034310713,0.0012754392,0.001355982,0.016651215],"category_scores_gemma":[0.0037824898,0.00060693064,0.0006649997,0.0018317638,0.00086044427,0.0050760405,0.0024659717,0.0019549727,0.022265544],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003073411,0.000048172893,0.00053530035,0.0005698242,0.000022243301,0.00021841314,0.00024603144,0.0013894654,0.11064109,0.010348995,0.025257744,0.8504154],"study_design_scores_gemma":[0.00008503216,0.00045418544,0.004367973,0.00068743783,0.000211381,0.001963812,0.0010722013,0.16853032,0.3410188,0.15578344,0.3256077,0.00021777087],"about_ca_topic_score_codex":0.0010095639,"about_ca_topic_score_gemma":0.0022400399,"teacher_disagreement_score":0.016651215,"about_ca_system_score_codex":0.0005001621,"about_ca_system_score_gemma":0.00084481825,"threshold_uncertainty_score":0.05570388},"labels":[],"label_agreement":null},{"id":"W4411846913","doi":"10.5430/elr.v14n2p1","title":"Recognizing Lexical Units in Portuguese","year":2025,"lang":"en","type":"article","venue":"English Linguistics Research","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Portuguese; Linguistics; Natural language processing; Computer science; Philosophy","score_opus":0.07688290686691437,"score_gpt":0.40432809617142224,"score_spread":0.32744518930450783,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411846913","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.64651835,0.004488424,0.13527857,0.0018601097,0.00027942137,0.00050125876,0.0022166546,0.0016177283,0.20723945],"genre_scores_gemma":[0.9388723,0.0009546094,0.048419163,0.0000797306,0.000032272128,0.000091147296,0.00067605433,0.0002210525,0.010653807],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99938655,0.00020330088,0.00007874337,0.00015691202,0.00012585097,0.000048684233],"domain_scores_gemma":[0.9978023,0.0014326939,0.00026598873,0.00019105354,0.00023239652,0.00007552391],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005652719,0.00032072086,0.00021976241,0.0019029712,0.001220393,0.0031873279,0.00038353776,0.0005480022,0.004459861],"category_scores_gemma":[0.0060709356,0.0003025618,0.00025850133,0.0015928654,0.0014780648,0.0024110533,0.000946327,0.0004930205,0.001343339],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004965322,0.00009146775,0.019183721,0.0020329785,0.00003075607,0.004560496,0.11522358,0.0026767047,0.10888911,0.09709774,0.005560136,0.6441567],"study_design_scores_gemma":[0.00008055952,0.00041200264,0.12528825,0.0019497712,0.00015381735,0.008800465,0.10185895,0.0347609,0.066637576,0.12106769,0.5387236,0.0002664006],"about_ca_topic_score_codex":0.006186789,"about_ca_topic_score_gemma":0.009108342,"teacher_disagreement_score":0.006186789,"about_ca_system_score_codex":0.0010418613,"about_ca_system_score_gemma":0.0009232312,"threshold_uncertainty_score":0.014919698},"labels":[],"label_agreement":null},{"id":"W4411868550","doi":"10.1016/j.jss.2026.112941","title":"Automatic Translation of Natural Language Requirements into Ctl Specifications Using Large Language Models: A Multi-Approach Evaluation ⋆","year":2025,"lang":"en","type":"preprint","venue":"Journal of Systems and Software","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Mitacs","keywords":"Computer science; Translation (biology); Natural language processing; CTL*; Natural language; Programming language; Universal Networking Language; Artificial intelligence; Machine translation; Programming language specification; Software engineering; Comprehension approach","score_opus":0.10288957519603693,"score_gpt":0.36164663536425906,"score_spread":0.25875706016822214,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411868550","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.24480553,0.00036005751,0.73179346,0.0006313364,0.00011060197,0.00069046975,0.0010566615,0.0114270495,0.009124755],"genre_scores_gemma":[0.52816147,0.00016689798,0.463558,0.00019540304,0.000025089925,0.00021359632,0.0025635653,0.0013877276,0.003728193],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9953446,0.0021902192,0.00027605114,0.000616469,0.0012396036,0.0003330235],"domain_scores_gemma":[0.9844326,0.009291434,0.0006677785,0.002158806,0.0031443022,0.0003052337],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033190474,0.0008958926,0.0007872217,0.0010725213,0.00064517313,0.0022101959,0.0016888011,0.0014379217,0.008814245],"category_scores_gemma":[0.013429196,0.00058928464,0.0013978586,0.0010330294,0.0008344901,0.002709272,0.0021457858,0.0014270038,0.002361237],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0039536776,0.0025706347,0.007699775,0.0016587304,0.00035249232,0.0013067988,0.001840989,0.18252885,0.10055932,0.041385494,0.014595005,0.6415483],"study_design_scores_gemma":[0.0003852809,0.00051906804,0.0012044095,0.000050410392,0.00012666937,0.00028128058,0.0006540961,0.9260546,0.053062063,0.013077607,0.0045393854,0.000045058274],"about_ca_topic_score_codex":0.005027529,"about_ca_topic_score_gemma":0.005216999,"teacher_disagreement_score":0.008814245,"about_ca_system_score_codex":0.0011413314,"about_ca_system_score_gemma":0.0029155295,"threshold_uncertainty_score":0.029486597},"labels":[],"label_agreement":null},{"id":"W4411950637","doi":"10.1109/forge66646.2025.00023","title":"MaRV: A Manually Validated Refactoring Dataset","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"EMI","keywords":"Code refactoring; Computer science; Programming language; Software","score_opus":0.012788646633431116,"score_gpt":0.30917543883313675,"score_spread":0.29638679219970565,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411950637","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.116929404,0.004157558,0.036330722,0.00082762586,0.00040175658,0.0009215232,0.7915484,0.036606275,0.012276757],"genre_scores_gemma":[0.03723201,0.0003602282,0.03250774,0.00018600201,0.000030839627,0.0006835969,0.92601484,0.0012358696,0.0017489552],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9925171,0.0015912378,0.0011639236,0.0016759973,0.002719079,0.00033268068],"domain_scores_gemma":[0.976712,0.008049036,0.0022037353,0.006273925,0.0062044165,0.0005570071],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0045798626,0.0019955938,0.0007697263,0.008038322,0.0011272341,0.0016319505,0.0036632768,0.0028696666,0.0030449228],"category_scores_gemma":[0.025000906,0.00060945336,0.0017241882,0.005741796,0.0008176145,0.0017230518,0.0018406993,0.00230153,0.0045688543],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011138103,0.001172811,0.044156846,0.007581396,0.00058374094,0.0015906967,0.0012194503,0.024601681,0.022912724,0.005515052,0.6486185,0.24093325],"study_design_scores_gemma":[0.00072812266,0.0008538166,0.07752848,0.001430877,0.00038786812,0.0023630152,0.0008186717,0.0689455,0.040082667,0.007951399,0.79848963,0.00041995116],"about_ca_topic_score_codex":0.00903334,"about_ca_topic_score_gemma":0.017461004,"teacher_disagreement_score":0.00903334,"about_ca_system_score_codex":0.0016800915,"about_ca_system_score_gemma":0.0026461177,"threshold_uncertainty_score":0.024220884},"labels":[],"label_agreement":null},{"id":"W4412039287","doi":"10.1111/lnc3.70016","title":"Conjoined Comparison and Variation in Degree Semantics","year":2025,"lang":"en","type":"article","venue":"Language and Linguistics Compass","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Variation (astronomy); Semantics (computer science); Degree (music); Linguistics; Computer science; Philosophy; Programming language; Physics; Astrophysics","score_opus":0.014711139817231163,"score_gpt":0.2968374589192176,"score_spread":0.28212631910198643,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412039287","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.63422555,0.00086464465,0.29058316,0.0009823446,0.00012641221,0.00008517596,0.00033236234,0.0004624787,0.07233785],"genre_scores_gemma":[0.98868316,0.00007064269,0.010127845,0.000038947284,0.00001738007,0.000018394483,0.00007094076,0.000045029647,0.00092770794],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99567777,0.002116611,0.00034636722,0.000752721,0.0008051371,0.00030147398],"domain_scores_gemma":[0.9948152,0.0028714763,0.0005901005,0.0008838298,0.0006337003,0.00020563055],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030219778,0.0003161842,0.0003762263,0.0022927425,0.0019410712,0.0032140834,0.0011198267,0.0007676182,0.0032322095],"category_scores_gemma":[0.0074277855,0.0003500661,0.00065591856,0.0024928262,0.007484675,0.010304341,0.0042418973,0.0013343894,0.00023262494],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000064528256,0.000013398297,0.001885145,0.0000703705,0.000018175724,0.00015298862,0.0045269914,0.00081858854,0.0022189855,0.9789836,0.00026106523,0.010986116],"study_design_scores_gemma":[0.000018186944,0.00005186139,0.0033202302,0.00005288769,0.000034291916,0.0005962478,0.0033409556,0.006638663,0.0028042311,0.96980625,0.013293746,0.00004240031],"about_ca_topic_score_codex":0.0014318193,"about_ca_topic_score_gemma":0.0011314334,"teacher_disagreement_score":0.0032322095,"about_ca_system_score_codex":0.001767586,"about_ca_system_score_gemma":0.0005709855,"threshold_uncertainty_score":0.015981913},"labels":[],"label_agreement":null},{"id":"W4412266620","doi":"","title":"Analogy Training Multilingual Encoderss","year":2021,"lang":"en","type":"article","venue":"Research at the University of Copenhagen (University of Copenhagen)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"H. Lundbeck A/S; Natural Sciences and Engineering Research Council of Canada; Lundbeckfonden","keywords":"Analogy; Training (meteorology); Computer science; Natural language processing; Artificial intelligence; Linguistics; Physics; Philosophy; Meteorology","score_opus":0.06711299574584237,"score_gpt":0.3121901835722294,"score_spread":0.24507718782638702,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412266620","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10172067,0.0008120926,0.863256,0.00051683956,0.00045123743,0.00021348108,0.0016738122,0.01649894,0.0148569625],"genre_scores_gemma":[0.56532437,0.00036877836,0.4062031,0.0005171961,0.00014697501,0.0003873583,0.0068580112,0.00088031817,0.019313868],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99908376,0.00021217894,0.000051131374,0.0003892868,0.00016139491,0.00010227117],"domain_scores_gemma":[0.9986645,0.00036745833,0.000052849417,0.00038855366,0.00045307938,0.000073486335],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00097504194,0.0010489663,0.0006858501,0.00067905424,0.00063767727,0.0008751558,0.001560375,0.0007514886,0.008589257],"category_scores_gemma":[0.004287593,0.0004743277,0.0006930583,0.0006176829,0.0005363219,0.002433113,0.002174696,0.0020596285,0.0035305559],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00046203172,0.00042765966,0.0055016726,0.00038549418,0.00021562733,0.0004207513,0.00044741272,0.07923984,0.0373561,0.037031442,0.030268574,0.80824333],"study_design_scores_gemma":[0.0001060907,0.00026128007,0.0013296538,0.00004288206,0.00009562556,0.0003175231,0.00020887141,0.8988334,0.039798092,0.034567483,0.02439081,0.000048350903],"about_ca_topic_score_codex":0.0041961283,"about_ca_topic_score_gemma":0.013450534,"teacher_disagreement_score":0.008589257,"about_ca_system_score_codex":0.00070092117,"about_ca_system_score_gemma":0.001719382,"threshold_uncertainty_score":0.028733969},"labels":[],"label_agreement":null},{"id":"W4412315723","doi":"","title":"\"… en plaffert af en pennepose\":lidt om opdatering af ordbogsdefinitioner","year":2012,"lang":"da","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"SKiN Health","funders":"","keywords":"Computer science","score_opus":0.01817051742315334,"score_gpt":0.27334144862991777,"score_spread":0.25517093120676443,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412315723","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012173903,0.004757345,0.4627537,0.032547392,0.013754306,0.00033059195,0.004770557,0.009818839,0.45909333],"genre_scores_gemma":[0.20274371,0.0049311323,0.25449505,0.0087127555,0.003495736,0.00051154994,0.00763484,0.010707226,0.5067679],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9937423,0.002443221,0.0004729566,0.0011730654,0.0017253143,0.0004431878],"domain_scores_gemma":[0.99697876,0.0008129907,0.00013494214,0.00062033435,0.0012902152,0.00016276135],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00488108,0.00091480673,0.0007209133,0.00198104,0.0042171534,0.009974549,0.0021215645,0.003053334,0.048303597],"category_scores_gemma":[0.008135336,0.0010744787,0.00086179154,0.0018141933,0.0052278643,0.016968232,0.0060241325,0.006918756,0.020599367],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017338064,0.00004470991,0.00042252836,0.0002598729,0.000013603203,0.00022284183,0.005450082,0.00028069777,0.0016416047,0.7199329,0.16262935,0.10892831],"study_design_scores_gemma":[0.000011815315,0.000016761187,0.00016758767,0.00014584031,0.000012982216,0.00015496474,0.001316576,0.0006727183,0.001856305,0.06076129,0.93484455,0.000038578],"about_ca_topic_score_codex":0.011151614,"about_ca_topic_score_gemma":0.0113641145,"teacher_disagreement_score":0.048303597,"about_ca_system_score_codex":0.0029293448,"about_ca_system_score_gemma":0.002965423,"threshold_uncertainty_score":0.16159165},"labels":[],"label_agreement":null},{"id":"W4412329597","doi":"","title":"Året, der gik - beretning fra center og ordbog","year":2021,"lang":"da","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"SKiN Health","funders":"","keywords":"Center (category theory); Geology; Environmental science; Geography; Chemistry","score_opus":0.014345325435465586,"score_gpt":0.2737015319927924,"score_spread":0.25935620655732683,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412329597","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.038728192,0.04393988,0.031220661,0.048049122,0.02473381,0.0004244415,0.030633308,0.014776586,0.76749396],"genre_scores_gemma":[0.0380749,0.0061037703,0.018091405,0.0016865623,0.00074902724,0.0001494011,0.009936066,0.004425661,0.92078316],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99645567,0.0005403386,0.00015335369,0.0011711922,0.0010820853,0.0005974908],"domain_scores_gemma":[0.9977672,0.00051492575,0.00013431942,0.00031152385,0.0007279728,0.000544152],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0022478837,0.002287751,0.002050733,0.0020934618,0.0033345704,0.008697215,0.0019362706,0.0035710705,0.31030908],"category_scores_gemma":[0.004044701,0.00087699137,0.0009986621,0.0018598381,0.0012093979,0.003920495,0.004000679,0.0044703805,0.2645631],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001340088,0.00047971384,0.003130583,0.0010754993,0.00008422966,0.00097281445,0.0016628939,0.001622977,0.0073529994,0.047389716,0.4921158,0.44277266],"study_design_scores_gemma":[0.00005276591,0.00008121838,0.0014867194,0.00018681557,0.000035032575,0.00035243927,0.0005707739,0.00068353553,0.0029211545,0.0060831653,0.9875125,0.000033841334],"about_ca_topic_score_codex":0.008047389,"about_ca_topic_score_gemma":0.007336035,"teacher_disagreement_score":0.31030908,"about_ca_system_score_codex":0.0026839247,"about_ca_system_score_gemma":0.004571093,"threshold_uncertainty_score":0.9837604},"labels":[],"label_agreement":null},{"id":"W4412377930","doi":"10.1145/3726302.3730331","title":"RankLLM: A Python Package for Reranking with LLMs","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Universitas Brawijaya","keywords":"Python (programming language); Computer science; Programming language","score_opus":0.007520732029637543,"score_gpt":0.27160608793856783,"score_spread":0.2640853559089303,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412377930","genre_codex":"software","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":"software","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0012378335,0.00018917587,0.39909452,0.00026463912,0.00014459662,0.00021611252,0.01586644,0.5782242,0.0047625145],"genre_scores_gemma":[0.037331097,0.0005411778,0.6559557,0.0011722116,0.00019092608,0.001983984,0.05889699,0.22554938,0.018378517],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99830437,0.00037836918,0.00019528798,0.0003354621,0.0006006231,0.00018582068],"domain_scores_gemma":[0.9971169,0.0013120245,0.0002364202,0.0006034858,0.0005651573,0.00016589483],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002372931,0.002462836,0.0014880473,0.0017697874,0.00092705456,0.0025161805,0.0047233384,0.0013702764,0.086483374],"category_scores_gemma":[0.012828072,0.0017324506,0.002415605,0.001667537,0.00094214274,0.0036365397,0.003928102,0.004004947,0.067643434],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006237182,0.00018859185,0.0026936354,0.0024266196,0.00030590405,0.00042646457,0.00045763637,0.030813474,0.0080279205,0.026289724,0.672726,0.25502032],"study_design_scores_gemma":[0.00048314885,0.00013727673,0.0019218794,0.00028308266,0.000103102524,0.00044918148,0.00011681431,0.45004168,0.025245564,0.098359786,0.42259586,0.00026247904],"about_ca_topic_score_codex":0.00547875,"about_ca_topic_score_gemma":0.009808464,"teacher_disagreement_score":0.086483374,"about_ca_system_score_codex":0.0012449417,"about_ca_system_score_gemma":0.0038231413,"threshold_uncertainty_score":0.2893157},"labels":[],"label_agreement":null},{"id":"W4412448221","doi":"10.1007/s44196-026-01341-9","title":"Infusing Syntax and Semantics into LLMs","year":2025,"lang":"en","type":"preprint","venue":"International Journal of Computational Intelligence Systems","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"Conselho Nacional de Desenvolvimento Científico e Tecnológico; Ministério da Ciência, Tecnologia e Inovação; Coordenação de Aperfeiçoamento de Pessoal de Nível Superior; Fundação de Amparo à Pesquisa do Estado de São Paulo; International Business Machines Corporation","keywords":"Syntax; Semantics (computer science); Linguistics; Computer science; Programming language; Natural language processing; Philosophy","score_opus":0.019795394015228947,"score_gpt":0.33857610026688,"score_spread":0.31878070625165106,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412448221","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07175998,0.00075310457,0.8775142,0.00991277,0.00073141756,0.0001552395,0.0010226397,0.024243265,0.013907474],"genre_scores_gemma":[0.7126942,0.000574749,0.26736194,0.003194995,0.0006524101,0.00017634455,0.0014689597,0.0072581917,0.006618217],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99232024,0.0046084435,0.00046866707,0.00076741603,0.0015264394,0.0003087628],"domain_scores_gemma":[0.9683464,0.021279603,0.0011941588,0.0063470765,0.0024546352,0.0003780219],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0067026247,0.0012583421,0.0010535497,0.0010296989,0.00095981365,0.00473407,0.0014315151,0.0016317251,0.009838111],"category_scores_gemma":[0.032859363,0.0010037745,0.0016279162,0.0010683164,0.0034518633,0.0110729225,0.0044472795,0.004481757,0.0032915133],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001084136,0.00038411148,0.0069585023,0.0009500239,0.00021438397,0.00079864723,0.001798183,0.13524997,0.0336708,0.6136982,0.031238448,0.17395456],"study_design_scores_gemma":[0.000060409748,0.00011342567,0.0004013462,0.00012640377,0.00012670986,0.00018044446,0.00025255862,0.5266993,0.025495678,0.42089233,0.025583075,0.0000682132],"about_ca_topic_score_codex":0.0022410292,"about_ca_topic_score_gemma":0.002317845,"teacher_disagreement_score":0.009838111,"about_ca_system_score_codex":0.0018673268,"about_ca_system_score_gemma":0.0019254267,"threshold_uncertainty_score":0.0354473},"labels":[],"label_agreement":null},{"id":"W4412473765","doi":"10.5281/zenodo.15644343","title":"Estilo Vancouver o Formato NLM: hacia una estandarización de la citación","year":2025,"lang":"es","type":"review","venue":"PubMed","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Standardization; Citation; Style (visual arts); Library science; Information retrieval; Computer science; History; Archaeology; Operating system","score_opus":0.012995012673574492,"score_gpt":0.2994052255598795,"score_spread":0.28641021288630497,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412473765","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":"reporting","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":"reporting","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0032317538,0.7314252,0.06879408,0.10064803,0.057707485,0.0003329298,0.00053371233,0.0006191425,0.03670767],"genre_scores_gemma":[0.10192745,0.66042256,0.11881771,0.043590594,0.04475532,0.0019340725,0.0012454937,0.0007232223,0.026583621],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9196806,0.045305327,0.010050661,0.0028223186,0.021351025,0.0007901507],"domain_scores_gemma":[0.8529677,0.08350402,0.015323941,0.012443747,0.034496594,0.0012640118],"candidate_categories":["metaresearch","scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.04981583,0.0010817884,0.0017999619,0.013703232,0.002810534,0.012781346,0.0028618183,0.0027698933,0.00465488],"category_scores_gemma":[0.16277318,0.00065704784,0.0014849268,0.01942304,0.009958391,0.01017202,0.0048265764,0.0046239686,0.0032205964],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024316095,0.000028678725,0.0015476797,0.028224088,0.0002698994,0.00025950355,0.0033443468,0.0004773906,0.0010375178,0.2006733,0.118394,0.6455004],"study_design_scores_gemma":[0.00005092122,0.00007746497,0.0018157663,0.023314726,0.00025050537,0.0010491976,0.001536167,0.0006096082,0.0009481776,0.05126159,0.91901106,0.00007483714],"about_ca_topic_score_codex":0.005748652,"about_ca_topic_score_gemma":0.007495185,"teacher_disagreement_score":0.9872187,"about_ca_system_score_codex":0.0055954303,"about_ca_system_score_gemma":0.014514106,"threshold_uncertainty_score":0.26345444},"labels":[],"label_agreement":null},{"id":"W4412583802","doi":"10.1017/s1366728925100333","title":"Bilinguals process incoming words using distributions across both languages","year":2025,"lang":"en","type":"article","venue":"Bilingualism Language and Cognition","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"National Institute on Deafness and Other Communication Disorders; National Institutes of Health; York University","keywords":"Psychology; Linguistics; Process (computing); Neuroscience of multilingualism; Cognitive psychology; Computer science; Programming language","score_opus":0.01411071206970388,"score_gpt":0.36025253099594556,"score_spread":0.3461418189262417,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412583802","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9882157,0.00006136385,0.004577963,0.000086746186,0.000011117095,0.000008744593,0.00009940218,0.00005643132,0.0068826047],"genre_scores_gemma":[0.99565166,0.00006661082,0.002491828,0.000058408434,0.000005390562,0.000009941991,0.0001581718,0.000053213953,0.0015048326],"study_design_codex":"bench_or_experimental","study_design_gemma":"observational","domain_scores_codex":[0.9995629,0.00009079466,0.000027969245,0.00016080629,0.0000917931,0.00006568659],"domain_scores_gemma":[0.99872476,0.00040739294,0.00028942715,0.00017465938,0.00023362025,0.00017009661],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000476858,0.00022799656,0.00026145193,0.0004558981,0.00028633792,0.0014494598,0.00017082652,0.00023367531,0.0035480994],"category_scores_gemma":[0.002978422,0.0002686526,0.00014171224,0.0002459846,0.0005101842,0.0014465924,0.00086148287,0.0003772421,0.0009558758],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001330878,0.00030484036,0.16020693,0.0001809301,0.00010442082,0.0008267018,0.010784714,0.0009948423,0.6959869,0.0060002385,0.0008535805,0.12242495],"study_design_scores_gemma":[0.00025884836,0.0012310924,0.79529333,0.0001221848,0.00018005914,0.0036504983,0.01698381,0.016483786,0.1299561,0.025199117,0.0104794055,0.00016168498],"about_ca_topic_score_codex":0.0036696729,"about_ca_topic_score_gemma":0.006413136,"teacher_disagreement_score":0.0036696729,"about_ca_system_score_codex":0.0003289538,"about_ca_system_score_gemma":0.00041082682,"threshold_uncertainty_score":0.01186955},"labels":[],"label_agreement":null},{"id":"W4412621474","doi":"10.1016/j.jml.2025.104672","title":"Reframing linguistic bootstrapping as joint inference using visually-grounded grammar induction models","year":2025,"lang":"en","type":"article","venue":"Journal of Memory and Language","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; Mila - Quebec Artificial Intelligence Institute; HEC Montréal","funders":"","keywords":"Bootstrapping (finance); Inference; Psychology; Cognitive reframing; Linguistics; Grammar; Syntax; Pragmatics; Grammar induction; Generative grammar; Joint (building); Natural language processing; Cognitive psychology; Artificial intelligence; Computer science; Social psychology; Philosophy","score_opus":0.030662942060368722,"score_gpt":0.3322244435602664,"score_spread":0.3015615014998977,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412621474","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03957938,0.0001276803,0.95405424,0.000934109,0.00002470519,0.000040192404,0.0000752992,0.00044136008,0.0047229663],"genre_scores_gemma":[0.794603,0.0002218233,0.20208836,0.0002910117,0.000042630138,0.00016975809,0.0001902138,0.00016896956,0.0022242055],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9981001,0.0009866703,0.00006538549,0.000390814,0.00033344293,0.0001235895],"domain_scores_gemma":[0.9889178,0.008120485,0.0006485338,0.001653587,0.00041082953,0.00024879907],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035611414,0.00063366117,0.0007801151,0.00095996266,0.00047799625,0.002477253,0.0026387298,0.0014549652,0.0030940615],"category_scores_gemma":[0.018924113,0.00075364293,0.001584452,0.00068708166,0.0037261439,0.0065844427,0.003626079,0.003172119,0.000503793],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015426872,0.0002085469,0.0057458854,0.00027370107,0.00020679773,0.00071283866,0.002465253,0.23079301,0.015398498,0.632156,0.0014344829,0.110450745],"study_design_scores_gemma":[0.000014507329,0.000039184662,0.0004966505,0.000020329511,0.000019872672,0.00010617003,0.000085385305,0.5043854,0.0023604212,0.49163207,0.00081509695,0.00002478516],"about_ca_topic_score_codex":0.0016999949,"about_ca_topic_score_gemma":0.002202152,"teacher_disagreement_score":0.0035611414,"about_ca_system_score_codex":0.0011458233,"about_ca_system_score_gemma":0.00092440797,"threshold_uncertainty_score":0.01883334},"labels":[],"label_agreement":null},{"id":"W4412661434","doi":"10.1101/2025.07.22.666196","title":"Efficient Grammar Compression via RLZ-based RePair","year":2025,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"Natural Sciences and Engineering Research Council of Canada; National Institutes of Health; National Science Foundation","keywords":"Bigram; Computer science; Parsing; Encoding (memory); Substring; Rule-based machine translation; Grammar; String (physics); Scalability; Artificial intelligence; Natural language processing; Database; Data structure; Programming language; Mathematics; Linguistics; Trigram","score_opus":0.00963221893959893,"score_gpt":0.23530764497259612,"score_spread":0.2256754260329972,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412661434","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02417491,0.0004677381,0.94750416,0.00032402112,0.00013095909,0.00015236207,0.0015086033,0.023467494,0.0022697954],"genre_scores_gemma":[0.15423061,0.00042548883,0.8277446,0.00045605947,0.00009888473,0.00037926497,0.0070196083,0.0040243347,0.0056211227],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.998206,0.0002775437,0.00017717692,0.0005570734,0.00065660296,0.00012552275],"domain_scores_gemma":[0.99613756,0.0015208045,0.0002488208,0.0015383967,0.00049462594,0.0000598241],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011699537,0.0011678782,0.0011738631,0.0018795531,0.00068395527,0.0013636436,0.0020151858,0.0012843295,0.0041135554],"category_scores_gemma":[0.007822033,0.00065385853,0.0012227393,0.0023694197,0.001344288,0.0024515525,0.002649339,0.0018407326,0.0040161214],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00042550848,0.00017974754,0.0020688,0.0007031967,0.00012408219,0.00088436966,0.0011110125,0.07749937,0.11590227,0.044098277,0.032351375,0.72465193],"study_design_scores_gemma":[0.00014756125,0.00019428793,0.0012729757,0.00009333851,0.0000995166,0.0010555997,0.0004469102,0.6784729,0.18566085,0.09202866,0.040401135,0.00012623725],"about_ca_topic_score_codex":0.0016660953,"about_ca_topic_score_gemma":0.0026868351,"teacher_disagreement_score":0.0041135554,"about_ca_system_score_codex":0.00075440423,"about_ca_system_score_gemma":0.001616079,"threshold_uncertainty_score":0.013761222},"labels":[],"label_agreement":null},{"id":"W4412703854","doi":"10.1145/3696630.3728568","title":"Can Generative AI Produce Test Cases? An Experience from the Automotive Domain","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"European Commission","keywords":"Automotive industry; Generative grammar; Computer science; Test (biology); Domain (mathematical analysis); Artificial intelligence; Engineering; Mathematics; Aerospace engineering","score_opus":0.01276337248502687,"score_gpt":0.3032364019357941,"score_spread":0.29047302945076725,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412703854","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.71373767,0.0011747343,0.2596636,0.002129753,0.000035956946,0.0003177997,0.000097144504,0.0017598629,0.021083415],"genre_scores_gemma":[0.8680184,0.0004851192,0.12857154,0.00038591743,0.000015187173,0.000085924665,0.0002217429,0.00035648356,0.0018596234],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9857087,0.010616409,0.0005791121,0.0006228497,0.0020708346,0.000402184],"domain_scores_gemma":[0.8883808,0.09760378,0.0015162809,0.005905443,0.0056326916,0.000960902],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012324515,0.0006673129,0.00035719856,0.0010386133,0.00071615935,0.0016139807,0.0022171976,0.0013505142,0.002264876],"category_scores_gemma":[0.05623541,0.00039064538,0.0005249012,0.0008661853,0.0019789198,0.002406304,0.001556679,0.0014022533,0.000769869],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009964699,0.002693764,0.04552092,0.0014903562,0.00014563341,0.004085678,0.056531493,0.06419841,0.046703804,0.040249515,0.008034193,0.7293498],"study_design_scores_gemma":[0.0008517057,0.0063660517,0.044748213,0.001290171,0.00042021132,0.0109503325,0.031020733,0.42730528,0.19933157,0.06634631,0.21096863,0.00040083463],"about_ca_topic_score_codex":0.002811761,"about_ca_topic_score_gemma":0.0033050822,"teacher_disagreement_score":0.012324515,"about_ca_system_score_codex":0.0011741929,"about_ca_system_score_gemma":0.0009102572,"threshold_uncertainty_score":0.06517905},"labels":[],"label_agreement":null},{"id":"W4412886796","doi":"10.18653/v1/2025.acl-long.111","title":"LLäMmlein: Transparent, Compact and Competitive German-Only Language Models from Scratch","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"Julius-Maximilians-Universität Würzburg; Deutsche Forschungsgemeinschaft","keywords":"Scratch; German; Computer science; Programming language; Linguistics; Philosophy","score_opus":0.015928012269990098,"score_gpt":0.30342456212995983,"score_spread":0.28749654985996975,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412886796","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.106503494,0.0027028387,0.5773537,0.0020964365,0.0013782525,0.00067304034,0.03215015,0.23894918,0.038192812],"genre_scores_gemma":[0.46132946,0.001261907,0.36391026,0.001701506,0.00014490585,0.0011534389,0.12329778,0.016938785,0.030261938],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9991229,0.00025259383,0.000059646,0.00026621783,0.00018142744,0.00011716048],"domain_scores_gemma":[0.9988398,0.0004766347,0.00005101116,0.00040078105,0.00016289203,0.00006892167],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015200763,0.0025899333,0.00091243815,0.0009019253,0.00044174385,0.0019458057,0.0033130618,0.0017985728,0.013150015],"category_scores_gemma":[0.0048560393,0.001326314,0.0013882066,0.0006594691,0.000549509,0.0039226296,0.0028059152,0.0025029762,0.0113808885],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001981875,0.0007690516,0.0043758443,0.0015185683,0.000778075,0.0010306046,0.00048016614,0.36111963,0.024717625,0.020043086,0.22728205,0.35590357],"study_design_scores_gemma":[0.00030763212,0.00033171097,0.00093636615,0.0001196412,0.00012930649,0.0003250991,0.00011187993,0.90925497,0.025794376,0.011204566,0.05137981,0.00010460569],"about_ca_topic_score_codex":0.008698596,"about_ca_topic_score_gemma":0.021420946,"teacher_disagreement_score":0.013150015,"about_ca_system_score_codex":0.0010666755,"about_ca_system_score_gemma":0.0016500509,"threshold_uncertainty_score":0.04399115},"labels":[],"label_agreement":null},{"id":"W4412886925","doi":"10.18653/v1/2025.acl-industry.56","title":"Enriching children’s stories with LLMs: Delivering multilingual data enrichment for children’s books at scale and across markets","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Nexen (Canada)","funders":"","keywords":"Scale (ratio); Computer science; Data science; Geography; Cartography","score_opus":0.011351127467519116,"score_gpt":0.2958524022356139,"score_spread":0.2845012747680948,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412886925","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17364761,0.0010560827,0.72018564,0.0021897824,0.00014847638,0.0009345628,0.005831598,0.042524613,0.053481545],"genre_scores_gemma":[0.3628719,0.0004622109,0.6099271,0.0006485782,0.0000546711,0.0007017054,0.0052755596,0.0037592715,0.016299037],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.998486,0.0005931422,0.00006493982,0.0003292052,0.00043599942,0.00009072453],"domain_scores_gemma":[0.99206614,0.005029222,0.0002627697,0.0014111097,0.0007964946,0.00043426437],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022647907,0.0007395712,0.0004280801,0.0015297432,0.000921668,0.0031168628,0.001350187,0.0007683135,0.0115377065],"category_scores_gemma":[0.013807385,0.00041808144,0.0005659361,0.0012507758,0.0011310568,0.0056133945,0.006217894,0.001415866,0.0042809965],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009059735,0.0005020946,0.018150477,0.0019805382,0.00014429673,0.0014670795,0.02344073,0.0061946157,0.09381591,0.020274274,0.030796207,0.8023279],"study_design_scores_gemma":[0.00022384868,0.0006268054,0.027536308,0.0005506517,0.00022817534,0.0017418672,0.017007904,0.0821497,0.23123495,0.054596845,0.5837544,0.0003485944],"about_ca_topic_score_codex":0.0028369776,"about_ca_topic_score_gemma":0.00905707,"teacher_disagreement_score":0.0115377065,"about_ca_system_score_codex":0.000842572,"about_ca_system_score_gemma":0.0012259824,"threshold_uncertainty_score":0.038597465},"labels":[],"label_agreement":null},{"id":"W4412887895","doi":"10.18653/v1/2025.findings-acl.1110","title":"TagRouter: Learning Route to LLMs through Tags for Open-Domain Text Generation Tasks","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Institute for Catastrophic Loss Reduction","keywords":"Computer science; Domain (mathematical analysis); Open domain; World Wide Web; Natural language processing; Question answering","score_opus":0.024820926393225973,"score_gpt":0.33181410655366506,"score_spread":0.3069931801604391,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412887895","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04385058,0.00093562965,0.8704668,0.00082023686,0.0003552698,0.00032271497,0.0015068469,0.07805411,0.003687799],"genre_scores_gemma":[0.32526556,0.00034780958,0.65353864,0.00087696075,0.00014412106,0.00047981704,0.0066583194,0.0044780737,0.0082107335],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99918276,0.0003031432,0.000042809745,0.00029289236,0.00011380213,0.00006468074],"domain_scores_gemma":[0.9977895,0.001293098,0.000083615654,0.0005017743,0.00021876808,0.00011325038],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015916403,0.001563234,0.0007631043,0.0012071357,0.0007737797,0.001536133,0.002646732,0.0023572876,0.007382111],"category_scores_gemma":[0.007748407,0.00066308025,0.0010424173,0.0009751168,0.0006705342,0.0040230877,0.0019794153,0.0023880214,0.0045658797],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000704747,0.00050218013,0.0052287937,0.0005483477,0.00023030982,0.0004890736,0.0007170877,0.14325239,0.022264756,0.013369234,0.08031632,0.73237675],"study_design_scores_gemma":[0.000076273536,0.000094810865,0.00020883081,0.00001821676,0.000033094122,0.00010534314,0.00009529592,0.97399,0.0072479867,0.011527659,0.006578891,0.000023590299],"about_ca_topic_score_codex":0.004708871,"about_ca_topic_score_gemma":0.011350011,"teacher_disagreement_score":0.007382111,"about_ca_system_score_codex":0.0011134783,"about_ca_system_score_gemma":0.0016626623,"threshold_uncertainty_score":0.024695694},"labels":[],"label_agreement":null},{"id":"W4412887991","doi":"10.18653/v1/2025.findings-acl.818","title":"A Fully Automated Pipeline for Conversational Discourse Annotation: Tree Scheme Generation and Labeling with Large Language Models","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Computer science; Scheme (mathematics); Annotation; Pipeline (software); Tree (set theory); Natural language processing; Artificial intelligence; Programming language","score_opus":0.013420934460764996,"score_gpt":0.3018274128048487,"score_spread":0.2884064783440837,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412887991","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004218245,0.00010108573,0.93498313,0.0003775186,0.000102542755,0.00025611045,0.0036158601,0.05309852,0.0032469812],"genre_scores_gemma":[0.04618467,0.0000980967,0.9363794,0.00016066385,0.00003766487,0.00048774775,0.0091908015,0.0045929034,0.0028681895],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9975068,0.00097259047,0.00017883617,0.00060219655,0.0006059951,0.00013363577],"domain_scores_gemma":[0.99208015,0.004157308,0.0002919887,0.0018047272,0.001378228,0.00028754293],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035740088,0.0016903395,0.0008568395,0.0024124058,0.0017996423,0.0024513577,0.00220162,0.001581539,0.016647967],"category_scores_gemma":[0.016484218,0.0011607152,0.001733231,0.0016908161,0.0008837171,0.0043656514,0.0038176535,0.003084868,0.013192277],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00045949782,0.0002635014,0.0028038837,0.0011416174,0.00011390167,0.00041786433,0.00459628,0.01697401,0.048498,0.058121577,0.15386595,0.7127439],"study_design_scores_gemma":[0.0000932795,0.00008933465,0.0014804734,0.00025482127,0.00007622107,0.0003207416,0.00127411,0.62761647,0.046224136,0.12823237,0.19417925,0.00015875658],"about_ca_topic_score_codex":0.006552664,"about_ca_topic_score_gemma":0.012018675,"teacher_disagreement_score":0.016647967,"about_ca_system_score_codex":0.00163181,"about_ca_system_score_gemma":0.0042805574,"threshold_uncertainty_score":0.05569297},"labels":[],"label_agreement":null},{"id":"W4412888004","doi":"10.18653/v1/2025.findings-acl.763","title":"REVS: Unlearning Sensitive Information in Language Models via Rank Editing in the Vocabulary Space","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Azrieli Foundation; European Commission; Open Philanthropy Project","keywords":"Vocabulary; Computer science; Rank (graph theory); Space (punctuation); Language model; Natural language processing; Artificial intelligence; Information retrieval; Linguistics; Mathematics; Operating system","score_opus":0.005588018851145987,"score_gpt":0.25062263813303615,"score_spread":0.24503461928189016,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412888004","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021256648,0.00064860587,0.9663404,0.00048141673,0.0001423429,0.00012477627,0.00045156784,0.009446566,0.0011076404],"genre_scores_gemma":[0.5415414,0.0006805538,0.44040102,0.0013384518,0.00039241413,0.00037149002,0.0035812885,0.0017880923,0.0099052815],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9974597,0.0009751059,0.00018880822,0.000627117,0.00058440684,0.00016490306],"domain_scores_gemma":[0.9938419,0.0032955236,0.000467412,0.001600767,0.0006346502,0.00015980496],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029315879,0.0020768514,0.0015890327,0.0011885585,0.0005540723,0.0015143314,0.002558419,0.001483334,0.0027115173],"category_scores_gemma":[0.014604487,0.0006806231,0.0016570741,0.0008538809,0.0013242521,0.0035693438,0.0026818363,0.0039251624,0.0021826574],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005269176,0.0003510314,0.0029112475,0.0004360265,0.00032581782,0.0003259897,0.00045673377,0.1882054,0.025569791,0.013269958,0.018250704,0.7493704],"study_design_scores_gemma":[0.000047042337,0.00018242991,0.00028167455,0.00002170289,0.000034626588,0.00013863415,0.00005486102,0.9666686,0.011672631,0.018158404,0.0027066455,0.000032860833],"about_ca_topic_score_codex":0.0032322647,"about_ca_topic_score_gemma":0.006736552,"teacher_disagreement_score":0.0032322647,"about_ca_system_score_codex":0.0007855222,"about_ca_system_score_gemma":0.001462114,"threshold_uncertainty_score":0.015503943},"labels":[],"label_agreement":null},{"id":"W4412889209","doi":"10.18653/v1/2025.bea-1.33","title":"LangEye: Toward ‘Anytime’ Learner-Driven Vocabulary Learning From Real-World Objects","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ontario Tech University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Vocabulary; Vocabulary learning; Human–computer interaction; Artificial intelligence; Multimedia; Natural language processing; Linguistics","score_opus":0.016007382050864133,"score_gpt":0.28462719855945984,"score_spread":0.2686198165085957,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412889209","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0690628,0.0011126288,0.8782616,0.00043935506,0.00010766874,0.0005850184,0.0008865597,0.039744876,0.009799373],"genre_scores_gemma":[0.24377507,0.0009057734,0.7280039,0.00043434853,0.000055837096,0.0010708022,0.0030114476,0.0026377805,0.020105004],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99907815,0.00034344406,0.00005035431,0.0002298334,0.00023296768,0.00006518034],"domain_scores_gemma":[0.99772006,0.0014119619,0.000099137644,0.00037566037,0.0002416953,0.00015145056],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010194833,0.0008865319,0.0004869667,0.00060163776,0.00026170854,0.0014612762,0.0019900354,0.0010133339,0.0065041985],"category_scores_gemma":[0.006428849,0.0004237169,0.0006857021,0.00026342814,0.00052429613,0.0031540003,0.0036794718,0.00093471847,0.0038453469],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007768202,0.0010430334,0.0055314978,0.0026470702,0.00015106268,0.0015145164,0.009150399,0.008817183,0.123603195,0.011548527,0.03746075,0.79775596],"study_design_scores_gemma":[0.00043696834,0.0023327805,0.01081637,0.0008805345,0.0002460867,0.0046696207,0.006581845,0.15606071,0.20842774,0.03767947,0.57142425,0.00044360667],"about_ca_topic_score_codex":0.00068433926,"about_ca_topic_score_gemma":0.0023829262,"teacher_disagreement_score":0.0065041985,"about_ca_system_score_codex":0.00028347713,"about_ca_system_score_gemma":0.0005767541,"threshold_uncertainty_score":0.021758735},"labels":[],"label_agreement":null},{"id":"W4412889244","doi":"10.18653/v1/2025.bea-1.38","title":"LLMs in alliance with Edit-based models: advancing In-Context Learning for Grammatical Error Correction by Specific Example Selection","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Selection (genetic algorithm); Computer science; Context (archaeology); Alliance; Artificial intelligence; Natural language processing; Machine learning; Political science; History","score_opus":0.016082606442576752,"score_gpt":0.26795653593910956,"score_spread":0.2518739294965328,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412889244","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.22961222,0.0021739448,0.72585356,0.0011031572,0.00044583608,0.00016921012,0.0008111009,0.03436064,0.0054703625],"genre_scores_gemma":[0.77569056,0.00033289552,0.21400845,0.0005674875,0.0001220138,0.00012591515,0.0017709936,0.0012742182,0.0061072996],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9985123,0.00076046155,0.00005824058,0.00038367114,0.00018593315,0.00009940598],"domain_scores_gemma":[0.99634343,0.001807639,0.000146851,0.0010420375,0.0005120943,0.00014787876],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002313992,0.0011195451,0.00089986005,0.00087320374,0.0005079826,0.001122492,0.0019171878,0.0012882634,0.003835159],"category_scores_gemma":[0.0073015024,0.00045028212,0.0006439841,0.0005503335,0.00055562,0.002497785,0.0021155414,0.0023207578,0.0021244243],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00086132565,0.00049447815,0.006936207,0.0003021298,0.0002816592,0.0003919467,0.0005875633,0.15182632,0.025822654,0.005386182,0.011262388,0.7958471],"study_design_scores_gemma":[0.000029766868,0.00018997936,0.0005058175,0.000024222714,0.000043025942,0.00009457871,0.00006939345,0.9745292,0.015677605,0.0061432123,0.0026737172,0.000019537216],"about_ca_topic_score_codex":0.00448893,"about_ca_topic_score_gemma":0.011188544,"teacher_disagreement_score":0.00448893,"about_ca_system_score_codex":0.0005045427,"about_ca_system_score_gemma":0.0010190338,"threshold_uncertainty_score":0.0128299},"labels":[],"label_agreement":null},{"id":"W4412889332","doi":"10.18653/v1/2025.africanlp-1","title":"Proceedings of the Sixth Workshop on African Natural Language Processing (AfricaNLP 2025)","year":2025,"lang":"en","type":"paratext","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Fundação para a Ciência e a Tecnologia; Universidade do Porto; National Research Foundation; Division of Mathematical Sciences; Government of Canada; African Institute for Mathematical Sciences; DeepMind; Wikimedia Foundation; Natural Sciences and Engineering Research Council of Canada; Bill and Melinda Gates Foundation","keywords":"Computer science; Natural (archaeology); Natural language; Programming language; Natural language processing; History; Archaeology","score_opus":0.010340328325992196,"score_gpt":0.2820152194514262,"score_spread":0.271674891125434,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412889332","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022371763,0.0671014,0.445233,0.11716872,0.06529019,0.002369114,0.019017631,0.013208453,0.24823971],"genre_scores_gemma":[0.06282845,0.037158143,0.2639222,0.012669517,0.010656037,0.0024998498,0.062327154,0.006989466,0.5409493],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9936591,0.0033332387,0.00041849684,0.0007894261,0.0012781606,0.00052171265],"domain_scores_gemma":[0.98902845,0.005257283,0.00024079373,0.0013684948,0.0022601127,0.0018448102],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012193059,0.0015191371,0.0013743078,0.0030406888,0.0020648108,0.008338282,0.003059675,0.0034429545,0.08905404],"category_scores_gemma":[0.016427984,0.00062794704,0.0012745404,0.0027762887,0.0019842053,0.010325843,0.007264493,0.0049106036,0.03400268],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003094669,0.0002593477,0.00039896916,0.0007360967,0.000055385026,0.00041311965,0.0011709309,0.00085513154,0.0036068624,0.011692711,0.70541775,0.2750843],"study_design_scores_gemma":[0.00002747349,0.00005472695,0.0005145101,0.0003253136,0.000019464629,0.0002264407,0.00066270906,0.002595331,0.0016285309,0.008387101,0.9855363,0.000022158485],"about_ca_topic_score_codex":0.005306927,"about_ca_topic_score_gemma":0.008883292,"teacher_disagreement_score":0.08905404,"about_ca_system_score_codex":0.00195741,"about_ca_system_score_gemma":0.0042277556,"threshold_uncertainty_score":0.29791546},"labels":[],"label_agreement":null},{"id":"W4412889469","doi":"10.18653/v1/2025.acl-srw.83","title":"DRUM: Learning Demonstration Retriever for Large MUlti-modal Models","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Modal; Labrador Retriever; Computer science; Drum; Artificial intelligence; Engineering; Mechanical engineering; Materials science; Medicine","score_opus":0.018251504182418603,"score_gpt":0.3018993524396694,"score_spread":0.2836478482572508,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412889469","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03372114,0.0018036101,0.92905986,0.0004450106,0.00018185769,0.0002758976,0.00089006743,0.031827383,0.0017951501],"genre_scores_gemma":[0.3698415,0.000772875,0.6125662,0.0010935375,0.00024212374,0.0007847713,0.0060919966,0.0015133716,0.0070936894],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99865997,0.0003680448,0.00008802306,0.0005154182,0.00023293568,0.00013556253],"domain_scores_gemma":[0.9978369,0.0010190628,0.00014586704,0.00052921503,0.00029680214,0.00017221845],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029980766,0.0029401234,0.001960424,0.0011007257,0.0006702898,0.0010499316,0.004486444,0.002251091,0.008222173],"category_scores_gemma":[0.007773464,0.0008436679,0.0017655797,0.0007940942,0.00069323805,0.004082715,0.003683365,0.0040576276,0.0032627564],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00066960196,0.00047551788,0.0018576407,0.0005615257,0.00026987414,0.00032296876,0.00020530022,0.13957861,0.015506009,0.0054287566,0.02965871,0.80546546],"study_design_scores_gemma":[0.000061261584,0.00020993751,0.00020829323,0.000014049829,0.00002800798,0.00007859342,0.000041270883,0.989235,0.004090387,0.0040858663,0.0019256013,0.000021756441],"about_ca_topic_score_codex":0.004726193,"about_ca_topic_score_gemma":0.0077069034,"teacher_disagreement_score":0.008222173,"about_ca_system_score_codex":0.0010476395,"about_ca_system_score_gemma":0.0011585343,"threshold_uncertainty_score":0.027505875},"labels":[],"label_agreement":null},{"id":"W4412889667","doi":"10.18653/v1/2025.acl-long.1572","title":"Where Are We? Evaluating LLM Performance on African Languages","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs; Bill and Melinda Gates Foundation","keywords":"Computer science; Natural language processing; Programming language","score_opus":0.021447731710652715,"score_gpt":0.33377185851615604,"score_spread":0.3123241268055033,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412889667","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.97464764,0.0027906438,0.004303932,0.0019740695,0.00019787508,0.00006894143,0.0029407945,0.0023168428,0.010759317],"genre_scores_gemma":[0.9798788,0.0006704686,0.010504716,0.00016585809,0.000047047917,0.000043967597,0.005708483,0.00017121964,0.0028093918],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99899,0.00052837963,0.00007399702,0.00013341138,0.00013225507,0.00014196854],"domain_scores_gemma":[0.9975139,0.0014612464,0.00012131275,0.00022663988,0.0004225719,0.00025427988],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001749413,0.00054421456,0.00046348904,0.0010797174,0.00070991355,0.0011604638,0.0006546526,0.00083293975,0.0034991181],"category_scores_gemma":[0.0066428846,0.00016554179,0.00028799323,0.0013165778,0.00030657463,0.0026525422,0.0014612352,0.00053806556,0.0023860433],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.006468008,0.00094468,0.0966348,0.0015146135,0.0003073874,0.0007465523,0.0036119283,0.012301203,0.033927504,0.003860324,0.0691418,0.77054113],"study_design_scores_gemma":[0.0010745133,0.004963915,0.24855636,0.0013084367,0.00071517297,0.0016146408,0.036167845,0.3986831,0.13936515,0.013334554,0.15388073,0.00033558471],"about_ca_topic_score_codex":0.0076940884,"about_ca_topic_score_gemma":0.009972519,"teacher_disagreement_score":0.0076940884,"about_ca_system_score_codex":0.00056010316,"about_ca_system_score_gemma":0.0005244671,"threshold_uncertainty_score":0.015298605},"labels":[],"label_agreement":null},{"id":"W4412900511","doi":"10.1145/3721251.3742863","title":"Stylization in Unreal Engine","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"BC Innovation Council","funders":"","keywords":"Computer science; Artificial intelligence; Computer graphics (images)","score_opus":0.00451688431136603,"score_gpt":0.2665896534636248,"score_spread":0.26207276915225874,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412900511","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.030393682,0.00031146637,0.80135083,0.0006890172,0.0006415285,0.00026559253,0.002482684,0.0480456,0.115819536],"genre_scores_gemma":[0.5126588,0.00036884577,0.3949141,0.0004692757,0.000090248875,0.00016675278,0.00446194,0.008591638,0.07827843],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99882334,0.00028106722,0.00010322219,0.00021640913,0.00046407018,0.000111919864],"domain_scores_gemma":[0.9992236,0.00021273,0.000025836676,0.00038800854,0.0001274008,0.00002240623],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00090359664,0.0005436383,0.00050944317,0.000842485,0.0007210373,0.002389053,0.0011571865,0.0007096728,0.03131786],"category_scores_gemma":[0.0034431282,0.0006426304,0.0007339819,0.0005778681,0.000740161,0.002454481,0.0017400798,0.0011713643,0.0075566694],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005783173,0.00012370832,0.0014956685,0.00032674728,0.000043595246,0.00066712854,0.0007529169,0.0289242,0.02066457,0.6299854,0.056380633,0.26005712],"study_design_scores_gemma":[0.000110544985,0.00009096652,0.0007103247,0.000116656935,0.00007704562,0.00067669305,0.0002513123,0.37473884,0.072382584,0.17748882,0.37326628,0.00008989593],"about_ca_topic_score_codex":0.0034635453,"about_ca_topic_score_gemma":0.005578206,"teacher_disagreement_score":0.03131786,"about_ca_system_score_codex":0.00080403156,"about_ca_system_score_gemma":0.0009723419,"threshold_uncertainty_score":0.10476869},"labels":[],"label_agreement":null},{"id":"W4412933877","doi":"10.1109/ethics65148.2025.11098285","title":"Impact Evaluation of AI-Language Access Models for Indigenous Language Communities in the City of Los Angeles","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Indigenous; Indigenous language; Computer science; Linguistics; Natural language processing; Artificial intelligence","score_opus":0.060987341053202596,"score_gpt":0.41688556187253306,"score_spread":0.3558982208193305,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412933877","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9803018,0.0001938548,0.0066687404,0.0007169089,0.00004672618,0.00081794313,0.00039839195,0.00075951667,0.010096143],"genre_scores_gemma":[0.9853373,0.00013320782,0.011399641,0.00013128178,0.000011916673,0.00055022194,0.00053244777,0.000051428236,0.0018525867],"study_design_codex":"simulation_or_modeling","study_design_gemma":"observational","domain_scores_codex":[0.9888903,0.008158317,0.00042456496,0.00075624225,0.0011821539,0.00058833504],"domain_scores_gemma":[0.95537466,0.033867378,0.00135465,0.0020114856,0.0053562685,0.0020356448],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01872676,0.0012867808,0.00066869275,0.0014733888,0.0012567963,0.004323235,0.0021590055,0.0013970901,0.0050766594],"category_scores_gemma":[0.049394496,0.0003939408,0.00088891,0.001072927,0.001524099,0.0035445443,0.0028980193,0.0015374332,0.00055627414],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.01543267,0.015137745,0.07689821,0.0018432373,0.0011050793,0.0005005106,0.0043172254,0.6403965,0.005967605,0.021752965,0.008881151,0.2077671],"study_design_scores_gemma":[0.0009972344,0.007815132,0.013043611,0.00017356445,0.00055004016,0.000056683315,0.0036794231,0.95856804,0.004966141,0.005183528,0.0048581287,0.00010845725],"about_ca_topic_score_codex":0.1250189,"about_ca_topic_score_gemma":0.06591737,"teacher_disagreement_score":0.1250189,"about_ca_system_score_codex":0.009848545,"about_ca_system_score_gemma":0.0057494156,"threshold_uncertainty_score":0.24858242},"labels":[],"label_agreement":null},{"id":"W4412944139","doi":"10.18653/v1/2025.trl-1.10","title":"Table Understanding and (Multimodal) LLMs: A Cross-Domain Case Study on Scientific vs. Non-Scientific Data","year":2025,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Deutsche Forschungsgemeinschaft; Institute for Catastrophic Loss Reduction","keywords":"Table (database); Computer science; Domain (mathematical analysis); Data science; Data mining; Mathematics","score_opus":0.0829351396306667,"score_gpt":0.37094880910032474,"score_spread":0.288013669469658,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412944139","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.777819,0.0070055993,0.14176479,0.00501266,0.00066138024,0.0015747588,0.023600718,0.019642143,0.022918988],"genre_scores_gemma":[0.7323015,0.0016947394,0.22587173,0.0016522617,0.00020542718,0.0007148769,0.029551048,0.001923665,0.0060847723],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9905902,0.005087885,0.0008251725,0.0015500749,0.0016319014,0.00031476893],"domain_scores_gemma":[0.9305237,0.05750067,0.0017311944,0.0060733417,0.003306689,0.0008642671],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0082236435,0.0011745191,0.0007303384,0.003037422,0.0008294193,0.0030036245,0.001476963,0.0022794944,0.005755227],"category_scores_gemma":[0.047752324,0.00029488796,0.0011518702,0.0029240919,0.0011029793,0.008015174,0.0039686924,0.0018991649,0.0022316012],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0043564807,0.0030234535,0.037395958,0.013438012,0.0006693424,0.0076932805,0.021701664,0.0473615,0.049464744,0.011462068,0.12542202,0.67801154],"study_design_scores_gemma":[0.00075675757,0.0031121653,0.06458434,0.0029228884,0.00065639196,0.011337249,0.04626337,0.33944574,0.12873891,0.034606766,0.36692312,0.0006523347],"about_ca_topic_score_codex":0.004218391,"about_ca_topic_score_gemma":0.004981005,"teacher_disagreement_score":0.0082236435,"about_ca_system_score_codex":0.0010612038,"about_ca_system_score_gemma":0.001078304,"threshold_uncertainty_score":0.043491244},"labels":[],"label_agreement":null},{"id":"W4412944500","doi":"10.18653/v1/2025.findings-acl.1364","title":"NBDESCRIB: A Dataset for Text Description Generation from Tables and Code in Jupyter Notebooks with Guidelines","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Code (set theory); Programming language; Data science; Natural language processing; Information retrieval","score_opus":0.05870056497911689,"score_gpt":0.3220833197584907,"score_spread":0.2633827547793738,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412944500","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.039243117,0.0020609933,0.017814193,0.00073903496,0.0004035458,0.001083683,0.8558703,0.07376622,0.009018921],"genre_scores_gemma":[0.017801134,0.00027593857,0.025908044,0.00021726833,0.000020333913,0.0007380201,0.9512051,0.0009933211,0.0028408854],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99866986,0.000265131,0.00018349472,0.0003779789,0.00041573078,0.0000879085],"domain_scores_gemma":[0.99631655,0.0017126908,0.00027375962,0.00074865855,0.00068163115,0.00026662883],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00090965815,0.003102669,0.0007657902,0.0032562453,0.0008465665,0.0012083263,0.002645266,0.002336027,0.014204894],"category_scores_gemma":[0.007577259,0.00064309203,0.001389606,0.0023759638,0.0005970872,0.002298606,0.0019365575,0.002039701,0.015407941],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00086305145,0.0005344905,0.008339714,0.004373044,0.00015663868,0.0013402643,0.00070766394,0.007674532,0.007567021,0.0019232411,0.8582422,0.10827807],"study_design_scores_gemma":[0.0010204838,0.00054530334,0.030912075,0.0008713881,0.00013664237,0.0015237593,0.0012025258,0.07985227,0.027728861,0.005110986,0.85078824,0.0003074269],"about_ca_topic_score_codex":0.022393972,"about_ca_topic_score_gemma":0.05270658,"teacher_disagreement_score":0.022393972,"about_ca_system_score_codex":0.001808957,"about_ca_system_score_gemma":0.0019073071,"threshold_uncertainty_score":0.0475201},"labels":[],"label_agreement":null},{"id":"W4412944786","doi":"10.18653/v1/2025.acl-long.1141","title":"Language Models Resist Alignment: Evidence From Data Compression","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"National Natural Science Foundation of China","keywords":"Resist; Computer science; Compression (physics); Data compression; Artificial intelligence; Materials science; Composite material","score_opus":0.05182750288185463,"score_gpt":0.34016920221343316,"score_spread":0.28834169933157855,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412944786","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6673137,0.007058656,0.25737536,0.018252114,0.001630532,0.00028840828,0.0031864408,0.0034010133,0.041493796],"genre_scores_gemma":[0.9545805,0.001420084,0.03444781,0.0017017396,0.0007618286,0.0001600212,0.0032439637,0.00056331226,0.0031207607],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9887488,0.006019089,0.0006939365,0.0014716367,0.0024654162,0.00060119724],"domain_scores_gemma":[0.7635953,0.16924144,0.010978481,0.04284872,0.011773172,0.0015629132],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013202698,0.0010468279,0.0018161045,0.002612076,0.0022614896,0.0037153435,0.0026680694,0.003271726,0.006367],"category_scores_gemma":[0.15508924,0.0010825471,0.000777858,0.0041414816,0.00459418,0.010226425,0.0036636302,0.0043565705,0.0042131003],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.010111587,0.0013523076,0.1016557,0.0014337845,0.0011460409,0.003564297,0.0038245376,0.07376916,0.020610547,0.16126153,0.059105672,0.56216484],"study_design_scores_gemma":[0.0010755482,0.00070139335,0.019753875,0.00026844844,0.0006770305,0.0029571205,0.0020714027,0.44155568,0.031238616,0.48282516,0.016654255,0.00022159916],"about_ca_topic_score_codex":0.0015964038,"about_ca_topic_score_gemma":0.0015360971,"teacher_disagreement_score":0.013202698,"about_ca_system_score_codex":0.0007373221,"about_ca_system_score_gemma":0.001709836,"threshold_uncertainty_score":0.069823384},"labels":[],"label_agreement":null},{"id":"W4412944910","doi":"10.18653/v1/2025.acl-long.875","title":"Enhancing Text Editing for Grammatical Error Correction: Arabic as a Case Study","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"York University; New York University Abu Dhabi","keywords":"Computer science; Arabic; Natural language processing; Artificial intelligence; Error analysis; Error detection and correction; Linguistics; Speech recognition; Algorithm; Mathematics; Philosophy","score_opus":0.015710222808662317,"score_gpt":0.3346478321451965,"score_spread":0.3189376093365342,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412944910","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6808166,0.0041816137,0.23566915,0.005276644,0.0018077818,0.000730561,0.0050174342,0.046424057,0.020076187],"genre_scores_gemma":[0.71501464,0.0010480046,0.2585824,0.0010065091,0.0002663379,0.00018837026,0.007317267,0.002234476,0.01434194],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9984101,0.00062291574,0.00012986944,0.00038356695,0.00035040049,0.00010319051],"domain_scores_gemma":[0.9906969,0.004890081,0.0003917732,0.0014119603,0.0022871573,0.00032218674],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00202993,0.0017343608,0.0007171008,0.001224077,0.0010090547,0.0015064627,0.0016458388,0.0015065373,0.0030159198],"category_scores_gemma":[0.012669183,0.0002278099,0.00062651595,0.0012466053,0.0006737791,0.0020757893,0.0011211833,0.002000463,0.0023120411],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010455833,0.00093444606,0.013503334,0.0013200064,0.00022766972,0.0039990963,0.0023449308,0.1675988,0.034253705,0.0061185616,0.06840714,0.70024675],"study_design_scores_gemma":[0.0001251523,0.00047046103,0.005158321,0.00016727166,0.00014282242,0.0019010438,0.0012939221,0.80710465,0.1017934,0.008167123,0.073525354,0.00015050796],"about_ca_topic_score_codex":0.010867525,"about_ca_topic_score_gemma":0.013936099,"teacher_disagreement_score":0.010867525,"about_ca_system_score_codex":0.00089250336,"about_ca_system_score_gemma":0.0012619855,"threshold_uncertainty_score":0.021608531},"labels":[],"label_agreement":null},{"id":"W4412945079","doi":"10.18653/v1/2025.acl-long.824","title":"Logical forms complement probability in understanding language model (and human) performance","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Complement (music); Computer science; Programming language; Natural language processing; Artificial intelligence; Theoretical computer science","score_opus":0.05872521618947497,"score_gpt":0.32412888227589837,"score_spread":0.2654036660864234,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412945079","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7825414,0.0017333238,0.19246176,0.002141104,0.00008063022,0.0002224748,0.0030068487,0.0016845327,0.016128052],"genre_scores_gemma":[0.9527607,0.0003281472,0.04353718,0.00019720092,0.000036873353,0.000082008635,0.0024612132,0.0001452728,0.00045132253],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99276805,0.0041945213,0.00044764712,0.0014624932,0.00093816506,0.00018919147],"domain_scores_gemma":[0.8805428,0.10010803,0.0065614977,0.009533357,0.0020942246,0.0011600788],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010002327,0.00090114964,0.00046799117,0.0022159384,0.0005593224,0.004957847,0.0009556044,0.0012653057,0.0044301003],"category_scores_gemma":[0.09491963,0.00056121737,0.00083675626,0.0013427001,0.0022239762,0.011110313,0.0025494169,0.0019357724,0.0010429174],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021539333,0.0011210653,0.4002649,0.0017219465,0.0010374382,0.00043732193,0.0133230295,0.087147005,0.017698942,0.052394718,0.008744162,0.41395554],"study_design_scores_gemma":[0.0001816137,0.0012351433,0.1720627,0.00033081646,0.00031618067,0.00088722433,0.004540271,0.5570829,0.012560716,0.23056962,0.019900672,0.00033220978],"about_ca_topic_score_codex":0.0047838367,"about_ca_topic_score_gemma":0.004578358,"teacher_disagreement_score":0.010002327,"about_ca_system_score_codex":0.00092868565,"about_ca_system_score_gemma":0.00087310886,"threshold_uncertainty_score":0.05289799},"labels":[],"label_agreement":null},{"id":"W4412945589","doi":"10.18653/v1/2025.acl-long.423","title":"LLMs can Perform Multi-Dimensional Analytic Writing Assessments: A Case Study of L2 Graduate-Level Academic English Writing","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Social Sciences and Humanities Research Council of Canada; Defense Advanced Research Projects Agency; Advanced Research Projects Agency; Yale University; Stony Brook University; Pennsylvania State University; University of Pennsylvania","keywords":"Academic writing; Computer science; Report writing; Graduate students; Mathematics education; Second language writing; Psychology; Linguistics; Second language; Pedagogy; Library science","score_opus":0.08607154982835231,"score_gpt":0.38865184903976413,"score_spread":0.30258029921141183,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412945589","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9377738,0.0006119854,0.04218173,0.0020872022,0.00016467654,0.0008291509,0.0015954965,0.005352081,0.0094038695],"genre_scores_gemma":[0.8596711,0.00025300047,0.13186415,0.0006891322,0.0001164572,0.0009190406,0.0014161195,0.00081913854,0.0042519495],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.97633874,0.0169449,0.0018708623,0.0013586212,0.00299666,0.00049023086],"domain_scores_gemma":[0.7577099,0.185689,0.009304581,0.01661852,0.026657604,0.0040204376],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0228152,0.0009624926,0.00068553514,0.0035090481,0.001955866,0.0023960115,0.0017228615,0.0017076763,0.0029191582],"category_scores_gemma":[0.16207013,0.0004467917,0.0005139053,0.002823014,0.0012068144,0.0031345838,0.004361001,0.0018125353,0.0025892796],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017417896,0.0026498616,0.07769526,0.0037732122,0.00022899453,0.007829518,0.18077222,0.007212378,0.046568733,0.005192181,0.050591093,0.6157448],"study_design_scores_gemma":[0.001306574,0.0061059664,0.17682514,0.002054869,0.00034496284,0.009665655,0.18868017,0.18641835,0.1193638,0.024231158,0.28366715,0.0013362096],"about_ca_topic_score_codex":0.002725317,"about_ca_topic_score_gemma":0.007057591,"teacher_disagreement_score":0.0228152,"about_ca_system_score_codex":0.0012847431,"about_ca_system_score_gemma":0.0018152611,"threshold_uncertainty_score":0.12065977},"labels":[],"label_agreement":null},{"id":"W4413019704","doi":"10.1016/j.ijar.2025.109545","title":"Selected papers from the Second International Joint Conference on Conceptual Knowledge Structures","year":2025,"lang":"en","type":"article","venue":"International Journal of Approximate Reasoning","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec en Outaouais","funders":"","keywords":"Joint (building); Computer science; Engineering; Structural engineering","score_opus":0.016790545037951063,"score_gpt":0.28954091363118745,"score_spread":0.27275036859323637,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413019704","genre_codex":"review","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010156903,0.27489537,0.15972033,0.069842964,0.2732922,0.00091187574,0.008681652,0.0018704142,0.20062825],"genre_scores_gemma":[0.035307266,0.19628678,0.07537472,0.0069224983,0.05333346,0.000703358,0.02834062,0.0022087116,0.6015226],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9979448,0.0003286818,0.0001941334,0.0004056461,0.00096088205,0.00016581659],"domain_scores_gemma":[0.99108326,0.0023558198,0.00022909402,0.0006191795,0.004666045,0.0010466092],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003715614,0.0013377201,0.0023669514,0.0064490307,0.0017078391,0.007708851,0.0018008403,0.001572123,0.08335132],"category_scores_gemma":[0.009257174,0.0005986178,0.0015794805,0.0098154275,0.0008295722,0.0053352895,0.0024332372,0.0028041159,0.02015348],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019250912,0.000111706926,0.00037821408,0.00087407016,0.00008258068,0.000114931834,0.00017131916,0.0007958029,0.0013819344,0.011061572,0.7135778,0.27125755],"study_design_scores_gemma":[0.00005412473,0.000050501345,0.00086808094,0.00052676885,0.00010141962,0.00015533777,0.00021272367,0.0016863606,0.0010683802,0.011271453,0.9839739,0.000030868116],"about_ca_topic_score_codex":0.006127597,"about_ca_topic_score_gemma":0.008877613,"teacher_disagreement_score":0.08335132,"about_ca_system_score_codex":0.0033476923,"about_ca_system_score_gemma":0.003958969,"threshold_uncertainty_score":0.27883792},"labels":[],"label_agreement":null},{"id":"W4413145645","doi":"10.1109/cvpr52734.2025.02813","title":"Exploring Simple Open-Vocabulary Semantic Segmentation","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Geomechanica (Canada)","funders":"","keywords":"Computer science; Simple (philosophy); Natural language processing; Vocabulary; Segmentation; Artificial intelligence; Linguistics","score_opus":0.0655176918182471,"score_gpt":0.33014782360309086,"score_spread":0.26463013178484374,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413145645","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07955872,0.0010517456,0.8887941,0.00095310266,0.00019814896,0.00018226974,0.0019700157,0.018557206,0.008734662],"genre_scores_gemma":[0.58959246,0.00060263975,0.38077468,0.001102708,0.00019648169,0.00031984076,0.01191636,0.0032869284,0.012207748],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9989722,0.000217251,0.000042309624,0.00050116004,0.00015895179,0.00010812789],"domain_scores_gemma":[0.9988158,0.0005334408,0.00009295869,0.0002985529,0.00016027899,0.0000989441],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012629138,0.0019258646,0.0011705668,0.0011238556,0.00071153475,0.0020873733,0.0029846828,0.0022516062,0.005923829],"category_scores_gemma":[0.003938889,0.0007843561,0.001343774,0.0010160243,0.0017288711,0.007333181,0.0024645082,0.0021038975,0.004657813],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010559926,0.00053442677,0.0051155123,0.0009311248,0.00027657297,0.0007251248,0.0011892697,0.20906188,0.07318144,0.07930188,0.041980717,0.58664614],"study_design_scores_gemma":[0.000041290645,0.0001212549,0.0004988684,0.000039613395,0.00003495613,0.0001536608,0.00021395284,0.90939856,0.014569766,0.06633485,0.008565287,0.000027890621],"about_ca_topic_score_codex":0.00717587,"about_ca_topic_score_gemma":0.012627535,"teacher_disagreement_score":0.00717587,"about_ca_system_score_codex":0.0014913873,"about_ca_system_score_gemma":0.0014358043,"threshold_uncertainty_score":0.019817233},"labels":[],"label_agreement":null},{"id":"W4413146477","doi":"10.1109/cvpr52734.2025.01381","title":"DeCLIP: Decoupled Learning for Open-Vocabulary Dense Perception","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"Shenzhen Science and Technology Innovation Program","keywords":"Computer science; Perception; Vocabulary learning; Vocabulary; Artificial intelligence; Psychology; Linguistics","score_opus":0.015283540374269954,"score_gpt":0.32194861875959147,"score_spread":0.30666507838532153,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413146477","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01735775,0.0004945175,0.96558213,0.0002826645,0.00012710287,0.00015564209,0.0007261671,0.012882816,0.002391116],"genre_scores_gemma":[0.4566648,0.0006050512,0.5222386,0.0012739543,0.00027097214,0.0006409117,0.007273776,0.0012945408,0.009737329],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9991742,0.00011704321,0.000029818319,0.00040581368,0.00015828697,0.000114818795],"domain_scores_gemma":[0.99852705,0.0006376217,0.00007363757,0.00042411152,0.0002034042,0.00013410352],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012100743,0.0018593876,0.001461478,0.0009312491,0.0006332063,0.001529744,0.004488312,0.0016663832,0.0073468694],"category_scores_gemma":[0.0040652696,0.00089918077,0.0012265783,0.0012367406,0.0011500814,0.005088547,0.0049610613,0.0040031327,0.0030333197],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00050176214,0.00055049534,0.0021559808,0.00037797703,0.00018423641,0.00027527538,0.0002650698,0.08446465,0.020189587,0.014066202,0.031835075,0.8451337],"study_design_scores_gemma":[0.000045726658,0.00011255694,0.0004909815,0.000020070314,0.000026373025,0.00008369414,0.00007763538,0.9643138,0.0055412804,0.025631962,0.0036314256,0.000024499046],"about_ca_topic_score_codex":0.0078415265,"about_ca_topic_score_gemma":0.012372144,"teacher_disagreement_score":0.0078415265,"about_ca_system_score_codex":0.0010510004,"about_ca_system_score_gemma":0.0016950696,"threshold_uncertainty_score":0.024577737},"labels":[],"label_agreement":null},{"id":"W4413157887","doi":"10.1109/cvpr52734.2025.02751","title":"VLsI: Verbalized Layers-to-Interactions from Large to Small Vision Language Models","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Computer science; Very-large-scale integration; Natural language processing; Artificial intelligence; Cognitive science; Programming language; Computer architecture; Psychology; Embedded system","score_opus":0.012489755155219814,"score_gpt":0.31690096221245456,"score_spread":0.30441120705723473,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413157887","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.028426345,0.00036519664,0.9253096,0.00057755446,0.00017739621,0.00018440484,0.0007532192,0.03574223,0.008464049],"genre_scores_gemma":[0.4181736,0.00023017895,0.56865305,0.00065022305,0.000053861877,0.0003899484,0.0021471435,0.0028409555,0.006861043],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99924743,0.0001819932,0.000057825375,0.0002368307,0.00019803953,0.00007792312],"domain_scores_gemma":[0.9985727,0.00063372066,0.000085782216,0.00043937488,0.00018786461,0.00008052225],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00096449105,0.0011416347,0.00048571275,0.0004059635,0.00041696263,0.0017465241,0.0033476667,0.0011066034,0.00830225],"category_scores_gemma":[0.005803698,0.00072784297,0.0010400484,0.00030009518,0.001013516,0.0042703873,0.0031758754,0.0027616932,0.0023423978],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00062721054,0.0003076386,0.0022608507,0.00063572085,0.00019493423,0.00034952365,0.0007366991,0.36439887,0.040068597,0.06830604,0.028468618,0.49364528],"study_design_scores_gemma":[0.000055483666,0.00011301216,0.00014217262,0.000033126733,0.000025428642,0.00005521604,0.00007902856,0.93966347,0.013855838,0.03547114,0.010476844,0.000029255381],"about_ca_topic_score_codex":0.005039484,"about_ca_topic_score_gemma":0.010452733,"teacher_disagreement_score":0.00830225,"about_ca_system_score_codex":0.0011148655,"about_ca_system_score_gemma":0.0020028995,"threshold_uncertainty_score":0.027773738},"labels":[],"label_agreement":null},{"id":"W4413178766","doi":"10.1109/gcwkshp64532.2024.11100791","title":"Enhancing Large Language Models for Telecom Networks Using Retrieval-Augmented Generation","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Regina","funders":"","keywords":"Computer science; Next-generation network; Telecommunications; Computer network; World Wide Web; The Internet","score_opus":0.02312514499307899,"score_gpt":0.3046900320327488,"score_spread":0.2815648870396698,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413178766","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07584215,0.0020029126,0.8877144,0.0009419001,0.00023151789,0.0003104573,0.0024191046,0.027467716,0.003069872],"genre_scores_gemma":[0.59971726,0.00058301043,0.3809533,0.0012239772,0.0002279049,0.0004796257,0.009900432,0.001123901,0.005790588],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.998838,0.00054851064,0.0000577228,0.00032585324,0.00013966288,0.00009024814],"domain_scores_gemma":[0.99748874,0.0016869599,0.0001090149,0.00041313868,0.00025060817,0.000051437077],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019451521,0.0015526497,0.0008834368,0.0010262429,0.00042768018,0.0011621314,0.0018116038,0.0012243878,0.0037147454],"category_scores_gemma":[0.006214144,0.0005281775,0.0013987755,0.0006763573,0.00048509132,0.002456954,0.0012098599,0.0020565612,0.0034414546],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007172768,0.0004753614,0.0036859745,0.00040694632,0.0002592382,0.00042024892,0.0003378106,0.39397788,0.027605254,0.0063671535,0.022044431,0.54370236],"study_design_scores_gemma":[0.000047369864,0.00007645331,0.00026906203,0.000009332983,0.000029044733,0.00007681581,0.000042253796,0.98621154,0.0049752914,0.005207884,0.003037348,0.000017614339],"about_ca_topic_score_codex":0.0070190434,"about_ca_topic_score_gemma":0.011224358,"teacher_disagreement_score":0.0070190434,"about_ca_system_score_codex":0.0010189246,"about_ca_system_score_gemma":0.001066706,"threshold_uncertainty_score":0.013956368},"labels":[],"label_agreement":null},{"id":"W4413232425","doi":"10.36834/cmej.82010","title":"Writing and artificial intelligence: apprivoiser one’s own paper","year":2025,"lang":"en","type":"article","venue":"Canadian Medical Education Journal","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Computer science; Artificial intelligence","score_opus":0.012225644574372009,"score_gpt":0.30441112026958556,"score_spread":0.2921854756952135,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413232425","genre_codex":"commentary","genre_gemma":"commentary","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"commentary","genre_consensus":"commentary","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0027561206,0.026010431,0.008293895,0.6521745,0.21764211,0.00005488853,0.00012075264,0.00039528392,0.09255205],"genre_scores_gemma":[0.06376544,0.031979952,0.015113172,0.16707282,0.20886257,0.00014596184,0.00028037545,0.002090806,0.5106889],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99378246,0.002364599,0.00036392565,0.00065388804,0.0024345391,0.0004005863],"domain_scores_gemma":[0.96660376,0.011655824,0.0010028734,0.0027754547,0.014285708,0.0036763859],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004783703,0.0008122706,0.00062296656,0.003902385,0.00571066,0.010332574,0.0018729807,0.004167532,0.017534077],"category_scores_gemma":[0.03982636,0.00032563167,0.0007861872,0.002121842,0.006803197,0.006928644,0.004161414,0.0069066207,0.009062014],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000009696618,0.000033137978,0.00015827682,0.00007157756,0.000009401127,0.00010455703,0.0014784753,0.000025879606,0.00012887943,0.014979417,0.9566225,0.026378175],"study_design_scores_gemma":[0.0000073559595,0.00001170213,0.00023226491,0.0002790219,0.000008379669,0.00054809323,0.0017280071,0.00015782472,0.00030456984,0.011149901,0.9855542,0.000018685192],"about_ca_topic_score_codex":0.005437372,"about_ca_topic_score_gemma":0.011572672,"teacher_disagreement_score":0.017534077,"about_ca_system_score_codex":0.002950358,"about_ca_system_score_gemma":0.0048275655,"threshold_uncertainty_score":0.05865735},"labels":[],"label_agreement":null},{"id":"W4413391566","doi":"10.18260/1-2--55492","title":"Automated Grading of Engineering Mechanics Assignments Using Large Language Models and Computer Vision: A Work in Progress","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Grading (engineering); Work (physics); Artificial intelligence; Natural language processing; Engineering drawing; Mechanical engineering; Engineering; Civil engineering","score_opus":0.010062536815914036,"score_gpt":0.2887079602246625,"score_spread":0.2786454234087485,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413391566","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.049115594,0.0006855421,0.9210765,0.0007412824,0.0002814663,0.00037323232,0.0019108402,0.022778114,0.0030375149],"genre_scores_gemma":[0.34264678,0.00046995885,0.63742113,0.0003268322,0.00019141415,0.00026848356,0.00917558,0.0013066593,0.0081931995],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9951184,0.0018351221,0.000298054,0.0009995209,0.0014378249,0.00031094998],"domain_scores_gemma":[0.989021,0.0034280058,0.00095036224,0.0022612736,0.0038394574,0.00049989455],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025289129,0.0017036005,0.0018996466,0.0035294641,0.00095652806,0.004249564,0.0037921597,0.0014699965,0.0075404723],"category_scores_gemma":[0.010101838,0.00064947223,0.0019508037,0.0018030684,0.0007292326,0.004647478,0.001753468,0.0025430662,0.005486092],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002921901,0.0010705574,0.003141188,0.00043702842,0.00012715462,0.00011511249,0.00015764276,0.03744296,0.027213844,0.0060570715,0.024295345,0.89964986],"study_design_scores_gemma":[0.00006886093,0.00015965666,0.0022254968,0.00005766444,0.000052308256,0.000106130145,0.00025015848,0.9425361,0.020425979,0.022266941,0.011783599,0.000067124536],"about_ca_topic_score_codex":0.007669572,"about_ca_topic_score_gemma":0.015278802,"teacher_disagreement_score":0.007669572,"about_ca_system_score_codex":0.0016858412,"about_ca_system_score_gemma":0.0032086587,"threshold_uncertainty_score":0.025225341},"labels":[],"label_agreement":null},{"id":"W4413423596","doi":"10.1016/b978-0-443-30046-2.00005-3","title":"Extending large language model capabilities beyond reasoning","year":2025,"lang":"en","type":"book-chapter","venue":"Elsevier eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Computer science; Programming language; Natural language processing; Cognitive science; Linguistics; Psychology; Philosophy","score_opus":0.008431370697338129,"score_gpt":0.2590292214334385,"score_spread":0.25059785073610036,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413423596","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0032442962,0.0016109118,0.89261013,0.0019842011,0.00029976247,0.00005330119,0.0005345699,0.0083599,0.091302976],"genre_scores_gemma":[0.13086969,0.0069606807,0.7334864,0.0013132105,0.0007010953,0.00022783958,0.0040776315,0.005084812,0.11727865],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992169,0.00018319506,0.000057632464,0.0001466728,0.00034466252,0.000050817947],"domain_scores_gemma":[0.9973942,0.0017274601,0.000043182445,0.0006453843,0.0001494122,0.000040290368],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009936644,0.0009045428,0.00060888147,0.0009857424,0.00057096075,0.003931581,0.001598999,0.00086186215,0.028020788],"category_scores_gemma":[0.0039815493,0.0009200375,0.0017211674,0.0011635914,0.0015457011,0.013380267,0.0029531815,0.0036802024,0.009306773],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000038990496,0.00006143187,0.00017727348,0.00040580938,0.00003848846,0.00021973535,0.00045398847,0.007973165,0.005721234,0.6408022,0.040238943,0.3038688],"study_design_scores_gemma":[0.000010701744,0.000012693276,0.00007817972,0.00012674072,0.000041467276,0.0002514809,0.00009388808,0.041697413,0.0048591853,0.7623845,0.19042227,0.000021513779],"about_ca_topic_score_codex":0.0016770773,"about_ca_topic_score_gemma":0.002022008,"teacher_disagreement_score":0.028020788,"about_ca_system_score_codex":0.00088746834,"about_ca_system_score_gemma":0.0007049415,"threshold_uncertainty_score":0.09373891},"labels":[],"label_agreement":null},{"id":"W4413423688","doi":"10.1016/b978-0-443-30046-2.00003-x","title":"Enabling retrieval-augmented generation and knowledge graphs with discourse analysis","year":2025,"lang":"en","type":"book-chapter","venue":"Elsevier eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Computer science; Knowledge graph; Information retrieval","score_opus":0.013230845914829266,"score_gpt":0.27108971357513295,"score_spread":0.2578588676603037,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413423688","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0047970423,0.00032150024,0.9699996,0.00028992546,0.00009228449,0.00009335353,0.00048127022,0.008244233,0.015680788],"genre_scores_gemma":[0.106253445,0.0006571852,0.86572397,0.00016077625,0.00007585376,0.00018046536,0.0025538246,0.0018482034,0.02254627],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99908817,0.00034751813,0.00006292594,0.00021212322,0.00024145773,0.000047940877],"domain_scores_gemma":[0.9981839,0.0012941224,0.000050728293,0.00027775808,0.00016148423,0.0000320692],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00086637674,0.0006931729,0.0004901273,0.0015172852,0.0005540253,0.0030014878,0.0012556931,0.00092446,0.016174689],"category_scores_gemma":[0.004173218,0.00049983704,0.00096317887,0.0014439959,0.0009544575,0.003440462,0.002370679,0.0010515016,0.0061474075],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001474945,0.00012593227,0.0003645662,0.0006702109,0.00005050425,0.00040342854,0.0012622536,0.018557834,0.024507342,0.17133605,0.02334341,0.75923103],"study_design_scores_gemma":[0.000059418515,0.00008935859,0.0006115984,0.00025500287,0.00012716488,0.0005232007,0.0006115463,0.37300503,0.067423485,0.3836784,0.1735386,0.000077235534],"about_ca_topic_score_codex":0.0022999549,"about_ca_topic_score_gemma":0.0025914342,"teacher_disagreement_score":0.016174689,"about_ca_system_score_codex":0.000618383,"about_ca_system_score_gemma":0.0007260477,"threshold_uncertainty_score":0.054109752},"labels":[],"label_agreement":null},{"id":"W4413459619","doi":"10.1109/icoct64433.2025.11118417","title":"Enhancing Interaction with Large Language Models: A Catalog of Prompt Engineering Techniques","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Hand and Upper Limb Clinic","funders":"","keywords":"Computer science; Data science; Software engineering; Programming language","score_opus":0.005215342687132969,"score_gpt":0.25430815348756364,"score_spread":0.24909281080043066,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413459619","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010172544,0.0049834372,0.9725864,0.000983903,0.000083639796,0.00026232295,0.00017464429,0.005462574,0.005290571],"genre_scores_gemma":[0.047618497,0.00516137,0.94225985,0.0002918156,0.000055361532,0.00035298092,0.0004740727,0.0010427209,0.002743253],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9933443,0.0027034786,0.00073880906,0.0008403103,0.0021736778,0.00019944076],"domain_scores_gemma":[0.9808014,0.012138597,0.0012930166,0.003450345,0.002070808,0.00024586378],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0068269903,0.0015764341,0.0008883067,0.0029516895,0.00079515314,0.002847228,0.0021780895,0.0014343344,0.0033297436],"category_scores_gemma":[0.023306856,0.0009010436,0.0013958745,0.002410372,0.0013493758,0.0061662295,0.0033372259,0.002669745,0.0017133541],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001399455,0.0002129275,0.0017732591,0.0027143895,0.000071624556,0.00023986063,0.0039040155,0.004293854,0.022396266,0.025530085,0.0058525424,0.9328714],"study_design_scores_gemma":[0.00021487707,0.0011730117,0.004151705,0.004125335,0.0005521816,0.0045070588,0.005546638,0.103429995,0.13462533,0.13690266,0.6042771,0.00049413124],"about_ca_topic_score_codex":0.00075704796,"about_ca_topic_score_gemma":0.0013564683,"teacher_disagreement_score":0.0068269903,"about_ca_system_score_codex":0.0008426516,"about_ca_system_score_gemma":0.001898133,"threshold_uncertainty_score":0.036105037},"labels":[],"label_agreement":null},{"id":"W4413491723","doi":"10.1007/978-3-031-97788-6_19","title":"Translation","year":2025,"lang":"en","type":"book-chapter","venue":"Health information technology standards","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"","keywords":"Translation (biology); Computer science; Biology; Genetics","score_opus":0.01253593387990665,"score_gpt":0.30618313849224876,"score_spread":0.2936472046123421,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413491723","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.000847097,0.00051274506,0.0054256315,0.003008681,0.007277038,0.00018249685,0.012197093,0.0018523671,0.9686968],"genre_scores_gemma":[0.0074986187,0.00096054655,0.0050310805,0.0011904103,0.00092042977,0.00015664632,0.014893846,0.0022992515,0.9670492],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.998963,0.00022747144,0.00008695261,0.00018677895,0.00043028555,0.00010550445],"domain_scores_gemma":[0.9980951,0.0003489975,0.000060603597,0.00046903235,0.000925846,0.00010054559],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.00087101874,0.0010111559,0.00056783255,0.0024460882,0.0014051803,0.004964277,0.00101311,0.0010073344,0.5962858],"category_scores_gemma":[0.004635646,0.00036162196,0.00047886476,0.002461221,0.0007913769,0.0027418572,0.0022123535,0.0017429092,0.49330175],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000055664623,0.00003252455,0.00006339348,0.00027049964,0.000003916658,0.000078389654,0.00029845277,0.000059585844,0.000807966,0.040730376,0.86722386,0.090375364],"study_design_scores_gemma":[0.000005506613,0.000006926405,0.00008476059,0.000042266496,0.0000018125111,0.000045611676,0.00006860261,0.000041994208,0.00037809587,0.0028401238,0.99648076,0.0000033788422],"about_ca_topic_score_codex":0.0032624742,"about_ca_topic_score_gemma":0.0036391995,"teacher_disagreement_score":0.5962858,"about_ca_system_score_codex":0.0015784146,"about_ca_system_score_gemma":0.0017786922,"threshold_uncertainty_score":0.5758493},"labels":[],"label_agreement":null},{"id":"W4413579816","doi":"10.1007/978-981-96-4317-2_35","title":"Design, Development, and Annotation of the Persian Spoken Learner Corpus","year":2025,"lang":"en","type":"book-chapter","venue":"Springer handbooks in languages and linguistics.","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Memorial University of Newfoundland; Cégep de l'Outaouais","funders":"","keywords":"Persian; Annotation; Natural language processing; Computer science; Artificial intelligence; Linguistics; Philosophy","score_opus":0.013986580375575534,"score_gpt":0.25966980510593585,"score_spread":0.2456832247303603,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413579816","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1002363,0.0018714343,0.60794955,0.0020922713,0.0012529718,0.012204014,0.1302233,0.045727976,0.098442115],"genre_scores_gemma":[0.10422319,0.00059592084,0.6216785,0.00056126824,0.00023076507,0.017741859,0.20712239,0.011885678,0.03596049],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.997273,0.00094655075,0.0002987489,0.0008033681,0.0005227594,0.000155501],"domain_scores_gemma":[0.9942866,0.0017701818,0.00019652655,0.0011143627,0.0021981413,0.00043422845],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004410136,0.0010732424,0.000944705,0.0048293504,0.002640403,0.0029367732,0.0025829093,0.0009807032,0.029449435],"category_scores_gemma":[0.0079291295,0.0011304057,0.0004442529,0.0032605026,0.0017048491,0.003902096,0.005788734,0.0026006745,0.024590028],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007731764,0.00045931278,0.006215888,0.0024383615,0.00006834045,0.0013357188,0.012924975,0.004340472,0.13512816,0.02563803,0.17263357,0.63804406],"study_design_scores_gemma":[0.00032474435,0.00056580355,0.01874802,0.00041833543,0.000117403695,0.0016754576,0.00892303,0.027379422,0.13042052,0.015238668,0.7959464,0.00024228328],"about_ca_topic_score_codex":0.010371455,"about_ca_topic_score_gemma":0.017546434,"teacher_disagreement_score":0.029449435,"about_ca_system_score_codex":0.0014366009,"about_ca_system_score_gemma":0.0049298382,"threshold_uncertainty_score":0.09851819},"labels":[],"label_agreement":null},{"id":"W4413612414","doi":"10.1007/978-3-031-88087-2_23","title":"Text Modification Research in L2 Korean: A Scoping Review","year":2025,"lang":"en","type":"book-chapter","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Linguistics; Psychology; History; Philosophy","score_opus":0.11858868567516893,"score_gpt":0.42236456936491723,"score_spread":0.3037758836897483,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413612414","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0010804175,0.99572855,0.00039084363,0.00051249814,0.0001454438,0.00004735859,0.00014141548,0.000012978271,0.0019404776],"genre_scores_gemma":[0.005331721,0.9910487,0.0015080314,0.000762503,0.00013307823,0.00012853785,0.000293552,0.000029009425,0.0007647927],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99832493,0.00043218394,0.0004740202,0.00031019544,0.0003756971,0.00008299095],"domain_scores_gemma":[0.9690887,0.02603405,0.0018201619,0.0005212579,0.0023076974,0.00022805214],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0065931506,0.000794863,0.0020502245,0.007895532,0.0007279481,0.003365125,0.0017735375,0.0015048455,0.0068603917],"category_scores_gemma":[0.018646875,0.0006359642,0.0016163936,0.009937484,0.0017167781,0.005533598,0.001911633,0.001370732,0.0013319446],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018322357,0.00014208983,0.0013969131,0.13018042,0.00036552767,0.0001921163,0.0012875614,0.00015046242,0.0014063843,0.0027082937,0.011166947,0.8508201],"study_design_scores_gemma":[0.000079588885,0.0004883212,0.015379625,0.39924207,0.004708893,0.0014539458,0.004771744,0.00031319028,0.0046427976,0.006236542,0.5625756,0.000107561216],"about_ca_topic_score_codex":0.005139304,"about_ca_topic_score_gemma":0.009610909,"teacher_disagreement_score":0.007895532,"about_ca_system_score_codex":0.0018896286,"about_ca_system_score_gemma":0.0071477825,"threshold_uncertainty_score":0.03486836},"labels":[],"label_agreement":null},{"id":"W4413614471","doi":"10.5430/wjel.v16n2p1","title":"Evaluating Three Neural Machine Translation Platforms for English-Arabic Translation: A Comparative Study of Linguistic Accuracy and Cultural Fidelity","year":2025,"lang":"en","type":"article","venue":"World Journal of English Language","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"King Faisal University","keywords":"Computer science; Fidelity; Arabic; Machine translation; Translation (biology); Natural language processing; Artificial intelligence; Linguistics; Philosophy; Chemistry","score_opus":0.061280691777733196,"score_gpt":0.38931185995646606,"score_spread":0.32803116817873285,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413614471","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9816098,0.00084078184,0.008037998,0.00034226273,0.00006717287,0.00018785041,0.00033363362,0.00035141647,0.008229148],"genre_scores_gemma":[0.98045105,0.0005080413,0.016263852,0.0000809665,0.00002450316,0.0001940495,0.0010841673,0.000084958156,0.0013083485],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.98828447,0.006394781,0.0016904923,0.00094997924,0.002431135,0.00024907975],"domain_scores_gemma":[0.9352312,0.0402918,0.0032985595,0.0045284205,0.015712665,0.00093722617],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01647607,0.0008195837,0.0008802037,0.0023770796,0.001438625,0.0036508557,0.001667587,0.0015430291,0.0017016578],"category_scores_gemma":[0.08134811,0.00034882684,0.0006931535,0.0027728134,0.0017474656,0.0039079813,0.0021301454,0.0011537563,0.001003961],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.010418407,0.0022076126,0.16167201,0.0036643546,0.0013027269,0.001564217,0.018820077,0.084998064,0.016576538,0.00824634,0.0046892064,0.6858405],"study_design_scores_gemma":[0.0011429844,0.010945346,0.25263575,0.0016341633,0.001977367,0.0022049712,0.02092909,0.6041518,0.06550356,0.013909104,0.024405578,0.00056021113],"about_ca_topic_score_codex":0.010448625,"about_ca_topic_score_gemma":0.008308471,"teacher_disagreement_score":0.01647607,"about_ca_system_score_codex":0.0029696487,"about_ca_system_score_gemma":0.0015717413,"threshold_uncertainty_score":0.08713478},"labels":[],"label_agreement":null},{"id":"W4413640367","doi":"10.1109/compsac65507.2025.00178","title":"Enhancing LLM-Based Code Generation with Complexity Metrics: A Feedback-Driven Approach","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Computer science; Code generation; Code (set theory); Programming language; Software engineering; Operating system","score_opus":0.03443670114263099,"score_gpt":0.28333948312434326,"score_spread":0.24890278198171228,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413640367","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15403482,0.001022585,0.76171035,0.001549592,0.0001805338,0.00087500905,0.0006957799,0.075652525,0.004278685],"genre_scores_gemma":[0.5263819,0.00019919412,0.4628181,0.000666645,0.00009402922,0.0006882308,0.0018064609,0.00506055,0.0022849725],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.98951924,0.0032177567,0.0007026591,0.0016445888,0.004401467,0.00051425415],"domain_scores_gemma":[0.94888794,0.027727075,0.0043409457,0.008326548,0.00970976,0.00100769],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006881955,0.0025694196,0.0012833232,0.0030461778,0.00065455126,0.0021881037,0.0032886628,0.0015808879,0.0025768029],"category_scores_gemma":[0.070202,0.0010501702,0.0009461566,0.0012130147,0.0012561238,0.0042587947,0.0033484767,0.002395983,0.0017892716],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00096186844,0.0012126806,0.021273604,0.00087112683,0.0001561758,0.0004151455,0.0013213822,0.13370307,0.07119501,0.0037760478,0.01388231,0.7512317],"study_design_scores_gemma":[0.00012603585,0.00037958488,0.0020636166,0.000067806955,0.000057205405,0.00016442625,0.00015228712,0.95519245,0.03293676,0.004900522,0.0038932743,0.00006598653],"about_ca_topic_score_codex":0.002903819,"about_ca_topic_score_gemma":0.0049262308,"teacher_disagreement_score":0.006881955,"about_ca_system_score_codex":0.0013677771,"about_ca_system_score_gemma":0.0034929847,"threshold_uncertainty_score":0.03639567},"labels":[],"label_agreement":null},{"id":"W4413761356","doi":"10.55092/sc20250021","title":"Fine-tuning large language models and evaluating retrieval methods for improved question answering on building codes","year":2025,"lang":"en","type":"article","venue":"Smart Construction","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Question answering; Computer science; Natural language processing; Information retrieval; Artificial intelligence","score_opus":0.021528780167426757,"score_gpt":0.38880374187797645,"score_spread":0.3672749617105497,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413761356","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19942111,0.0018828138,0.7647059,0.0009543224,0.0002507974,0.00042252836,0.0013958304,0.027464833,0.0035018893],"genre_scores_gemma":[0.58839196,0.00052056764,0.39963743,0.00045381082,0.00016764393,0.00034509588,0.0058287024,0.0015139382,0.003140738],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9961739,0.0017018405,0.000251442,0.00088693085,0.0007000083,0.00028588707],"domain_scores_gemma":[0.9822858,0.012757128,0.0004310786,0.0025779386,0.0016309142,0.00031713012],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0042552575,0.0014849367,0.0018648817,0.0020802785,0.0009324529,0.0022816947,0.0032197377,0.00248776,0.0039244746],"category_scores_gemma":[0.02284961,0.0006858439,0.0016529858,0.0017498079,0.0011358535,0.006786839,0.002153657,0.0023575155,0.0024744126],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017008613,0.0010491046,0.005350747,0.0010015153,0.0003250542,0.00021011487,0.00054129655,0.27783313,0.03549864,0.012813709,0.026243681,0.6374321],"study_design_scores_gemma":[0.000064616215,0.00010414749,0.0003306694,0.000014737948,0.000050045906,0.00003926538,0.000100344754,0.9829776,0.007850518,0.0073069516,0.0011458847,0.000015202388],"about_ca_topic_score_codex":0.011398521,"about_ca_topic_score_gemma":0.016189499,"teacher_disagreement_score":0.011398521,"about_ca_system_score_codex":0.0017664824,"about_ca_system_score_gemma":0.0020930625,"threshold_uncertainty_score":0.022664368},"labels":[],"label_agreement":null},{"id":"W4413775151","doi":"10.5121/csit.2025.151603","title":"Towards stable AI systems for Evaluating Arabic Pronunciations","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"York University; New York University Abu Dhabi","keywords":"Arabic; Computer science; Natural language processing; Artificial intelligence; Writing system; Linguistics; Philosophy","score_opus":0.03002963023902807,"score_gpt":0.360582316062258,"score_spread":0.33055268582322994,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413775151","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05305629,0.0012293486,0.91075075,0.0006988743,0.0005732985,0.00027367202,0.0025152853,0.023069968,0.007832663],"genre_scores_gemma":[0.46326914,0.00059708476,0.5170008,0.00045595877,0.00020655851,0.0006163644,0.008421835,0.0009371262,0.008495112],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9968214,0.0010489279,0.00024105747,0.0010227263,0.00067324325,0.00019256919],"domain_scores_gemma":[0.9946286,0.0019845604,0.00035108687,0.0006609585,0.0021845838,0.00019011885],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003623138,0.001746984,0.000773099,0.0022929104,0.00061702303,0.003219326,0.0019118786,0.0014257975,0.005098764],"category_scores_gemma":[0.017981604,0.0004478253,0.00054443016,0.0013381393,0.0007855821,0.0027009654,0.0021339434,0.0021918078,0.006078239],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005806431,0.00026448365,0.008762614,0.00042139998,0.0002679251,0.00016450346,0.0004681747,0.08994993,0.05052672,0.009268273,0.020777183,0.81854814],"study_design_scores_gemma":[0.00001854436,0.00013668324,0.002775536,0.00005332442,0.000036319212,0.0000734762,0.00022237399,0.9593145,0.02329073,0.008734111,0.0053037438,0.00004065542],"about_ca_topic_score_codex":0.006359845,"about_ca_topic_score_gemma":0.0069360924,"teacher_disagreement_score":0.006359845,"about_ca_system_score_codex":0.0011402218,"about_ca_system_score_gemma":0.0011813398,"threshold_uncertainty_score":0.019161224},"labels":[],"label_agreement":null},{"id":"W4413795518","doi":"10.11647/obp.0482.02","title":"Helpful Information for Reading the Latin Text","year":2025,"lang":"en","type":"book-chapter","venue":"Open Book Publishers","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Bishop's University","funders":"","keywords":"Reading (process); Latin Americans; Computer science; Information retrieval; Linguistics; Philosophy","score_opus":0.01701951597253395,"score_gpt":0.2660466089124783,"score_spread":0.24902709293994432,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413795518","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00070240657,0.0043241098,0.005415219,0.0035316665,0.004116877,0.00021622755,0.004726877,0.0020475457,0.974919],"genre_scores_gemma":[0.0038754488,0.0028610944,0.007452728,0.0015735194,0.0010617038,0.0001468734,0.0041887113,0.0029158082,0.9759241],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.999634,0.000079540354,0.000026952881,0.00005859766,0.00017433241,0.000026511618],"domain_scores_gemma":[0.99919695,0.00024824782,0.0000355692,0.00006959623,0.00037030107,0.00007929204],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.00040212588,0.0011005223,0.00079227594,0.0023369428,0.0015750965,0.002275927,0.0007604785,0.0006798698,0.6735705],"category_scores_gemma":[0.002440158,0.00032372924,0.00037130312,0.0026859818,0.0005798198,0.0034597719,0.0015352257,0.0015644022,0.4808798],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000020254402,0.000021172007,0.00008753512,0.00025992354,0.0000014357147,0.00010632283,0.00043345615,0.000041473224,0.00051136216,0.014887898,0.9013423,0.08228679],"study_design_scores_gemma":[0.0000019150605,0.000003084054,0.000085658816,0.00007306245,5.942632e-7,0.00007987022,0.00010692832,0.000015506866,0.00006064604,0.00106219,0.9985084,0.0000021259111],"about_ca_topic_score_codex":0.0016233943,"about_ca_topic_score_gemma":0.0046414626,"teacher_disagreement_score":0.6735705,"about_ca_system_score_codex":0.0009769845,"about_ca_system_score_gemma":0.0009411963,"threshold_uncertainty_score":0.465612},"labels":[],"label_agreement":null},{"id":"W4413842841","doi":"10.1101/2025.08.25.672171","title":"Song familiarity relies on evidence accumulation","year":2025,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"MacEwan University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Business","score_opus":0.03777816707369322,"score_gpt":0.2950092126521874,"score_spread":0.2572310455784942,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413842841","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9431712,0.00035156147,0.047905616,0.00019794982,0.00004205132,0.000047612553,0.00018828058,0.00040937238,0.0076863985],"genre_scores_gemma":[0.99489224,0.000050870458,0.00456226,0.000021361828,0.00002044081,0.0000089107025,0.0000600943,0.000028842387,0.0003551425],"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990357,0.00016890647,0.00008471042,0.00032111618,0.00028647465,0.00010323272],"domain_scores_gemma":[0.9893372,0.005762711,0.0022371004,0.0009034597,0.0011768499,0.000582703],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010235923,0.0002979349,0.00040386405,0.00078716845,0.00027520276,0.0015115653,0.00038270905,0.0004513332,0.0043476443],"category_scores_gemma":[0.015058965,0.00039161855,0.0002242382,0.00040480224,0.0007977688,0.0023722134,0.0010427153,0.0007724741,0.00062345096],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016971987,0.00018382928,0.08410493,0.00041665646,0.0001828695,0.0010580682,0.0010183227,0.0038779194,0.7530474,0.0056372727,0.00082997343,0.14794555],"study_design_scores_gemma":[0.00007419475,0.0013800908,0.6635565,0.00014171457,0.00020183716,0.0034025481,0.00082193146,0.09349872,0.20864888,0.025590597,0.0024557677,0.00022706843],"about_ca_topic_score_codex":0.0006791911,"about_ca_topic_score_gemma":0.000725448,"teacher_disagreement_score":0.0043476443,"about_ca_system_score_codex":0.00027353122,"about_ca_system_score_gemma":0.00022697324,"threshold_uncertainty_score":0.014544368},"labels":[],"label_agreement":null},{"id":"W4413874636","doi":"10.1038/s41592-025-02810-3","title":"Translon: a single term for translated regions","year":2025,"lang":"en","type":"letter","venue":"Nature Methods","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; McGill University Health Centre; Centre Hospitalier Universitaire de Sherbrooke; Université de Sherbrooke","funders":"Mikrobiologický Ústav, Akademie Věd České Republiky; Institute of Molecular and Cell Biology; Wenzhou Institute of Biomaterials and Engineering; Directorate for Biological Sciences; University of Chinese Academy of Sciences; State Key Laboratory of Ophthalmology; Tartu Ülikool; Centre National de la Recherche Scientifique; Universiteit Gent; University of Oxford; Chinese Academy of Sciences; York University; Akademie Věd České Republiky; Weizmann Institute of Science; National Institute of Allergy and Infectious Diseases; Sun Yat-sen University; NHLBI Division of Intramural Research; McGill University; European Cooperation in Science and Technology; Massachusetts Institute of Technology; Division of Intramural Research, National Institute of Allergy and Infectious Diseases","keywords":"Term (time); Computational biology; Computer science; Biology; Physics; Astronomy","score_opus":0.03239577879587022,"score_gpt":0.3925208336633362,"score_spread":0.36012505486746593,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413874636","genre_codex":"methods","genre_gemma":"commentary","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"commentary","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.003495076,0.0050751916,0.5083219,0.2007507,0.11809079,0.0002302505,0.0036486238,0.018927773,0.14145973],"genre_scores_gemma":[0.07018407,0.003174461,0.36145374,0.112692066,0.050063767,0.000611688,0.005027637,0.016708931,0.3800837],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99820733,0.0004986973,0.00016940932,0.00028760004,0.0006696969,0.00016723601],"domain_scores_gemma":[0.99310726,0.0029663777,0.0003023364,0.0013780125,0.001902272,0.00034370925],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020290483,0.00094426004,0.0009030974,0.0010258905,0.0017062529,0.0026616137,0.0014037791,0.004204924,0.044698462],"category_scores_gemma":[0.010491937,0.0004262556,0.00071577623,0.0008870748,0.0021683923,0.0043305787,0.0019418678,0.0058876015,0.03633223],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019499836,0.000015255152,0.00008558357,0.00023220913,0.000011586377,0.00023329035,0.00013901776,0.00024558543,0.0053695478,0.074381374,0.82725704,0.09183455],"study_design_scores_gemma":[0.000030413958,0.000027990422,0.000113710834,0.00009751717,0.000015248651,0.0006609896,0.000058656988,0.0047298213,0.0066773025,0.034855712,0.95270383,0.000028889648],"about_ca_topic_score_codex":0.0015326221,"about_ca_topic_score_gemma":0.0038146207,"teacher_disagreement_score":0.044698462,"about_ca_system_score_codex":0.0014233217,"about_ca_system_score_gemma":0.0011880429,"threshold_uncertainty_score":0.14953119},"labels":[],"label_agreement":null},{"id":"W4413942424","doi":"10.1007/978-3-032-04354-2_18","title":"Overview of the CLEF 2025 JOKER Lab: Humour in Machine","year":2025,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba","funders":"","keywords":"Clef; Computer science; Artificial intelligence; Information retrieval; World Wide Web; Management","score_opus":0.01832890902460905,"score_gpt":0.28679542091333915,"score_spread":0.2684665118887301,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413942424","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.003940809,0.15675037,0.16422911,0.023812652,0.0077710925,0.0010097658,0.019070104,0.034228977,0.5891871],"genre_scores_gemma":[0.043671396,0.08440758,0.24085742,0.011859316,0.012076104,0.0015980585,0.07712512,0.013708529,0.5146965],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99788564,0.00067627645,0.0000927944,0.00029667062,0.0008809008,0.00016769739],"domain_scores_gemma":[0.997332,0.0007788074,0.00008073867,0.0003489286,0.00089458795,0.0005649892],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026184383,0.0013603236,0.0012083896,0.00579311,0.0014961703,0.004791034,0.0017000507,0.0017631878,0.108186886],"category_scores_gemma":[0.004314905,0.0007591987,0.00078069844,0.0055865305,0.0007617057,0.00500959,0.0031224503,0.0024609342,0.086956024],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000042087693,0.000052774496,0.0001254696,0.00044951637,0.000011051296,0.00003622931,0.000056825953,0.00051805325,0.0010199347,0.01141241,0.66427034,0.3220053],"study_design_scores_gemma":[0.000013767199,0.000024850322,0.00050471135,0.00017408276,0.000005590254,0.00016500996,0.000030085243,0.0011642704,0.0011003717,0.0064754616,0.99031633,0.000025432792],"about_ca_topic_score_codex":0.0039989618,"about_ca_topic_score_gemma":0.008502805,"teacher_disagreement_score":0.108186886,"about_ca_system_score_codex":0.002410443,"about_ca_system_score_gemma":0.0022527205,"threshold_uncertainty_score":0.3619212},"labels":[],"label_agreement":null},{"id":"W4414087786","doi":"10.7152/nasko.v7i1.95650","title":"Cosine Similarity Indexing of Word Embeddings Using Knowledge Organization Systems","year":2025,"lang":"en","type":"article","venue":"NASKO","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Search engine indexing; Vector space model; Cosine similarity; Word (group theory); Similarity (geometry); Latent semantic analysis; Knowledge organization; Context (archaeology); Probabilistic latent semantic analysis","score_opus":0.014119261269350725,"score_gpt":0.30395903105734384,"score_spread":0.28983976978799314,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414087786","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04368873,0.0018563017,0.946301,0.0003828095,0.00025793823,0.00024330575,0.0014426736,0.0014138734,0.0044133984],"genre_scores_gemma":[0.3628636,0.0011525669,0.6268894,0.00014263246,0.0003158479,0.0004851051,0.0047296085,0.0002781384,0.0031431736],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9960246,0.0010084165,0.0006551981,0.0008940251,0.0012277105,0.00019005878],"domain_scores_gemma":[0.9956909,0.0017372649,0.00060064497,0.0010359575,0.0007806822,0.00015461407],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021096887,0.0007105735,0.0013973789,0.008495568,0.0009829337,0.0034769499,0.0014778045,0.0009764178,0.0028878327],"category_scores_gemma":[0.015018634,0.00038707134,0.0011945473,0.0111423265,0.001195433,0.00777178,0.0027768803,0.0013289371,0.001338219],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00043360365,0.00034924998,0.005480519,0.00080254214,0.00025183737,0.00022403043,0.0012538746,0.030844446,0.010873379,0.162407,0.010520865,0.77655876],"study_design_scores_gemma":[0.00010414363,0.00034838953,0.0051443526,0.00017343595,0.000113995215,0.0005177284,0.00091451086,0.622859,0.010402894,0.33006436,0.029190583,0.00016659731],"about_ca_topic_score_codex":0.0032330174,"about_ca_topic_score_gemma":0.002713016,"teacher_disagreement_score":0.008495568,"about_ca_system_score_codex":0.0013898542,"about_ca_system_score_gemma":0.0012987822,"threshold_uncertainty_score":0.011157274},"labels":[],"label_agreement":null},{"id":"W4414230282","doi":"10.1101/2025.09.11.675507","title":"Dual-LLM Adversarial Framework for Information Extraction from Research Literature","year":2025,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Prompt (Canada)","funders":"","keywords":"Adversarial system; Consistency (knowledge bases); Information extraction; Task (project management); Natural language; Process (computing); Data extraction; Task analysis","score_opus":0.021104434407886206,"score_gpt":0.30974943557203544,"score_spread":0.2886450011641492,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414230282","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02237809,0.0011078859,0.9696604,0.0015638482,0.0001050817,0.00013241125,0.0007345718,0.0022910305,0.0020266282],"genre_scores_gemma":[0.5756463,0.00068587385,0.4075729,0.0015940528,0.00028346674,0.0005067602,0.0034595092,0.00042174556,0.009829412],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9977863,0.0010515124,0.000112114474,0.0004277826,0.00047526596,0.00014690639],"domain_scores_gemma":[0.9922031,0.005783496,0.00063638936,0.0006426713,0.0005455958,0.00018871992],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0057901805,0.001248676,0.0010668976,0.0020849898,0.0006006781,0.0012608586,0.0019827576,0.0017621801,0.003004323],"category_scores_gemma":[0.011195322,0.0005388953,0.0010110538,0.001235472,0.0014089979,0.001870986,0.0027524868,0.002138312,0.001048243],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00032096423,0.00012917619,0.0020867137,0.00031665846,0.00018072051,0.00033699037,0.00022154402,0.8206501,0.0063071884,0.019805953,0.010225668,0.13941833],"study_design_scores_gemma":[0.000009015905,0.000022569024,0.0001306572,0.000018920318,0.000011575027,0.000031873413,0.000009916483,0.9877528,0.0014247892,0.009526668,0.0010522085,0.000009029232],"about_ca_topic_score_codex":0.002592466,"about_ca_topic_score_gemma":0.0038464153,"teacher_disagreement_score":0.0057901805,"about_ca_system_score_codex":0.001488138,"about_ca_system_score_gemma":0.0017229846,"threshold_uncertainty_score":0.030621707},"labels":[],"label_agreement":null},{"id":"W4414231430","doi":"10.1109/amlds63918.2025.11159378","title":"Error Analysis for POS Tagging of Hindi-English Code-Mixed Data","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Sentence; Noun; Task (project management); Hindi; Spelling; Word (group theory); Error detection and correction; Error analysis","score_opus":0.03519192064712283,"score_gpt":0.33931331869499026,"score_spread":0.3041213980478674,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414231430","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.86222017,0.0007015524,0.12478476,0.0004272571,0.00053685386,0.0002729374,0.0061740116,0.0025196623,0.0023628187],"genre_scores_gemma":[0.91162527,0.0001148873,0.07242899,0.00021415664,0.000053482156,0.00021895084,0.013178335,0.0004604683,0.0017054763],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9889065,0.0043525184,0.0017330036,0.0021337736,0.0025114354,0.00036272247],"domain_scores_gemma":[0.89240944,0.07524398,0.007243545,0.010049045,0.01430836,0.0007455573],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011242187,0.0006581835,0.00050608424,0.0031937398,0.0012445836,0.0013519003,0.0008952479,0.0009825957,0.0010257199],"category_scores_gemma":[0.053683992,0.00026905455,0.0006562249,0.0032771106,0.0011339692,0.0011920314,0.0013153396,0.0011137192,0.00088276365],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0055895755,0.0006972827,0.4554346,0.001900762,0.000845881,0.005753664,0.008714879,0.047550943,0.059611037,0.008929946,0.016072523,0.38889894],"study_design_scores_gemma":[0.00014698488,0.0006976384,0.27822685,0.0003308441,0.0004099492,0.004511384,0.003451649,0.499773,0.16482024,0.015663316,0.031673048,0.00029505097],"about_ca_topic_score_codex":0.0051918505,"about_ca_topic_score_gemma":0.005588025,"teacher_disagreement_score":0.011242187,"about_ca_system_score_codex":0.0011875718,"about_ca_system_score_gemma":0.0009390554,"threshold_uncertainty_score":0.059455037},"labels":[],"label_agreement":null},{"id":"W4414249370","doi":"10.1038/s43588-025-00861-2","title":"On the compatibility of generative AI and generative linguistics","year":2025,"lang":"en","type":"review","venue":"Nature Computational Science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Mila - Quebec Artificial Intelligence Institute; HEC Montréal","funders":"","keywords":"Generative grammar; Cognitive linguistics; Applied linguistics; Computational linguistics; Language and Communication Technologies; Theoretical linguistics; Quantitative linguistics; Compatibility (geochemistry)","score_opus":0.02464684558482763,"score_gpt":0.3833029036527605,"score_spread":0.3586560580679329,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414249370","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.000077678764,0.9927879,0.0010821909,0.0016000145,0.00029968328,0.0000025808063,0.000013802374,0.000008641771,0.004127586],"genre_scores_gemma":[0.0024386926,0.99292976,0.0012460555,0.0013388726,0.00077979575,0.000011884643,0.000041425264,0.0000094595,0.0012041951],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9994784,0.00018014832,0.00004742862,0.00008933438,0.00017287675,0.00003187616],"domain_scores_gemma":[0.99599886,0.003218364,0.00012740133,0.00011716677,0.0004571041,0.00008115487],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019264641,0.0007417582,0.0013344187,0.0030892247,0.00047733827,0.0019573243,0.001447641,0.0018199007,0.0055264547],"category_scores_gemma":[0.0045004687,0.00036822318,0.0003983712,0.0039004304,0.0021528536,0.0045079286,0.0013706478,0.0028451153,0.0027732619],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000054571872,0.000045464243,0.00018861143,0.009121332,0.00010909884,0.00015141023,0.00014155392,0.0006055662,0.0004936069,0.17549649,0.050183743,0.7634086],"study_design_scores_gemma":[0.000014853806,0.000030577543,0.00039860694,0.0050762272,0.00008816446,0.0004352864,0.000061668274,0.0002473433,0.00025983178,0.06713356,0.9262328,0.000021064758],"about_ca_topic_score_codex":0.0023083251,"about_ca_topic_score_gemma":0.003288619,"teacher_disagreement_score":0.0055264547,"about_ca_system_score_codex":0.0014930989,"about_ca_system_score_gemma":0.0023399699,"threshold_uncertainty_score":0.018487811},"labels":[],"label_agreement":null},{"id":"W4414253714","doi":"10.1007/978-3-032-04624-6_33","title":"DocAnnot - Accelerating the Creation of Key Information Extraction Datasets with GenAI-Powered Auto-annotation","year":2025,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Glycemic Index Laboratories","funders":"","keywords":"Key (lock); Benchmark (surveying); Task (project management); Annotation; Process (computing); Minimum bounding box; Matching (statistics); Automation","score_opus":0.012205942969146325,"score_gpt":0.2750500807634131,"score_spread":0.26284413779426674,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414253714","genre_codex":"software","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011059771,0.0014773582,0.28986126,0.00090937666,0.0017254526,0.00071864174,0.12405581,0.55343515,0.016757134],"genre_scores_gemma":[0.028561838,0.00061087514,0.5706949,0.0008767544,0.00020840239,0.00094148976,0.34720778,0.031455845,0.019442081],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99716055,0.00041036028,0.00025893838,0.0009891117,0.0009930744,0.00018799743],"domain_scores_gemma":[0.99542636,0.0013044567,0.0001763371,0.001851067,0.0010681761,0.0001736516],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027774137,0.002887716,0.0017316988,0.004260186,0.0022491864,0.0037341267,0.0028689287,0.001367486,0.03505001],"category_scores_gemma":[0.006971816,0.001639685,0.0026823347,0.005648568,0.0008710842,0.0042778123,0.0047566597,0.002590234,0.047864333],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009430797,0.0002245981,0.0019140725,0.0013485702,0.00032648272,0.00027749586,0.0004998769,0.0013821848,0.022767652,0.0063009174,0.66792065,0.2960945],"study_design_scores_gemma":[0.0004544061,0.00027227352,0.0046005463,0.00031025862,0.00026183142,0.0007205664,0.0005494071,0.062520154,0.114643246,0.026417844,0.7889892,0.00026034986],"about_ca_topic_score_codex":0.007138337,"about_ca_topic_score_gemma":0.016204758,"teacher_disagreement_score":0.03505001,"about_ca_system_score_codex":0.0014244058,"about_ca_system_score_gemma":0.0028825814,"threshold_uncertainty_score":0.11725396},"labels":[],"label_agreement":null},{"id":"W4414266922","doi":"10.14778/3750601.3750685","title":"Beyond Quacking: Deep Integration of Language Models and RAG into DuckDB","year":2025,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; Polytechnique Montréal","funders":"","keywords":"Schema (genetic algorithms); Data integration; Data modeling; Relational database; Context (archaeology); SQL; Language model; Context model; Rapid prototyping","score_opus":0.008653647840289753,"score_gpt":0.2656501601318978,"score_spread":0.256996512291608,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414266922","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012242266,0.0016308263,0.86595446,0.0017149124,0.00028729418,0.00027268205,0.005726524,0.1044629,0.0077081774],"genre_scores_gemma":[0.11741411,0.00089793996,0.8488277,0.0011892325,0.00008124131,0.00030233944,0.017661657,0.008379709,0.005245999],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99700254,0.00084503647,0.0003512854,0.0006520029,0.000999023,0.00015006254],"domain_scores_gemma":[0.9959798,0.0014994112,0.00012862557,0.0017060064,0.0005069432,0.00017920196],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0050998665,0.00091859174,0.0012209076,0.0020315826,0.0009506996,0.0055923136,0.0038577374,0.0010955678,0.0058353585],"category_scores_gemma":[0.014781449,0.0012374253,0.0015436639,0.0018593735,0.0013597911,0.007818766,0.005409986,0.0028982467,0.0035204447],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00061044964,0.00044937877,0.007800912,0.0012094226,0.00046225687,0.00047319685,0.0015381898,0.13683805,0.005464096,0.17954634,0.10210148,0.5635063],"study_design_scores_gemma":[0.00008479683,0.000059870406,0.00064593536,0.00018117872,0.000064971915,0.00019672618,0.0002470612,0.69953233,0.007195883,0.14579868,0.14587241,0.00012014109],"about_ca_topic_score_codex":0.017485162,"about_ca_topic_score_gemma":0.029461414,"teacher_disagreement_score":0.017485162,"about_ca_system_score_codex":0.0019914764,"about_ca_system_score_gemma":0.0031319982,"threshold_uncertainty_score":0.034766793},"labels":[],"label_agreement":null},{"id":"W4414270172","doi":"10.1101/2025.09.12.675926","title":"Medical Abbreviation Disambiguation with Large Language Models: Zero- and Few-Shot Evaluation on the MeDAL Dataset","year":2025,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Brock University","funders":"","keywords":"Interpretability; Readability; Task (project management); Unified Medical Language System; Language model; Resource (disambiguation); Named entity; Information extraction","score_opus":0.025252830765958086,"score_gpt":0.2839559906917193,"score_spread":0.2587031599257612,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414270172","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7870056,0.017285502,0.11494147,0.0044584284,0.0020657293,0.00089797063,0.017610159,0.04725093,0.008484268],"genre_scores_gemma":[0.75705,0.0020200755,0.16932614,0.002445217,0.00046697838,0.00044579108,0.06140898,0.0014763818,0.0053605167],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99434125,0.0026712096,0.0005120526,0.0014747649,0.0007518572,0.00024876063],"domain_scores_gemma":[0.9829214,0.013126743,0.00036857632,0.0016603137,0.0011889555,0.00073397526],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010077278,0.002299072,0.0019134134,0.0026026927,0.0014433312,0.002171242,0.0033134823,0.0034837022,0.0025580577],"category_scores_gemma":[0.024115318,0.00052598957,0.0018047497,0.0015383074,0.0016599464,0.0043149954,0.0030075184,0.0029458012,0.0017202195],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.006753265,0.004149744,0.011931068,0.004714977,0.0023394346,0.0026137996,0.0020463795,0.22165956,0.029619504,0.00514475,0.112950146,0.5960774],"study_design_scores_gemma":[0.0006542421,0.0013187097,0.005165262,0.00021940297,0.00046146536,0.0011811298,0.0014452818,0.9413854,0.02516992,0.008166536,0.01462151,0.00021116788],"about_ca_topic_score_codex":0.014423233,"about_ca_topic_score_gemma":0.022116523,"teacher_disagreement_score":0.014423233,"about_ca_system_score_codex":0.0018550484,"about_ca_system_score_gemma":0.0028412126,"threshold_uncertainty_score":0.05329442},"labels":[],"label_agreement":null},{"id":"W4414360309","doi":"10.24963/ijcai.2025/643","title":"DUQ: Dual Uncertainty Quantification for Text-Video Retrieval","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Similarity (geometry); Granularity; Construct (python library); Benchmark (surveying); Dual (grammatical number); Feature (linguistics); Uncertainty reduction theory; Video retrieval; Trustworthiness","score_opus":0.018763589555484135,"score_gpt":0.31421871010135777,"score_spread":0.29545512054587364,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414360309","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006019543,0.0006562141,0.99132746,0.00017983395,0.00002871225,0.00007168399,0.00015683901,0.0006251528,0.00093455677],"genre_scores_gemma":[0.6072689,0.00091919216,0.38538474,0.00049620046,0.00024458722,0.00040293403,0.0010210805,0.00026228695,0.0039999937],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9973015,0.00059820054,0.00018768721,0.0005411568,0.0011879415,0.00018352189],"domain_scores_gemma":[0.9969541,0.0016134081,0.00037877454,0.00040484942,0.00053956424,0.00010935309],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028075287,0.0011203181,0.0013548855,0.0020037757,0.0005519607,0.0019057083,0.0025816984,0.0012941394,0.0022531569],"category_scores_gemma":[0.011584442,0.0005089413,0.0009366892,0.0018201008,0.0010804604,0.0050360537,0.0033992478,0.0016853783,0.00065960013],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004408348,0.00019182805,0.0020127916,0.00042019426,0.00015225468,0.00020400155,0.00035111842,0.42702413,0.018623767,0.06566545,0.008157915,0.47675574],"study_design_scores_gemma":[0.000010451506,0.000046728182,0.00021711392,0.000010820274,0.000016330043,0.000052409803,0.000020227777,0.97871757,0.0028918597,0.016275542,0.0017199655,0.00002094824],"about_ca_topic_score_codex":0.0052045295,"about_ca_topic_score_gemma":0.0035723993,"teacher_disagreement_score":0.0052045295,"about_ca_system_score_codex":0.0021246944,"about_ca_system_score_gemma":0.001320347,"threshold_uncertainty_score":0.015415847},"labels":[],"label_agreement":null},{"id":"W4414447584","doi":"10.1109/iccv51701.2025.02118","title":"Plug-in Feedback Self-Adaptive Attention in CLIP for Training-Free Open-Vocabulary Segmentation","year":2025,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University; McMaster University; University of Toronto","funders":"","keywords":"Segmentation; Coherence (philosophical gambling strategy); Semantics (computer science); Consistency (knowledge bases); Perspective (graphical); Pattern recognition (psychology)","score_opus":0.04225797506368045,"score_gpt":0.33173249115709075,"score_spread":0.2894745160934103,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414447584","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.035794094,0.00087207026,0.8708769,0.00032618394,0.00030599657,0.00023752125,0.00090913975,0.08524643,0.0054316083],"genre_scores_gemma":[0.46132764,0.00037494354,0.5105873,0.001134247,0.0002337963,0.00048410002,0.006164789,0.006751675,0.0129416175],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993461,0.00009961477,0.000023688624,0.0002738457,0.00016107189,0.00009564049],"domain_scores_gemma":[0.9991672,0.00035057857,0.000030617062,0.00023613423,0.00014314044,0.00007238028],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008814768,0.0016012805,0.0010431852,0.00066644565,0.0005593128,0.0011362609,0.003268145,0.0016034356,0.009000064],"category_scores_gemma":[0.0034111363,0.00065399345,0.00077984715,0.00068656064,0.0008313836,0.002278216,0.0026122052,0.0025048782,0.0033918056],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000775346,0.0003570684,0.001745309,0.0004680615,0.00020960701,0.00040269864,0.0003764493,0.13092543,0.055444453,0.005563642,0.051610015,0.75212187],"study_design_scores_gemma":[0.00006663888,0.00008501147,0.0005199088,0.000015863076,0.00003822941,0.00010092358,0.00007002208,0.9618825,0.025370322,0.004724382,0.007103018,0.000023251016],"about_ca_topic_score_codex":0.008547417,"about_ca_topic_score_gemma":0.01743313,"teacher_disagreement_score":0.009000064,"about_ca_system_score_codex":0.0009062341,"about_ca_system_score_gemma":0.0011825288,"threshold_uncertainty_score":0.030108213},"labels":[],"label_agreement":null},{"id":"W4414465542","doi":"10.1037/xhp0001373","title":"Contributions of action representations to tool naming.","year":2025,"lang":"en","type":"article","venue":"Journal of Experimental Psychology Human Perception & Performance","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Action (physics); Object (grammar); Task (project management); Orientation (vector space); Dimension (graph theory); Feature (linguistics); Laterality; Sequence (biology); Semantics (computer science)","score_opus":0.0290836165024124,"score_gpt":0.43343067248555267,"score_spread":0.40434705598314025,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414465542","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9296331,0.0015479437,0.056732807,0.0003215441,0.00012827208,0.00007882344,0.00050751225,0.00047999958,0.010570108],"genre_scores_gemma":[0.98140866,0.0002905414,0.016579838,0.000075227756,0.000035052515,0.000055017066,0.0003749719,0.00009962466,0.0010810221],"study_design_codex":"bench_or_experimental","study_design_gemma":"observational","domain_scores_codex":[0.9977175,0.0006305449,0.0001941727,0.0007482545,0.00055349636,0.00015595609],"domain_scores_gemma":[0.95923984,0.024087206,0.0073802373,0.0068798424,0.0014447425,0.0009680769],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002899231,0.00071627705,0.00054811576,0.0008301724,0.0003190244,0.0021725069,0.0009410484,0.0011140179,0.005378455],"category_scores_gemma":[0.036366086,0.0005686084,0.00069089496,0.0005272379,0.0012042973,0.0033931856,0.0016851462,0.0011340312,0.00060881994],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0032677408,0.00039766487,0.04327421,0.0017544241,0.00043353578,0.0012073284,0.0030237995,0.006727669,0.5820989,0.015820144,0.0009884384,0.34100616],"study_design_scores_gemma":[0.0004899329,0.004161962,0.54975015,0.0003273769,0.0011306898,0.009308006,0.0012951188,0.09689455,0.21280001,0.10729054,0.01607312,0.0004785884],"about_ca_topic_score_codex":0.00092344626,"about_ca_topic_score_gemma":0.0006463983,"teacher_disagreement_score":0.005378455,"about_ca_system_score_codex":0.0004992587,"about_ca_system_score_gemma":0.00041881844,"threshold_uncertainty_score":0.017992735},"labels":[],"label_agreement":null},{"id":"W4414529254","doi":"10.5539/ijel.v15n5p13","title":"The Case for Subject-Verb Dependency Distance as a Measure of Complexity and Readability","year":2025,"lang":"en","type":"article","venue":"International Journal of English Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"University of Leicester","keywords":"Readability; Dependency (UML); Sentence; Noun; Subject (documents); Noun phrase; Measure (data warehouse); Context (archaeology)","score_opus":0.018362209025360236,"score_gpt":0.32041718879630476,"score_spread":0.30205497977094453,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414529254","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5880647,0.0073717274,0.24437898,0.01499877,0.0008874672,0.00044570616,0.0011718238,0.00044089806,0.1422399],"genre_scores_gemma":[0.9565592,0.0005031948,0.037904892,0.00068648445,0.00033756014,0.00025166257,0.00031954856,0.00023625013,0.0032011485],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9778992,0.011165955,0.0017652034,0.0040284265,0.0046430873,0.0004981944],"domain_scores_gemma":[0.8249652,0.13570394,0.015718853,0.012870684,0.008490407,0.0022509736],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.015883263,0.00084601453,0.00078580185,0.0056610946,0.0014766009,0.00737728,0.0013316523,0.0017490907,0.006231291],"category_scores_gemma":[0.13130356,0.00055860187,0.0009665167,0.005823302,0.008296666,0.014833947,0.0055364147,0.003970876,0.0007969232],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011549315,0.00039585537,0.24016635,0.0024512296,0.0009479663,0.0011601939,0.080285795,0.003179612,0.015594208,0.38623822,0.0054673674,0.26295832],"study_design_scores_gemma":[0.0001174964,0.00088515895,0.42396906,0.0011112656,0.0002946406,0.0021014777,0.020398298,0.012764567,0.007379171,0.47419724,0.056382596,0.00039898572],"about_ca_topic_score_codex":0.0022253194,"about_ca_topic_score_gemma":0.0018427708,"teacher_disagreement_score":0.015883263,"about_ca_system_score_codex":0.0018417869,"about_ca_system_score_gemma":0.0012312065,"threshold_uncertainty_score":0.08399975},"labels":[],"label_agreement":null},{"id":"W4414596238","doi":"10.1007/978-3-032-04614-7_24","title":"MATATA: Weakly Supervised End-to-End MAthematical Tool-Augmented Reasoning for Tabular Applications","year":2025,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Merck Canada Inc. (Canada); University of Toronto","funders":"","keywords":"Planner; Commonsense reasoning; Automated reasoning; Outcome (game theory); Model-based reasoning; Language model; Non-monotonic logic; Reasoning system","score_opus":0.01252736102412455,"score_gpt":0.2740569173297306,"score_spread":0.26152955630560604,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414596238","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011206419,0.00022958044,0.93540823,0.00017572251,0.00009133647,0.00031977362,0.0010133849,0.04810531,0.0034502044],"genre_scores_gemma":[0.13187306,0.000117588315,0.85239106,0.0003081365,0.00003215373,0.00058122433,0.005990885,0.0018471782,0.006858762],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987128,0.00037824898,0.000084544736,0.00045494063,0.00027865288,0.00009074444],"domain_scores_gemma":[0.9974343,0.0011720684,0.00014462385,0.0007041762,0.00042281995,0.00012199963],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017693444,0.0018874004,0.0007917054,0.0007522018,0.0006648429,0.0017542336,0.0041157007,0.0016349163,0.009270102],"category_scores_gemma":[0.006666967,0.00088654883,0.0017558335,0.00054982206,0.00093859126,0.0028179581,0.003404966,0.0036145481,0.0055982727],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007026993,0.00069438736,0.0027016224,0.0008081693,0.0002569715,0.00033134007,0.0007611491,0.26593047,0.021183893,0.013453281,0.053271774,0.6399042],"study_design_scores_gemma":[0.000044672925,0.00007619693,0.00019152308,0.000026983993,0.000020272853,0.000042357435,0.00005949654,0.9699424,0.008401887,0.012677316,0.00850048,0.000016357288],"about_ca_topic_score_codex":0.0056064995,"about_ca_topic_score_gemma":0.013387539,"teacher_disagreement_score":0.009270102,"about_ca_system_score_codex":0.0011147461,"about_ca_system_score_gemma":0.0025816222,"threshold_uncertainty_score":0.031011581},"labels":[],"label_agreement":null},{"id":"W4414604439","doi":"10.1109/ismac65024.2025.11175996","title":"Optimizing Filipino Text-to-Speech Synthesis: Integration of Generative Models","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Naturalness; Intelligibility (philosophy); Prosody; Mean opinion score; Speech synthesis; Active listening; Preprocessor; Generative grammar; Normalization (sociology)","score_opus":0.02097474468132208,"score_gpt":0.2858517800405878,"score_spread":0.2648770353592657,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414604439","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07446816,0.0007016345,0.9086718,0.00021986593,0.00015317851,0.00013937436,0.00044013798,0.007062233,0.008143696],"genre_scores_gemma":[0.7703837,0.00038828028,0.2159191,0.00019817709,0.00007379851,0.00022585572,0.0016009645,0.0008377334,0.010372341],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9998109,0.00003724728,0.000009348952,0.000060832845,0.000058338355,0.000023448365],"domain_scores_gemma":[0.9997942,0.00011579136,0.000009670593,0.000020083944,0.000046247267,0.000014098299],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00037773928,0.00087647716,0.00042306387,0.00024151908,0.00020646125,0.0005986058,0.00054790213,0.0005248439,0.0034547118],"category_scores_gemma":[0.000795211,0.00022603682,0.0004945278,0.00015110907,0.00021978839,0.0005218056,0.0006324614,0.0006065065,0.0016490552],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003891545,0.000109771296,0.0014143491,0.0002867971,0.00008995178,0.00041490042,0.00021610543,0.6251093,0.07985151,0.003187364,0.004105533,0.28482533],"study_design_scores_gemma":[0.0000130451335,0.00008264774,0.00031453438,0.000011225506,0.000022592416,0.000088312285,0.000036636295,0.9746629,0.021007163,0.00081635686,0.002932103,0.000012376447],"about_ca_topic_score_codex":0.0033662047,"about_ca_topic_score_gemma":0.006854274,"teacher_disagreement_score":0.0034547118,"about_ca_system_score_codex":0.00035109837,"about_ca_system_score_gemma":0.00050889887,"threshold_uncertainty_score":0.011557162},"labels":[],"label_agreement":null},{"id":"W4414643097","doi":"10.61091/ars164-04","title":"Counting adjacencies with difference at most one in -ary words","year":2025,"lang":"en","type":"article","venue":"Ars Combinatoria","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Statistic; Generating function; Set (abstract data type); Function (biology); Distribution (mathematics); Chebyshev filter","score_opus":0.008487833351066884,"score_gpt":0.2355380503270227,"score_spread":0.22705021697595584,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414643097","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8317195,0.00019317574,0.15757887,0.0002728471,0.000055975914,0.00004508536,0.00031144565,0.00024890158,0.009574173],"genre_scores_gemma":[0.97192276,0.00012026022,0.024318235,0.000054220007,0.00006464072,0.000068226036,0.0004147599,0.00009240496,0.0029445728],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9989273,0.00021266773,0.000097714656,0.00025247838,0.00029740031,0.0002123769],"domain_scores_gemma":[0.98962003,0.006865207,0.0012122237,0.0010183715,0.00082108076,0.0004631048],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011592552,0.00031275206,0.0006323707,0.0018921124,0.001059368,0.0020161364,0.0018949709,0.0007078929,0.0052586044],"category_scores_gemma":[0.011240236,0.0003246224,0.00038520465,0.0019762423,0.0017313847,0.0035743723,0.0011412296,0.00075052923,0.0006162702],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00056085736,0.00014245545,0.027131151,0.0002550227,0.000045086745,0.00080974965,0.0012842274,0.029187027,0.024619386,0.8526866,0.002308393,0.06097015],"study_design_scores_gemma":[0.00004059154,0.0002328549,0.008185119,0.000061652994,0.00007658737,0.0015544533,0.0006667331,0.23744778,0.03870838,0.7060239,0.00690632,0.000095613315],"about_ca_topic_score_codex":0.00078546605,"about_ca_topic_score_gemma":0.0011625383,"teacher_disagreement_score":0.0052586044,"about_ca_system_score_codex":0.0010289972,"about_ca_system_score_gemma":0.00054499554,"threshold_uncertainty_score":0.017591715},"labels":[],"label_agreement":null},{"id":"W4414652086","doi":"10.31234/osf.io/v28nf_v1","title":"Extracting Prototypes from Lexical Feature Norms for Settlement Concepts","year":2025,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Context (archaeology); Feature (linguistics); Human settlement; Settlement (finance); Task (project management); Categorization; Abstraction; Semantics (computer science)","score_opus":0.02277203231235816,"score_gpt":0.35464437999823556,"score_spread":0.3318723476858774,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414652086","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7551741,0.00038465913,0.23183258,0.00041845083,0.00009589397,0.0004545851,0.0016180675,0.0014293756,0.008592374],"genre_scores_gemma":[0.8894923,0.00012257174,0.10703911,0.000034783414,0.000025042375,0.00034147862,0.002059142,0.00013061415,0.00075500977],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99748236,0.0009124227,0.00022757312,0.0006645795,0.0005753716,0.0001376318],"domain_scores_gemma":[0.9885693,0.0069103343,0.0011994741,0.0011561794,0.0018507243,0.00031395815],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002881995,0.00066120294,0.0005234917,0.003751965,0.00068632123,0.00262642,0.0011226279,0.0009994388,0.0026941858],"category_scores_gemma":[0.023323316,0.00050379493,0.00078219443,0.0019021273,0.0011957592,0.005712057,0.0018144319,0.0009964871,0.0007555532],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013087732,0.00042327304,0.08277544,0.0020777492,0.00023347777,0.002479833,0.074874304,0.0065950835,0.071316555,0.075490735,0.011300906,0.67112386],"study_design_scores_gemma":[0.00045592707,0.0011673752,0.12172264,0.0009791601,0.0004072745,0.0043459544,0.100501105,0.32386842,0.041595615,0.3288571,0.075533964,0.00056551286],"about_ca_topic_score_codex":0.002411114,"about_ca_topic_score_gemma":0.0028539805,"teacher_disagreement_score":0.003751965,"about_ca_system_score_codex":0.0011344256,"about_ca_system_score_gemma":0.0010119366,"threshold_uncertainty_score":0.015241623},"labels":[],"label_agreement":null},{"id":"W4414970751","doi":"10.5430/wjel.v16n2p114","title":"Exploring AI-Generated Texts vs. Human-Written Texts in EFL Academic Writing: A Case Study of Qassim University in Saudi Arabia","year":2025,"lang":"en","type":"article","venue":"World Journal of English Language","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"Qassim University","keywords":"Grammar; Curriculum; Process (computing); Literacy; Foreign language; Coherence (philosophical gambling strategy); Academic writing; English language","score_opus":0.029323336890730824,"score_gpt":0.2997891607760918,"score_spread":0.270465823885361,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414970751","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9977004,0.00019453671,0.00017627266,0.00043735464,0.000008922638,0.00002063504,0.000015352058,0.0000071886425,0.0014393303],"genre_scores_gemma":[0.99764556,0.00023393579,0.00046498803,0.0001991575,0.000015620315,0.0000119559245,0.000018125102,0.000010661154,0.0013999784],"study_design_codex":"qualitative","study_design_gemma":"observational","domain_scores_codex":[0.99680233,0.0018622351,0.00021295268,0.00024287881,0.0005136966,0.00036592066],"domain_scores_gemma":[0.9879006,0.006717761,0.0022127277,0.00040472945,0.0015397784,0.001224333],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033061253,0.00050460664,0.0003417877,0.0024719306,0.00697603,0.0033444595,0.0010554674,0.0017282612,0.0023883088],"category_scores_gemma":[0.012248208,0.00026559064,0.00024447465,0.001634619,0.0031603465,0.0016245486,0.0023174006,0.0012543531,0.00059522974],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009742087,0.0004376174,0.051415455,0.00031772966,0.000014015332,0.027849762,0.8898114,0.00013870218,0.003531112,0.00094154093,0.0009215072,0.024523742],"study_design_scores_gemma":[0.000013749405,0.00021582066,0.041919906,0.00015942693,0.0000151516015,0.009232957,0.930607,0.000503412,0.0027126546,0.00029604562,0.014295531,0.000028263517],"about_ca_topic_score_codex":0.008347762,"about_ca_topic_score_gemma":0.021671891,"teacher_disagreement_score":0.008347762,"about_ca_system_score_codex":0.0031507888,"about_ca_system_score_gemma":0.002045871,"threshold_uncertainty_score":0.022860706},"labels":[],"label_agreement":null},{"id":"W4414973481","doi":"10.48550/arxiv.2510.05132","title":"Training Large Language Models To Reason In Parallel With Global Forking Tokens","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"University of Toronto; Amazon Web Services","keywords":"Set (abstract data type); Matching (statistics); Bipartite graph; Automated reasoning; Diversity (politics); Code (set theory); Training (meteorology)","score_opus":0.055165109209034044,"score_gpt":0.3201223017253664,"score_spread":0.2649571925163323,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414973481","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13087235,0.00065099733,0.8286871,0.0012214324,0.00022206419,0.00022969017,0.00087676494,0.03211119,0.005128513],"genre_scores_gemma":[0.6904565,0.00015533713,0.3004623,0.0007433937,0.00009420604,0.00024057113,0.0021579769,0.0011800431,0.0045095775],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99909854,0.00022110734,0.000046770114,0.00035988557,0.00016382078,0.00010991727],"domain_scores_gemma":[0.9970145,0.0015925977,0.00018829734,0.0008101592,0.00025891082,0.00013560441],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014330422,0.0014332626,0.0008839984,0.00063329923,0.00049667276,0.0012571322,0.0026970292,0.001238421,0.0037403689],"category_scores_gemma":[0.006542333,0.00075287774,0.0015526983,0.0006442978,0.0010890664,0.0032345855,0.0015162316,0.0037640422,0.0018165989],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00032080628,0.0003757608,0.0043793833,0.00024264274,0.00015666401,0.00018209031,0.00017860378,0.76901424,0.009522674,0.008171098,0.011680372,0.19577566],"study_design_scores_gemma":[0.000014191516,0.000018100924,0.00008612061,0.0000035797523,0.0000075338094,0.000011303312,0.000011118416,0.9920065,0.0012039607,0.006194014,0.00043966624,0.0000040130913],"about_ca_topic_score_codex":0.006992288,"about_ca_topic_score_gemma":0.025407836,"teacher_disagreement_score":0.006992288,"about_ca_system_score_codex":0.0014003409,"about_ca_system_score_gemma":0.0023057165,"threshold_uncertainty_score":0.013903141},"labels":[],"label_agreement":null},{"id":"W4414973489","doi":"10.1109/icmla66185.2025.00227","title":"NLD-LLM: A systematic framework for evaluating small language transformer models on natural language description","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of New Brunswick","funders":"","keywords":"Natural language; Transformer; Language model; Natural language understanding; Source code; Process (computing); Iterative and incremental development; Set (abstract data type); Task (project management)","score_opus":0.03913978433267463,"score_gpt":0.3440755121825993,"score_spread":0.3049357278499247,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414973489","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0848055,0.0014618391,0.87731445,0.00079138763,0.0002527845,0.003951063,0.0060411184,0.019177936,0.006203986],"genre_scores_gemma":[0.25859284,0.00043262026,0.7265422,0.00031617295,0.000026618707,0.0036363897,0.008257081,0.0013404769,0.0008556864],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9786023,0.012868912,0.001970804,0.0017270116,0.004369223,0.00046178745],"domain_scores_gemma":[0.9103602,0.06791096,0.0039709858,0.010555856,0.0062021217,0.0009998964],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02491151,0.003180841,0.0011082703,0.005510787,0.00096634636,0.0031184917,0.0035750363,0.0019958122,0.0034156428],"category_scores_gemma":[0.115704454,0.0010583386,0.0021123104,0.002407348,0.0024304572,0.0060925125,0.006273867,0.0039189053,0.0009644061],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020466603,0.002506025,0.018227339,0.0060014995,0.0011012328,0.0003359201,0.0017438278,0.5068885,0.020508535,0.05149113,0.026150512,0.36299884],"study_design_scores_gemma":[0.00033247747,0.0013775723,0.0018386673,0.00034208602,0.00016266815,0.00013202823,0.0004550602,0.9415239,0.018604612,0.026245886,0.008864224,0.00012094824],"about_ca_topic_score_codex":0.008011717,"about_ca_topic_score_gemma":0.0140387425,"teacher_disagreement_score":0.02491151,"about_ca_system_score_codex":0.0038247595,"about_ca_system_score_gemma":0.0055629197,"threshold_uncertainty_score":0.13174623},"labels":[],"label_agreement":null},{"id":"W4415045391","doi":"10.7202/1118950ar","title":"Machines à écrire","year":2024,"lang":"fr","type":"article","venue":"Sens public","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Context (archaeology); Order (exchange); Period (music)","score_opus":0.028366634382013603,"score_gpt":0.2971950064287017,"score_spread":0.26882837204668814,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415045391","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02233769,0.0069289473,0.7540363,0.033411447,0.004458376,0.00028136358,0.0011482224,0.004493513,0.17290403],"genre_scores_gemma":[0.34089646,0.0066100764,0.46505263,0.009726136,0.004944872,0.00065295596,0.0028662754,0.001886152,0.1673645],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.992861,0.0021708247,0.0004772122,0.002668364,0.0014061738,0.0004163716],"domain_scores_gemma":[0.98902357,0.0050532417,0.0005752013,0.0036278185,0.0012830327,0.00043712437],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004535182,0.001518986,0.0010809108,0.0018502771,0.0030108402,0.009443803,0.0025372005,0.0039688903,0.029273491],"category_scores_gemma":[0.01999993,0.0009057175,0.0019281335,0.0016983415,0.0067701484,0.020413954,0.005104575,0.0041020266,0.013516729],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020715968,0.00009065409,0.0020519109,0.00062948075,0.000104258455,0.00043756136,0.0038520053,0.0027390611,0.0070290025,0.8020349,0.028572517,0.15225144],"study_design_scores_gemma":[0.000038252623,0.00011816982,0.0010517685,0.00030828145,0.000110516856,0.0010133974,0.0015845189,0.012553015,0.0065110843,0.49721265,0.47941583,0.00008254934],"about_ca_topic_score_codex":0.0014598026,"about_ca_topic_score_gemma":0.0012333861,"teacher_disagreement_score":0.029273491,"about_ca_system_score_codex":0.0013958825,"about_ca_system_score_gemma":0.0015474395,"threshold_uncertainty_score":0.09792954},"labels":[],"label_agreement":null},{"id":"W4415312099","doi":"10.48550/arxiv.2506.14598","title":"Learning From the Past with Cascading Eligibility Traces","year":2025,"lang":"en","type":"preprint","venue":"White Rose Research Online (University of Leeds, The University of Sheffield, University of York)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research; Nvidia; National Science Foundation","keywords":"Cascade; Work (physics); Information cascade; Reentry; Set (abstract data type); Basis (linear algebra)","score_opus":0.03615657059625891,"score_gpt":0.28387430344960957,"score_spread":0.24771773285335066,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415312099","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08883749,0.000152622,0.90752465,0.0004277637,0.00004352382,0.00005009905,0.00018286749,0.0008016517,0.0019793785],"genre_scores_gemma":[0.92546296,0.00024721006,0.069763936,0.00010530944,0.000038515354,0.00012979905,0.00022575312,0.00011622604,0.0039104153],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992717,0.00017681211,0.00007277089,0.00022547059,0.00017227474,0.00008105439],"domain_scores_gemma":[0.99416465,0.0037989004,0.00062083517,0.0007701281,0.00035280257,0.00029260665],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015861298,0.00067134626,0.0007312852,0.0010858662,0.0004484562,0.0013583959,0.0015269803,0.0010170252,0.003682046],"category_scores_gemma":[0.011264459,0.0005645827,0.0010216088,0.0008442111,0.001763296,0.0036435074,0.0018329831,0.002309294,0.00041940177],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016240047,0.00007710524,0.0025758909,0.0000947166,0.00006513316,0.0002453362,0.0002641148,0.83391494,0.0027943437,0.1135896,0.0006982136,0.045518234],"study_design_scores_gemma":[0.000006490983,0.000014997939,0.000115653784,0.0000057529855,0.000006336769,0.000020205369,0.000007798048,0.93496436,0.0004625668,0.06415631,0.00023194405,0.0000075366556],"about_ca_topic_score_codex":0.0050221123,"about_ca_topic_score_gemma":0.0061580385,"teacher_disagreement_score":0.0050221123,"about_ca_system_score_codex":0.0010055576,"about_ca_system_score_gemma":0.0010376178,"threshold_uncertainty_score":0.0123176575},"labels":[],"label_agreement":null},{"id":"W4415318169","doi":"10.48550/arxiv.2510.07203","title":"Sunflower: A New Approach To Expanding Coverage of African Languages in Large Language Models","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"International Development Research Centre","keywords":"Swahili; Comprehension; Languages of Africa; Bantu languages; State (computer science); Language model","score_opus":0.031715355184456305,"score_gpt":0.3121366835627905,"score_spread":0.2804213283783342,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415318169","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013469735,0.00031759936,0.9552786,0.0010174235,0.00009973276,0.00028015915,0.0042273733,0.020929426,0.004380036],"genre_scores_gemma":[0.14544387,0.00042044636,0.83883923,0.00047066613,0.00009147342,0.00081461435,0.006652076,0.0030869322,0.004180676],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99886996,0.00048425267,0.0000912255,0.0002546943,0.00022689994,0.00007289754],"domain_scores_gemma":[0.9950912,0.003514098,0.00021054066,0.0007310746,0.00026665797,0.00018633364],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0044379113,0.0010566129,0.00073759304,0.0015243364,0.0008576555,0.0021630866,0.002731424,0.0012077101,0.010142119],"category_scores_gemma":[0.014438105,0.0010482541,0.002708338,0.0011300499,0.0008176297,0.007438005,0.0031850561,0.0026097267,0.0016525232],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00087627716,0.0004291982,0.010229192,0.00084048195,0.0004881866,0.00059844175,0.0031424898,0.29736096,0.006054507,0.25574023,0.077314705,0.3469253],"study_design_scores_gemma":[0.000089885165,0.00005322443,0.00027842936,0.00006845692,0.000046351022,0.00008078005,0.00012531366,0.84106195,0.0012710213,0.11033431,0.046564125,0.0000262341],"about_ca_topic_score_codex":0.011747878,"about_ca_topic_score_gemma":0.028781,"teacher_disagreement_score":0.011747878,"about_ca_system_score_codex":0.001448369,"about_ca_system_score_gemma":0.0020986414,"threshold_uncertainty_score":0.033928752},"labels":[],"label_agreement":null},{"id":"W4415320279","doi":"10.48550/arxiv.2505.14804","title":"Automated Journalistic Questions: A New Method for Extracting 5W1H in French","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Pipeline (software); Task (project management); Journalism; Information extraction; Task analysis","score_opus":0.054139781314937935,"score_gpt":0.40108874387464505,"score_spread":0.3469489625597071,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415320279","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.039038785,0.0014930121,0.8790445,0.0025096668,0.00033447702,0.0010949502,0.017667636,0.042747263,0.016069742],"genre_scores_gemma":[0.12736756,0.0005873096,0.8292653,0.00048596918,0.0003542859,0.00064667454,0.0282372,0.0018227346,0.011232987],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9980019,0.00051023933,0.00018533721,0.00069888326,0.00042802747,0.00017560768],"domain_scores_gemma":[0.99720806,0.0012707709,0.00024006813,0.00026910802,0.00090792263,0.000104080485],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001966056,0.0016779238,0.00070785073,0.008361829,0.0021301017,0.0029642223,0.000956262,0.0019359215,0.01186756],"category_scores_gemma":[0.005118319,0.00075049384,0.0015631268,0.0029322559,0.0011573116,0.0035451958,0.0020273256,0.0015357718,0.005062083],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022963302,0.00015299396,0.0052874614,0.0011840484,0.00013591301,0.0009599192,0.0046321987,0.0033215804,0.053944822,0.019997187,0.060889266,0.849265],"study_design_scores_gemma":[0.00016961929,0.00031158247,0.023324292,0.00034405367,0.00029162053,0.0033593723,0.0042676767,0.22283657,0.09006756,0.042622313,0.61207086,0.00033449862],"about_ca_topic_score_codex":0.034443367,"about_ca_topic_score_gemma":0.04245107,"teacher_disagreement_score":0.034443367,"about_ca_system_score_codex":0.002049766,"about_ca_system_score_gemma":0.0038333975,"threshold_uncertainty_score":0.0684858},"labels":[],"label_agreement":null},{"id":"W4415337949","doi":"10.48550/arxiv.2509.00404","title":"Metis: Training LLMs with FP4 Quantization","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Metis; Quantization (signal processing); Singular value decomposition; Training (meteorology); Spectral line; Matching (statistics)","score_opus":0.051549079629148446,"score_gpt":0.3046366567186559,"score_spread":0.25308757708950747,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415337949","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.035367787,0.00081574335,0.91999334,0.0008078956,0.00043898332,0.0001619567,0.0010054141,0.03292511,0.008483792],"genre_scores_gemma":[0.3777367,0.00025316633,0.5994915,0.0011173435,0.00012456092,0.0004670981,0.004307846,0.0037351637,0.01276669],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99905175,0.00025335405,0.00006434155,0.00024460675,0.00027641736,0.00010953072],"domain_scores_gemma":[0.9988292,0.0005389897,0.00005511609,0.00025499667,0.00025093224,0.00007084697],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015383697,0.0012865247,0.00076669926,0.00061655923,0.000612879,0.0012401834,0.0021270006,0.0015682919,0.012027516],"category_scores_gemma":[0.0073111304,0.0005916813,0.0006692376,0.00063456193,0.00089431653,0.0020904974,0.0019358819,0.0027208303,0.0054442897],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007167323,0.00021493131,0.0014941487,0.00030641645,0.00012808849,0.0002208473,0.00028067932,0.29039225,0.023077674,0.021423295,0.051707268,0.6100376],"study_design_scores_gemma":[0.000040668427,0.000052157025,0.00012853512,0.000021051266,0.0000071314644,0.000033652184,0.000039499144,0.9774774,0.007813441,0.010306348,0.0040670773,0.000013140322],"about_ca_topic_score_codex":0.008994213,"about_ca_topic_score_gemma":0.015698396,"teacher_disagreement_score":0.012027516,"about_ca_system_score_codex":0.0013163747,"about_ca_system_score_gemma":0.0018126036,"threshold_uncertainty_score":0.040235996},"labels":[],"label_agreement":null},{"id":"W4415368940","doi":"10.37236/11555","title":"Complement Avoidance in Binary Words","year":2025,"lang":"en","type":"article","venue":"The Electronic Journal of Combinatorics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Winnipeg","funders":"","keywords":"Complement (music); Binary number; Word (group theory); Binary data; Factor (programming language)","score_opus":0.006246664915193381,"score_gpt":0.27100077517690413,"score_spread":0.26475411026171075,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415368940","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9302115,0.00091382983,0.024977751,0.00029441243,0.00010632468,0.00003726685,0.00008424871,0.000092459515,0.043282244],"genre_scores_gemma":[0.9813768,0.00046729913,0.008055465,0.00010557652,0.00012133971,0.000060798244,0.00012710087,0.000075504686,0.009610093],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9996094,0.00006893684,0.00002455788,0.00006189738,0.00011501964,0.00012020836],"domain_scores_gemma":[0.9988668,0.00056729076,0.0001931385,0.00006388937,0.00015452699,0.00015441106],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00032862593,0.00055256765,0.00046457382,0.0015291615,0.0015906836,0.0023908403,0.00035009364,0.00073412084,0.0051003],"category_scores_gemma":[0.0023979305,0.00031068246,0.000436955,0.0006741494,0.0028870706,0.0020597195,0.0013284271,0.00094469596,0.0005908444],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020743009,0.000047725665,0.0016977276,0.00016510994,0.000013264028,0.0005394253,0.0008808287,0.0030777357,0.012514862,0.9637037,0.0018461361,0.015306158],"study_design_scores_gemma":[0.000059233622,0.00012333649,0.0015375956,0.00013749945,0.000037062888,0.0008019756,0.0006748869,0.019265374,0.009574772,0.9520914,0.015646633,0.00005016436],"about_ca_topic_score_codex":0.00089555245,"about_ca_topic_score_gemma":0.0010850091,"teacher_disagreement_score":0.0051003,"about_ca_system_score_codex":0.00087906246,"about_ca_system_score_gemma":0.0004446085,"threshold_uncertainty_score":0.017062247},"labels":[],"label_agreement":null},{"id":"W4415381531","doi":"10.1145/3736731.3746137","title":"LLMs in Citation Intent Classification: Progress, Precision, and Reproducibility Challenges","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"Good Ventures Foundation; Open Philanthropy Project; Alfred P. Sloan Foundation","keywords":"Citation; Reproducibility; Reliability (semiconductor)","score_opus":0.04366004339709035,"score_gpt":0.3280326195008369,"score_spread":0.28437257610374655,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415381531","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"reproducibility","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"reproducibility","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14666677,0.11258271,0.60594803,0.0488127,0.009520143,0.0012552068,0.016441353,0.038203128,0.020569904],"genre_scores_gemma":[0.5650694,0.010341269,0.36892834,0.005527781,0.011948254,0.00085247477,0.01901334,0.0034966348,0.014822592],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.8224731,0.099852435,0.016050175,0.017868709,0.04050911,0.0032465067],"domain_scores_gemma":[0.40022948,0.39498895,0.018075606,0.12125349,0.060386866,0.0050656185],"candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.18147197,0.00317771,0.0075658364,0.02413376,0.004609883,0.02337981,0.010352737,0.006748307,0.0037774784],"category_scores_gemma":[0.3580924,0.0022232481,0.0033625448,0.019089978,0.005259148,0.020286946,0.010324951,0.010230769,0.008299362],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00097342866,0.00078769523,0.044562798,0.0022377353,0.0018741295,0.000104958686,0.0014682658,0.009087473,0.0022584684,0.015163934,0.090143465,0.83133763],"study_design_scores_gemma":[0.00045380398,0.0007638147,0.03095859,0.0021076936,0.00170985,0.0007141889,0.0024530108,0.58797723,0.01560615,0.2398938,0.116760306,0.00060157024],"about_ca_topic_score_codex":0.014693917,"about_ca_topic_score_gemma":0.023146592,"teacher_disagreement_score":0.97586626,"about_ca_system_score_codex":0.0031932464,"about_ca_system_score_gemma":0.006833194,"threshold_uncertainty_score":0.959727},"labels":[],"label_agreement":null},{"id":"W4415428125","doi":"10.3233/faia251331","title":"ALF: A Fine-Grained French Analogical Dataset for Evaluating Lexical Knowledge of Large Language Models","year":2025,"lang":"","type":"book-chapter","venue":"Frontiers in artificial intelligence and applications","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Fluency; Lexical item; Lexicographical order; Key (lock); Vocabulary; Computational linguistics","score_opus":0.07860967414722973,"score_gpt":0.3867682798869631,"score_spread":0.30815860573973336,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415428125","genre_codex":"empirical","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4150492,0.011648959,0.091034435,0.0025544174,0.0009647503,0.001234141,0.39102894,0.04779035,0.038694777],"genre_scores_gemma":[0.28966913,0.0009089584,0.07821795,0.00076756923,0.00017947998,0.0006951228,0.6232417,0.0010190327,0.0053010606],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99824226,0.0006879481,0.00015277276,0.00047090996,0.00034991087,0.00009628515],"domain_scores_gemma":[0.9954176,0.002718149,0.00016536875,0.0009912148,0.0005129859,0.0001946941],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019351799,0.0018665796,0.0007092468,0.0031439757,0.0009386918,0.0017987243,0.0021472843,0.002358094,0.008822985],"category_scores_gemma":[0.0113591915,0.00033068762,0.0015196515,0.0024679068,0.00063497276,0.0018903213,0.0012882021,0.0012833525,0.004300178],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017717496,0.0013300775,0.029381573,0.0034908736,0.0010813789,0.0015919664,0.000695419,0.08501699,0.009320603,0.014710554,0.4943448,0.35726404],"study_design_scores_gemma":[0.0013012917,0.0012732003,0.044137772,0.00036363106,0.00031209565,0.002219245,0.0010010537,0.5939105,0.013416377,0.030770572,0.31100866,0.0002855883],"about_ca_topic_score_codex":0.028423348,"about_ca_topic_score_gemma":0.04682894,"teacher_disagreement_score":0.028423348,"about_ca_system_score_codex":0.00158359,"about_ca_system_score_gemma":0.0010805108,"threshold_uncertainty_score":0.056515813},"labels":[],"label_agreement":null},{"id":"W4415428196","doi":"10.3233/faia251188","title":"PoT-PTQ:Two-Step Power-of-Two Post-Training for LLMs","year":2025,"lang":"","type":"book-chapter","venue":"Frontiers in artificial intelligence and applications","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Mila - Quebec Artificial Intelligence Institute; Université de Montréal; McGill University","funders":"","keywords":"Quantization (signal processing); Inference; Software deployment; Integer (computer science); Floating point; Natural language; Language model","score_opus":0.04209750365547548,"score_gpt":0.3277798479616582,"score_spread":0.2856823443061827,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415428196","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009033522,0.00033644107,0.97221506,0.00037062116,0.00021099954,0.00016416839,0.00026972173,0.015754517,0.0016449351],"genre_scores_gemma":[0.21696097,0.00030912395,0.7673351,0.0010582485,0.0001704527,0.00060996885,0.0021480303,0.0025925913,0.008815517],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99882656,0.00022291389,0.00010200604,0.00028761933,0.0004228692,0.00013806524],"domain_scores_gemma":[0.99761605,0.0010846071,0.00012839436,0.0005765748,0.00048320816,0.00011118197],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017005597,0.0017312972,0.0012154862,0.0007818003,0.0008523129,0.00166517,0.0038894208,0.0019906445,0.01850654],"category_scores_gemma":[0.012897038,0.0008959037,0.0009962787,0.0008758778,0.001118632,0.0042149574,0.0035994896,0.0042421934,0.0058035473],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006323069,0.00025695865,0.0014129593,0.0003703599,0.00012307489,0.00033906059,0.00030805395,0.11389476,0.021483019,0.018353906,0.027814077,0.8150115],"study_design_scores_gemma":[0.000073643336,0.00011070208,0.0002786582,0.000028528422,0.000019851614,0.000097697506,0.000052727326,0.9664951,0.012185542,0.015413731,0.0052187643,0.000025030162],"about_ca_topic_score_codex":0.005238566,"about_ca_topic_score_gemma":0.010762325,"teacher_disagreement_score":0.01850654,"about_ca_system_score_codex":0.0010849382,"about_ca_system_score_gemma":0.002236665,"threshold_uncertainty_score":0.06191051},"labels":[],"label_agreement":null},{"id":"W4415524320","doi":"10.1109/mlsp62443.2025.11204231","title":"Colflor: Towards Bert-Size Vision-Language Document Retrieval Models","year":2025,"lang":"","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Document retrieval; Encoding (memory); Process (computing); Image retrieval; Document clustering; Visual Word; Infographic; Data retrieval; Vector space model","score_opus":0.009472087359197887,"score_gpt":0.309669048924067,"score_spread":0.3001969615648691,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415524320","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016747022,0.0026708352,0.95428634,0.0009416185,0.00020490418,0.00053381664,0.0025494073,0.017907701,0.004158431],"genre_scores_gemma":[0.16038004,0.0023085389,0.80863345,0.0015223816,0.00036936463,0.0011072885,0.008713573,0.0014715848,0.015493758],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986808,0.00026656507,0.00010290818,0.0003999781,0.0004314331,0.000118387194],"domain_scores_gemma":[0.9971982,0.0012494153,0.00019835414,0.000526734,0.0007291219,0.00009801718],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021673474,0.0015867309,0.0018851928,0.0025944097,0.0006510293,0.002767018,0.0043990747,0.0025359567,0.006189877],"category_scores_gemma":[0.008690778,0.0008057331,0.0020231178,0.0021354163,0.0006961869,0.0048140753,0.0016118975,0.002077132,0.0068395464],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009264685,0.0005277988,0.0014789797,0.00074305705,0.00027468425,0.00026446462,0.00019487515,0.22095641,0.019532282,0.01723316,0.059485313,0.6783826],"study_design_scores_gemma":[0.0000713744,0.000121373036,0.00026527635,0.00003035183,0.000040819956,0.00017352625,0.000030777188,0.9755568,0.0050512673,0.009795401,0.008828717,0.00003430455],"about_ca_topic_score_codex":0.016601404,"about_ca_topic_score_gemma":0.016847892,"teacher_disagreement_score":0.016601404,"about_ca_system_score_codex":0.0023659642,"about_ca_system_score_gemma":0.0017390032,"threshold_uncertainty_score":0.03300959},"labels":[],"label_agreement":null},{"id":"W4415602683","doi":"10.36227/techrxiv.176162061.19169845/v1","title":"MaplePT: A Generalized Instruction-Tuned Canadian Language Model Trained on Distributed Low-Cost Compute","year":2025,"lang":"","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Leverage (statistics); Cloud computing; Natural language; Language model; Inference; Scaling; Fluency; Quantization (signal processing)","score_opus":0.010891640599484729,"score_gpt":0.26823010323244634,"score_spread":0.2573384626329616,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415602683","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.34419942,0.0014752289,0.43170837,0.0033515175,0.0012206482,0.00089487364,0.026096625,0.13515091,0.055902343],"genre_scores_gemma":[0.6805158,0.00049051456,0.23700461,0.0010154048,0.00007086556,0.0005186309,0.03828425,0.0038561546,0.0382437],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9997093,0.00002731906,0.000009038706,0.000108282315,0.00007612294,0.00006990531],"domain_scores_gemma":[0.9995401,0.00012785065,0.000011710142,0.000062832645,0.00020873485,0.000048804544],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00047250904,0.0011847035,0.0004810489,0.00044019127,0.0010994053,0.0011790522,0.0023931935,0.0007308787,0.006786904],"category_scores_gemma":[0.0020673925,0.00043692667,0.00094676076,0.0008112399,0.0006519514,0.0013419669,0.0008908097,0.0024350977,0.002630269],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00073587935,0.00023629192,0.005302487,0.00031180002,0.0002788464,0.0003926013,0.0004730871,0.5564458,0.019595843,0.011104459,0.15007538,0.25504753],"study_design_scores_gemma":[0.000045989116,0.000048397782,0.00078906695,0.000010380509,0.000025743586,0.000037367714,0.000090053996,0.9803073,0.004751452,0.0021327818,0.011730928,0.000030623436],"about_ca_topic_score_codex":0.6277694,"about_ca_topic_score_gemma":0.74466425,"teacher_disagreement_score":0.3722306,"about_ca_system_score_codex":0.004535225,"about_ca_system_score_gemma":0.009911479,"threshold_uncertainty_score":0.74884546},"labels":[],"label_agreement":null},{"id":"W4416033657","doi":"10.18653/v1/2025.winlp-main.37","title":"Reference-Guided Verdict: LLMs-as-Judges in Automatic Evaluation of Free-Form QA","year":2025,"lang":"","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Alliance de recherche numérique du Canada; Research Nova Scotia","keywords":"Process (computing); Automation; Identification (biology); Set (abstract data type); Measure (data warehouse)","score_opus":0.04708086877994515,"score_gpt":0.36197880997188403,"score_spread":0.31489794119193887,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416033657","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.23905785,0.003191333,0.6249752,0.0032966996,0.0012710404,0.00090541044,0.005044929,0.08065299,0.041604515],"genre_scores_gemma":[0.8420695,0.00013548958,0.14001897,0.00053595135,0.00019047198,0.00018975385,0.0042730807,0.003668665,0.008918223],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.96759063,0.021996375,0.0014679838,0.002809162,0.0049294336,0.0012064773],"domain_scores_gemma":[0.96014297,0.022626942,0.0012860015,0.006231889,0.008200411,0.0015116717],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.016293053,0.0012316662,0.0015928675,0.0021392168,0.0014362329,0.003201304,0.0032911475,0.0035263025,0.01365771],"category_scores_gemma":[0.06778133,0.00059429667,0.0006910695,0.00084028824,0.0014732435,0.0039357436,0.005023778,0.0028405397,0.007473405],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.005793632,0.000665605,0.010066781,0.0012435623,0.00028705172,0.00088441215,0.0036713253,0.055709083,0.027211206,0.040623702,0.16115494,0.69268864],"study_design_scores_gemma":[0.00043898457,0.00064151344,0.004352362,0.0002735274,0.00011225584,0.00047913758,0.00096646027,0.86570394,0.039547402,0.05371813,0.033566874,0.00019946146],"about_ca_topic_score_codex":0.005503122,"about_ca_topic_score_gemma":0.011351585,"teacher_disagreement_score":0.016293053,"about_ca_system_score_codex":0.0014839859,"about_ca_system_score_gemma":0.002541105,"threshold_uncertainty_score":0.08616692},"labels":[],"label_agreement":null},{"id":"W4416033845","doi":"10.18653/v1/2025.wmt-1.69","title":"MSLC25: Metric Performance on Low-Quality Machine Translation, Empty Strings, and Language Variants","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Metric (unit); Machine translation; Focus (optics); Variety (cybernetics); Translation (biology); Range (aeronautics)","score_opus":0.012527109567021417,"score_gpt":0.3015938198665457,"score_spread":0.2890667102995243,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416033845","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.531623,0.03295628,0.10031789,0.0042282287,0.008233773,0.0019526676,0.14090596,0.11119318,0.06858901],"genre_scores_gemma":[0.5201012,0.0018240378,0.12042565,0.0013276732,0.0009905437,0.0013749918,0.32602543,0.013333057,0.014597492],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.97167385,0.011220657,0.0035579957,0.0044329427,0.0075451313,0.001569457],"domain_scores_gemma":[0.9580573,0.017189087,0.0017522303,0.009672921,0.011307414,0.0020210424],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.016644532,0.005851497,0.0036163223,0.01074411,0.00301707,0.005195859,0.0035639042,0.0037337667,0.0076346877],"category_scores_gemma":[0.061219282,0.000610425,0.0021178436,0.009803848,0.0022324529,0.0056880424,0.0044705844,0.003301732,0.010153409],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004363091,0.0013364478,0.023008557,0.005303952,0.0020431152,0.0007769254,0.0011234384,0.05262688,0.018392887,0.0067377267,0.51795113,0.36633575],"study_design_scores_gemma":[0.0023417003,0.0055434806,0.07417456,0.0013021354,0.001090407,0.004116704,0.0030323004,0.45132703,0.120615,0.038714778,0.2963772,0.0013646231],"about_ca_topic_score_codex":0.015614907,"about_ca_topic_score_gemma":0.02495846,"teacher_disagreement_score":0.016644532,"about_ca_system_score_codex":0.0033112138,"about_ca_system_score_gemma":0.0032946263,"threshold_uncertainty_score":0.08802581},"labels":[],"label_agreement":null},{"id":"W4416035125","doi":"10.18653/v1/2025.emnlp-main.1803","title":"Which Word Orders Facilitate Length Generalization in LMs? An Investigation with GCG-Based Artificial Languages","year":2025,"lang":"","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Generalization; Word (group theory); Natural language; Feature (linguistics)","score_opus":0.020435529820189872,"score_gpt":0.28471531903156166,"score_spread":0.2642797892113718,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416035125","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8951005,0.00019410504,0.093488805,0.0010451736,0.000047012098,0.00008384698,0.00034573738,0.0008816783,0.008812995],"genre_scores_gemma":[0.9766707,0.00007053934,0.022310302,0.00013991132,0.000016379463,0.000055417833,0.000174203,0.00011965036,0.00044290462],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9968309,0.0018752048,0.00017249695,0.00059574394,0.00040363925,0.00012198363],"domain_scores_gemma":[0.9644658,0.025476469,0.002071224,0.006084613,0.0014342995,0.0004675873],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0042905523,0.00042261963,0.0003290134,0.0006485565,0.00041745484,0.0017802641,0.00092338957,0.00076443836,0.0033824085],"category_scores_gemma":[0.040495846,0.00032214585,0.00047074945,0.0006244765,0.0020462507,0.005189122,0.0017651141,0.0015844995,0.00050222373],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012659879,0.0005986877,0.10753407,0.0013679183,0.00022234538,0.0011919902,0.021770915,0.05758032,0.13404642,0.34638438,0.0060377643,0.3219992],"study_design_scores_gemma":[0.00015637487,0.0007478605,0.03601963,0.00015917131,0.00017437457,0.00092520664,0.0061920583,0.33992448,0.067104556,0.53029114,0.018135196,0.00016991125],"about_ca_topic_score_codex":0.0008243179,"about_ca_topic_score_gemma":0.0009943348,"teacher_disagreement_score":0.0042905523,"about_ca_system_score_codex":0.00083101017,"about_ca_system_score_gemma":0.00050039333,"threshold_uncertainty_score":0.022690892},"labels":[],"label_agreement":null},{"id":"W4416036783","doi":"10.18653/v1/2025.emnlp-main.442","title":"Beyond Seen Data: Improving KBQA Generalization Through Schema-Guided Logical Form Generation","year":2025,"lang":"","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Atomic Energy of Canada Limited; Institute for Catastrophic Loss Reduction","keywords":"Generalization; Set (abstract data type); Feature (linguistics); Relation (database)","score_opus":0.0634349050262608,"score_gpt":0.3447075123731154,"score_spread":0.2812726073468546,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416036783","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11516323,0.0059313495,0.8217143,0.004526578,0.0003817077,0.0011554695,0.007860047,0.0348282,0.008439157],"genre_scores_gemma":[0.43906865,0.0015415874,0.5223401,0.0033931308,0.00020179557,0.0005184932,0.027817464,0.0010405338,0.004078348],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9968135,0.0011619077,0.00027743023,0.0009791692,0.0006297038,0.00013828326],"domain_scores_gemma":[0.9895858,0.005077242,0.00031118433,0.0033317045,0.0015142205,0.00017984735],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0049670553,0.0012482872,0.0011332041,0.0022371856,0.00067745306,0.0022909597,0.003884934,0.0018161754,0.004192255],"category_scores_gemma":[0.026853189,0.00054271595,0.0021836732,0.0019363288,0.0012600427,0.008316038,0.004273807,0.0032005052,0.0023501944],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00042840684,0.00089901994,0.011826386,0.0009762418,0.00036218498,0.00048159406,0.0015108826,0.10665126,0.014953096,0.018559935,0.061356943,0.7819941],"study_design_scores_gemma":[0.00025137665,0.00023070475,0.0020846385,0.00015778586,0.0002247411,0.00041476867,0.0005129034,0.88959897,0.012917746,0.062251806,0.031287372,0.000067215726],"about_ca_topic_score_codex":0.011637869,"about_ca_topic_score_gemma":0.016189575,"teacher_disagreement_score":0.011637869,"about_ca_system_score_codex":0.0010155994,"about_ca_system_score_gemma":0.0021968328,"threshold_uncertainty_score":0.026268601},"labels":[],"label_agreement":null},{"id":"W4416037105","doi":"10.18653/v1/2025.emnlp-main.69","title":"LingGym: How Far Are LLMs from Thinking Like Field Linguists?","year":2025,"lang":"","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Field (mathematics); Perspective (graphical)","score_opus":0.011029987447588336,"score_gpt":0.2717450584029037,"score_spread":0.2607150709553154,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416037105","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.54390967,0.006225882,0.25016457,0.019773236,0.0008925192,0.00056072336,0.010217932,0.05997256,0.10828295],"genre_scores_gemma":[0.82306236,0.0011250704,0.14771995,0.003026315,0.00015942183,0.00030559127,0.013167871,0.0030678269,0.008365612],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99426717,0.00306851,0.00026603724,0.0010923698,0.0010064222,0.00029953851],"domain_scores_gemma":[0.968793,0.0174971,0.0010012413,0.00898417,0.0020828834,0.0016416091],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012281821,0.0012184903,0.00078716024,0.0020031475,0.0012210651,0.005404155,0.0029976848,0.0026170288,0.0110963555],"category_scores_gemma":[0.049177434,0.0005825032,0.0007744517,0.0015681254,0.0026892128,0.01712847,0.0044765286,0.0036319483,0.0058438326],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020902865,0.00078970543,0.061429147,0.0018683799,0.00040179817,0.00036576917,0.008511277,0.023843471,0.009955204,0.067095466,0.09501864,0.7286309],"study_design_scores_gemma":[0.0005260214,0.0013082967,0.029918136,0.0012249915,0.00021671066,0.00081695896,0.01022808,0.25548685,0.025183614,0.4233727,0.25143972,0.00027802697],"about_ca_topic_score_codex":0.006071735,"about_ca_topic_score_gemma":0.007496345,"teacher_disagreement_score":0.012281821,"about_ca_system_score_codex":0.0017654031,"about_ca_system_score_gemma":0.0031696395,"threshold_uncertainty_score":0.06495327},"labels":[],"label_agreement":null},{"id":"W4416037239","doi":"10.18653/v1/2025.emnlp-main.6","title":"QFrCoLA: a Quebec-French Corpus of Linguistic Acceptability Judgments","year":2025,"lang":"","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Université Laval","funders":"Fonds de recherche du Québec – Nature et technologies; Natural Sciences and Engineering Research Council of Canada","keywords":"Corpus linguistics; Set (abstract data type); Term (time); Subject (documents)","score_opus":0.011369933077680435,"score_gpt":0.2944397419745004,"score_spread":0.28306980889682,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416037239","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4178775,0.0035203244,0.022445753,0.0036654468,0.00073375105,0.0011030197,0.463331,0.014376355,0.07294691],"genre_scores_gemma":[0.45880082,0.00059946626,0.028535752,0.00085723854,0.00012370048,0.00078814436,0.49119335,0.0011572036,0.017944338],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9977412,0.0007114934,0.00012170766,0.00047916328,0.0007786843,0.00016771503],"domain_scores_gemma":[0.99348736,0.0019848682,0.00023656864,0.000770661,0.0032063522,0.00031424503],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015472994,0.0014750523,0.00040668395,0.002693589,0.0024662989,0.0014357037,0.0015494839,0.0014933982,0.012247804],"category_scores_gemma":[0.010569607,0.00026976038,0.0005182489,0.002016645,0.0011042714,0.0010593281,0.0012816241,0.0019518324,0.0044856365],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007770937,0.00057558133,0.028051093,0.0016001664,0.00021520982,0.0012432693,0.004588186,0.0061412174,0.015211856,0.00699332,0.7470951,0.1875079],"study_design_scores_gemma":[0.00033389576,0.00024008524,0.20298135,0.0006379941,0.00012287908,0.001235286,0.006655212,0.048948176,0.014854543,0.004533101,0.7190596,0.00039788682],"about_ca_topic_score_codex":0.62758505,"about_ca_topic_score_gemma":0.77726555,"teacher_disagreement_score":0.37241495,"about_ca_system_score_codex":0.0060879695,"about_ca_system_score_gemma":0.0047711157,"threshold_uncertainty_score":0.7492163},"labels":[],"label_agreement":null},{"id":"W4416037633","doi":"10.18653/v1/2025.arabicnlp-sharedtasks.99","title":"NADI 2025: The First Multidialectal Arabic Speech Processing Shared Task","year":2025,"lang":"","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Arabic; Task (project management); Speech processing; Natural (archaeology); Natural language","score_opus":0.012154805244740947,"score_gpt":0.2754053313416078,"score_spread":0.26325052609686683,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416037633","genre_codex":"empirical","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3088586,0.009556786,0.19422524,0.0118349325,0.014838301,0.010861054,0.237536,0.092001945,0.12028708],"genre_scores_gemma":[0.3149979,0.00068099267,0.20092294,0.0030038755,0.0010934804,0.006994495,0.4282687,0.0043413616,0.039696164],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99338526,0.0030160178,0.00045697758,0.0013836427,0.0010496916,0.00070844183],"domain_scores_gemma":[0.9932715,0.0017194767,0.00014431051,0.0021015818,0.0014368971,0.0013261955],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0056548603,0.0038511555,0.0029958112,0.0015294113,0.003905708,0.0039351755,0.0039445017,0.004260152,0.016660266],"category_scores_gemma":[0.012957952,0.0010875325,0.001478554,0.001172856,0.0011173312,0.0059239324,0.013448314,0.004443492,0.021400452],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.005783125,0.0024268408,0.00450763,0.001716952,0.0005877432,0.0022023194,0.0027863246,0.0036992007,0.034047935,0.0039616385,0.6255841,0.31269616],"study_design_scores_gemma":[0.0034489883,0.0034042047,0.027572969,0.00055160606,0.0006781326,0.0038272294,0.010205147,0.15257615,0.09546336,0.02855038,0.6726928,0.0010291069],"about_ca_topic_score_codex":0.021998914,"about_ca_topic_score_gemma":0.034721166,"teacher_disagreement_score":0.021998914,"about_ca_system_score_codex":0.0018999405,"about_ca_system_score_gemma":0.0051760455,"threshold_uncertainty_score":0.055734158},"labels":[],"label_agreement":null},{"id":"W4416037652","doi":"10.18653/v1/2025.arabicnlp-main.10","title":"Lemmatizing Dialectal Arabic with Sequence-to-Sequence Models","year":2025,"lang":"","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"York University; New York University Abu Dhabi","keywords":"Arabic; Feature (linguistics); Nasalization; Natural language","score_opus":0.03910445339087106,"score_gpt":0.3050145840558795,"score_spread":0.26591013066500846,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416037652","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.038395308,0.00038843445,0.94777995,0.00046364413,0.000182495,0.0001343905,0.0018978296,0.006396245,0.004361752],"genre_scores_gemma":[0.46116126,0.00062724005,0.5168963,0.00052503287,0.00013658148,0.0002396922,0.010068256,0.0012170913,0.009128561],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9994149,0.0001938401,0.000049461974,0.00020334542,0.000099848585,0.000038647013],"domain_scores_gemma":[0.99810934,0.0010674337,0.00011227973,0.0002756709,0.0003928806,0.000042384818],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00091873534,0.0008538084,0.00040710834,0.0008339448,0.00043271808,0.0016002883,0.0007426469,0.0006454397,0.0050231633],"category_scores_gemma":[0.0040084384,0.00040292248,0.00091281463,0.00080472056,0.0005839248,0.0021210224,0.0010566904,0.0016486525,0.0057411348],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00063412345,0.00018564209,0.007972623,0.0005267507,0.00025090153,0.000984481,0.0013698165,0.390781,0.04305152,0.044112045,0.025243279,0.48488784],"study_design_scores_gemma":[0.000014182682,0.00003708607,0.00048036233,0.000023912,0.00002526143,0.00015751233,0.00018917944,0.95389867,0.0112972725,0.023615463,0.010240857,0.00002020685],"about_ca_topic_score_codex":0.0059915795,"about_ca_topic_score_gemma":0.014103393,"teacher_disagreement_score":0.0059915795,"about_ca_system_score_codex":0.0007141853,"about_ca_system_score_gemma":0.0013726988,"threshold_uncertainty_score":0.0168041},"labels":[],"label_agreement":null},{"id":"W4416037888","doi":"10.18653/v1/2025.disrpt-1.3","title":"CLaC at DISRPT 2025: Hierarchical Adapters for Cross-Framework Multi-lingual Discourse Relation Classification","year":2025,"lang":"","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Relation (database); Feature (linguistics); Field (mathematics); Identity (music)","score_opus":0.04473939019565893,"score_gpt":0.40433033078628805,"score_spread":0.3595909405906291,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416037888","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12950692,0.0067074765,0.42987913,0.007623688,0.008794366,0.0039340714,0.0763634,0.2941045,0.043086413],"genre_scores_gemma":[0.21077757,0.00080407615,0.48963678,0.0031840613,0.0008526021,0.003295461,0.24419668,0.014946928,0.032305885],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.98801047,0.0045169056,0.0005372971,0.0041680797,0.0017528256,0.0010144353],"domain_scores_gemma":[0.98453325,0.00523716,0.00042659606,0.0048601534,0.0032289154,0.0017139061],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0133954715,0.0053801285,0.0024486273,0.0032574278,0.0032309177,0.006679531,0.007691119,0.0059247725,0.030559057],"category_scores_gemma":[0.031591345,0.0018778989,0.0034063142,0.0023204088,0.0020042402,0.011199561,0.0115187485,0.009655697,0.028875524],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014887167,0.0012363136,0.0033254323,0.0014437094,0.00041764198,0.0006620286,0.0013671281,0.015179325,0.01641283,0.0119159315,0.557937,0.38861388],"study_design_scores_gemma":[0.001352176,0.0014763726,0.00690037,0.0005917893,0.00034347794,0.001332745,0.0025044223,0.49166736,0.043820806,0.036906544,0.41260678,0.00049711263],"about_ca_topic_score_codex":0.028369904,"about_ca_topic_score_gemma":0.049268797,"teacher_disagreement_score":0.030559057,"about_ca_system_score_codex":0.005011499,"about_ca_system_score_gemma":0.0064710085,"threshold_uncertainty_score":0.10223025},"labels":[],"label_agreement":null},{"id":"W4416078514","doi":"10.1145/3746252.3760808","title":"Approximating Gradient-Based Influence for Scalable Instruction Data Selection","year":2025,"lang":"","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; York University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Scalability; Selection (genetic algorithm); Fraction (chemistry); Regression; Sample (material); Model selection; Computation; Feature selection","score_opus":0.02190326696587988,"score_gpt":0.30609991062364217,"score_spread":0.2841966436577623,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416078514","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05222367,0.00092057826,0.92146075,0.00030399312,0.00013493292,0.00017268852,0.00064199464,0.021281524,0.0028599103],"genre_scores_gemma":[0.5442448,0.0003261732,0.44213855,0.00044111244,0.00015551654,0.00046439734,0.0036606824,0.003011985,0.00555677],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99894196,0.00022794699,0.00006545374,0.00030619494,0.00033765627,0.000120783465],"domain_scores_gemma":[0.99795437,0.0010926293,0.000115241586,0.0003260954,0.00042194166,0.000089686415],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011443562,0.0014197532,0.0013000155,0.0012375873,0.00053943926,0.0011283251,0.0014993359,0.00079390086,0.0035602837],"category_scores_gemma":[0.010110796,0.0005603558,0.00093407935,0.0010630576,0.0006995368,0.0018341299,0.0013402208,0.0016098514,0.0025908118],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00059360027,0.0002477942,0.0056636115,0.00029373606,0.000106799904,0.0001664686,0.00026108712,0.25541213,0.02646954,0.0065418202,0.017851649,0.6863918],"study_design_scores_gemma":[0.000023789511,0.000052666557,0.0005650709,0.000008338489,0.000011260735,0.00003177086,0.000023455557,0.9865861,0.0070045004,0.00357725,0.0021027212,0.000013093713],"about_ca_topic_score_codex":0.0077665346,"about_ca_topic_score_gemma":0.015241872,"teacher_disagreement_score":0.0077665346,"about_ca_system_score_codex":0.0009439555,"about_ca_system_score_gemma":0.0015949778,"threshold_uncertainty_score":0.015442669},"labels":[],"label_agreement":null},{"id":"W4416129820","doi":"10.7717/peerj-cs.3209","title":"Exploring prompting for dialectical machine translation: a focus on north Jordanian Arabic","year":2025,"lang":"en","type":"article","venue":"PeerJ Computer Science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"Jordan University of Science and Technology","keywords":"Dialectic; Focus (optics); Modern Standard Arabic; Arabic; Machine translation; Metric (unit); Process (computing)","score_opus":0.0685314119341759,"score_gpt":0.3054673907473636,"score_spread":0.23693597881318773,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416129820","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.90922886,0.006036244,0.048055433,0.001836943,0.0004904374,0.0006061967,0.007178043,0.010836817,0.015730977],"genre_scores_gemma":[0.84729666,0.0014871658,0.10668885,0.0010321869,0.00018708725,0.00047794954,0.03486397,0.0007585616,0.007207537],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9983353,0.00077944,0.00009902854,0.0005362397,0.00016277032,0.000087276945],"domain_scores_gemma":[0.99619764,0.002031239,0.00013012555,0.0008804387,0.00053238386,0.000228061],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022868381,0.0013930246,0.0010360344,0.0011508913,0.0013664138,0.0018508866,0.0013250086,0.0012893976,0.003497479],"category_scores_gemma":[0.00972335,0.00030456713,0.0007621203,0.0015925348,0.00082654745,0.0026419242,0.0021162904,0.0023678648,0.0029042263],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0034392674,0.0021657806,0.021882528,0.0025292858,0.0003613631,0.0017591728,0.0042954413,0.08060082,0.034599394,0.0042492133,0.053129356,0.79098827],"study_design_scores_gemma":[0.0010380924,0.0017837648,0.03944663,0.00055439054,0.00044165822,0.002490024,0.010086286,0.6969129,0.07725039,0.0197122,0.14990054,0.00038312512],"about_ca_topic_score_codex":0.010909997,"about_ca_topic_score_gemma":0.01907124,"teacher_disagreement_score":0.010909997,"about_ca_system_score_codex":0.001028889,"about_ca_system_score_gemma":0.0013463495,"threshold_uncertainty_score":0.021692991},"labels":[],"label_agreement":null},{"id":"W4416186693","doi":"10.48550/arxiv.2509.08105","title":"MERLIN: Multi-Stage Curriculum Alignment for Multilingual Encoder-LLM Integration in Cross-Lingual Reasoning","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Institut de Valorisation des Données; Canada First Research Excellence Fund","keywords":"Benchmark (surveying); Curriculum; Set (abstract data type); Data set; Training set; Computational linguistics; Language acquisition","score_opus":0.05660995828063751,"score_gpt":0.3846209979870951,"score_spread":0.32801103970645756,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416186693","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020025328,0.00076965994,0.8794284,0.00043120555,0.00021564895,0.00016982392,0.0013698223,0.091683,0.0059070773],"genre_scores_gemma":[0.20851666,0.0002844365,0.77133864,0.000491257,0.000096022646,0.00026452265,0.0064880783,0.0042854915,0.008234942],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99826366,0.0005818315,0.00010272864,0.0005979972,0.00029617254,0.00015765333],"domain_scores_gemma":[0.998478,0.0006007418,0.00007093934,0.0005059562,0.00025918186,0.00008527379],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002397562,0.002046314,0.0010534022,0.0016205944,0.0010747034,0.002053992,0.0034687154,0.0018402159,0.019917006],"category_scores_gemma":[0.0070509645,0.0013396448,0.0012758297,0.0014405595,0.00072207354,0.005660343,0.004403268,0.0035256643,0.00998744],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00039630587,0.00034576567,0.0017710204,0.0003684084,0.00023094856,0.00021016256,0.0003970678,0.04257741,0.011237831,0.013002025,0.04382105,0.885642],"study_design_scores_gemma":[0.00014750093,0.00016458433,0.0006924034,0.0000710687,0.000086935885,0.00018068659,0.0002724934,0.90255606,0.027301675,0.0464866,0.021982182,0.00005772678],"about_ca_topic_score_codex":0.0070838034,"about_ca_topic_score_gemma":0.023152415,"teacher_disagreement_score":0.019917006,"about_ca_system_score_codex":0.0013331147,"about_ca_system_score_gemma":0.0024622187,"threshold_uncertainty_score":0.06662905},"labels":[],"label_agreement":null},{"id":"W4416219021","doi":"10.48550/arxiv.2508.15212","title":"SparK: Query-Aware Unstructured Sparsity with Recoverable KV Cache Channel Pruning","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Strong; Institute for Catastrophic Loss Reduction","keywords":"Cache; Computation; Robustness (evolution); Pruning; Channel (broadcasting); Byte; Quantization (signal processing); SPARK (programming language); Inference","score_opus":0.02914784969433545,"score_gpt":0.26311013946853956,"score_spread":0.2339622897742041,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416219021","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.035194222,0.00065189856,0.7870973,0.0006211034,0.0003190969,0.00029002846,0.002646828,0.16998616,0.0031934062],"genre_scores_gemma":[0.3488569,0.00031829072,0.6272639,0.000776353,0.00020669396,0.0005419218,0.008086718,0.0087413555,0.0052078427],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99911076,0.00014415043,0.00006188998,0.0002676907,0.00028608146,0.00012940652],"domain_scores_gemma":[0.9980305,0.0008536251,0.000092071314,0.00056253065,0.0002732039,0.0001881166],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012513632,0.0015616786,0.0013174703,0.0007975211,0.00069399923,0.0013019477,0.004951633,0.0013257212,0.005884655],"category_scores_gemma":[0.007184391,0.00086457276,0.0012415319,0.0007931273,0.0010440592,0.0031179874,0.0033259857,0.0025411039,0.0034435454],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020681163,0.00072068133,0.0077618808,0.0008819873,0.0003590957,0.0008704537,0.0011423422,0.16255888,0.043007344,0.021577364,0.14877708,0.6102748],"study_design_scores_gemma":[0.00015530159,0.00007816259,0.00029638965,0.000015413967,0.000021606847,0.00010092848,0.000082566774,0.9742172,0.009318127,0.010476272,0.005208266,0.000029839439],"about_ca_topic_score_codex":0.010700817,"about_ca_topic_score_gemma":0.026877942,"teacher_disagreement_score":0.010700817,"about_ca_system_score_codex":0.00084653066,"about_ca_system_score_gemma":0.003110176,"threshold_uncertainty_score":0.02127707},"labels":[],"label_agreement":null},{"id":"W4416371877","doi":"10.2316/j.2026.206-1214","title":"FUZZY LOGIC-BASED ERROR DETECTION AND CORRECTION IN ENGLISH WRITING FOR LANGUAGE LEARNING. 85-97","year":2025,"lang":"en","type":"article","venue":"International Journal of Robotics and Automation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Fuzzy logic; Error detection and correction; Natural language; Fuzzy control system; Error analysis","score_opus":0.009513769704676007,"score_gpt":0.2914994519304855,"score_spread":0.2819856822258095,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416371877","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.41571325,0.0014513351,0.55832195,0.0010610052,0.0004091499,0.0005174732,0.0006433603,0.00466814,0.017214311],"genre_scores_gemma":[0.85025334,0.0002522691,0.14199482,0.00008254843,0.00002873237,0.00008414324,0.0003300234,0.00009725821,0.006876911],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.998922,0.00032974212,0.00009673797,0.00018792371,0.00037391734,0.000089791094],"domain_scores_gemma":[0.99644226,0.0016806833,0.00020300313,0.00025129176,0.0013114301,0.00011132643],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019373834,0.00043691357,0.0005419576,0.001173977,0.00047526974,0.0012934383,0.0008990077,0.0006023323,0.004276842],"category_scores_gemma":[0.009476972,0.0002070959,0.00041763936,0.00056101737,0.00037169343,0.0010904558,0.0004696652,0.0008211088,0.0013567248],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018215373,0.00045368876,0.00961481,0.00020099962,0.000088523506,0.00023325374,0.00037584957,0.01931135,0.032251157,0.0021216858,0.0045603258,0.9289669],"study_design_scores_gemma":[0.000119572556,0.0004410458,0.01898913,0.00009723655,0.00012596347,0.000313799,0.00046721997,0.8744763,0.09634853,0.0047258474,0.0038412386,0.000054119977],"about_ca_topic_score_codex":0.010834473,"about_ca_topic_score_gemma":0.012190762,"teacher_disagreement_score":0.010834473,"about_ca_system_score_codex":0.000854265,"about_ca_system_score_gemma":0.0012947829,"threshold_uncertainty_score":0.021542847},"labels":[],"label_agreement":null},{"id":"W4416374244","doi":"10.48550/arxiv.2510.04919","title":"Do LLMs Align with My Task? Evaluating Text-to-SQL via Dataset Alignment","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada","keywords":"Generalization; SQL; Training set; Selection (genetic algorithm); Natural language; Data modeling","score_opus":0.04374433755076232,"score_gpt":0.34421671627954536,"score_spread":0.3004723787287831,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416374244","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.82440114,0.0038441655,0.10848853,0.0032145572,0.0007491185,0.0006028183,0.008672487,0.042180013,0.007847031],"genre_scores_gemma":[0.8989899,0.00038733843,0.069154866,0.0011964716,0.0001277035,0.00027678133,0.02604783,0.0016885268,0.0021306458],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9914526,0.003826531,0.00062393857,0.0025652864,0.0011063347,0.00042531043],"domain_scores_gemma":[0.98017603,0.011289698,0.0009915861,0.0049693407,0.0017184848,0.00085489074],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012947221,0.0014552895,0.0010263216,0.0012593841,0.00084513956,0.0025098287,0.002245034,0.0022459999,0.0027310357],"category_scores_gemma":[0.05249271,0.00055932644,0.0010712526,0.0014244078,0.0011164768,0.0051448313,0.0025500427,0.0027941389,0.0033223727],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.003922578,0.0021248746,0.08893962,0.0015282349,0.0012347318,0.00057796424,0.0014072707,0.27847278,0.033479765,0.0036978207,0.07913024,0.5054841],"study_design_scores_gemma":[0.00038625897,0.0014360666,0.018801976,0.00010501555,0.000192944,0.00032530175,0.0010557864,0.9319315,0.024181146,0.009699041,0.011786628,0.00009824644],"about_ca_topic_score_codex":0.008397946,"about_ca_topic_score_gemma":0.010847088,"teacher_disagreement_score":0.012947221,"about_ca_system_score_codex":0.0013629269,"about_ca_system_score_gemma":0.0021063876,"threshold_uncertainty_score":0.068472266},"labels":[],"label_agreement":null},{"id":"W4416655590","doi":"10.1080/0907676x.2025.2590066","title":"Measuring lexical distance between parallel corpora: the case of AI-generated news translation","year":2025,"lang":"en","type":"article","venue":"Perspectives","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Institut de Valorisation des Données; Canada First Research Excellence Fund","keywords":"Translation (biology); Machine translation; Feature (linguistics); Measure (data warehouse); Lexical item","score_opus":0.042178125932252605,"score_gpt":0.3094155154243002,"score_spread":0.26723738949204756,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416655590","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5671891,0.002989261,0.4018571,0.0018240713,0.0002864752,0.00058346824,0.001123624,0.0009151915,0.023231672],"genre_scores_gemma":[0.66598403,0.00050753256,0.3289727,0.00010498632,0.0001018236,0.0003426797,0.0015175297,0.00028652913,0.002182072],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.98035413,0.01067685,0.0016546331,0.0024902846,0.004410608,0.00041360676],"domain_scores_gemma":[0.9194137,0.05654335,0.004874115,0.009062611,0.009583557,0.00052268454],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011345545,0.00066222675,0.0010641377,0.0059680915,0.0024366016,0.004415732,0.0017326223,0.0016646279,0.0017379931],"category_scores_gemma":[0.08774677,0.0006415781,0.0006030032,0.012301579,0.0030487045,0.006528333,0.0029782562,0.0014754877,0.00092848524],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016870837,0.0006986132,0.05346472,0.002185561,0.00049236335,0.00635165,0.026894813,0.061407678,0.024309881,0.13853388,0.006618479,0.6773553],"study_design_scores_gemma":[0.00026934402,0.0008687706,0.056500804,0.0005592681,0.00035440142,0.0061899875,0.019505672,0.5627391,0.0593121,0.23154654,0.061789066,0.00036492292],"about_ca_topic_score_codex":0.004200558,"about_ca_topic_score_gemma":0.004405372,"teacher_disagreement_score":0.011345545,"about_ca_system_score_codex":0.0019548482,"about_ca_system_score_gemma":0.0013586883,"threshold_uncertainty_score":0.06000173},"labels":[],"label_agreement":null},{"id":"W4416809269","doi":"10.1016/j.jml.2025.104705","title":"A predictive coding model for online sentence processing","year":2025,"lang":"en","type":"article","venue":"Journal of Memory and Language","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Division of Graduate Education; Office of Naval Research; National Science Foundation","keywords":"Predictive coding; Coding (social sciences); Sentence; Sentence processing; Information processing","score_opus":0.014590982264796607,"score_gpt":0.30681840577869596,"score_spread":0.29222742351389935,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416809269","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07388784,0.0002686659,0.9148312,0.0015306525,0.000091635906,0.00006778714,0.00036912053,0.0005304706,0.00842258],"genre_scores_gemma":[0.9166083,0.0003257573,0.07557746,0.00026217083,0.00015073206,0.00022104173,0.00030690394,0.00013118377,0.006416546],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99932504,0.00023727026,0.000034912227,0.00015789682,0.00014013934,0.000104760504],"domain_scores_gemma":[0.9926059,0.0054157106,0.0005882831,0.0005194042,0.00062412315,0.00024643866],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017787192,0.0005348588,0.0006476096,0.0012567976,0.00060682074,0.0019110115,0.0017984555,0.0011083537,0.006165044],"category_scores_gemma":[0.011199851,0.00041907243,0.001122707,0.00095430604,0.0019586056,0.004127142,0.0010123049,0.0018915541,0.0009469863],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022717919,0.00013547996,0.0035620003,0.00016755697,0.00009549913,0.00045135897,0.0012898876,0.21182139,0.0065396414,0.7210091,0.0032330519,0.051467974],"study_design_scores_gemma":[0.000009043144,0.000025370691,0.00047224978,0.000010989999,0.000016472835,0.000073510324,0.000040496572,0.78088015,0.00039463083,0.21766004,0.00039972836,0.000017270133],"about_ca_topic_score_codex":0.0040326696,"about_ca_topic_score_gemma":0.0026870817,"teacher_disagreement_score":0.006165044,"about_ca_system_score_codex":0.0012623792,"about_ca_system_score_gemma":0.0010334384,"threshold_uncertainty_score":0.02062416},"labels":[],"label_agreement":null},{"id":"W49270455","doi":"","title":"Using Unigram and Bigram Language Models for Monolingual and Cross-Language IR","year":2007,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Bigram; Computer science; Natural language processing; Artificial intelligence; Word (group theory); Search engine indexing; Machine translation; Cross-language information retrieval; Translation (biology); Character (mathematics); Speech recognition; Linguistics; Mathematics; Trigram","score_opus":0.031127130277414847,"score_gpt":0.36343266854436534,"score_spread":0.3323055382669505,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W49270455","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.27119607,0.008025695,0.69358015,0.0006032787,0.0003964211,0.00036221018,0.0007058073,0.009965812,0.015164597],"genre_scores_gemma":[0.7174289,0.0019159652,0.26851198,0.00029602647,0.00022351155,0.00028921867,0.0013083416,0.0006440549,0.009382006],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9969522,0.0020005773,0.00017321925,0.00036674412,0.00036727145,0.0001400266],"domain_scores_gemma":[0.99331665,0.0050133877,0.00022216103,0.00060151954,0.00070548453,0.00014074746],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0065165935,0.0014622256,0.0013190052,0.0027212494,0.0009276019,0.0021501551,0.0009017732,0.0011324335,0.0032820345],"category_scores_gemma":[0.01097731,0.00041004975,0.00097429677,0.0015189036,0.00043746427,0.0066424906,0.001315928,0.0012796334,0.0027729638],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0022219785,0.00073698466,0.0056229862,0.0008493849,0.00076572254,0.0002940917,0.0006874813,0.075193934,0.014808169,0.0075983903,0.00695177,0.8842692],"study_design_scores_gemma":[0.0001206908,0.00062132545,0.002328268,0.00005304612,0.00024371469,0.0003061605,0.000329671,0.96673167,0.011356475,0.013724245,0.0040525654,0.00013223283],"about_ca_topic_score_codex":0.0052013113,"about_ca_topic_score_gemma":0.006810311,"teacher_disagreement_score":0.0065165935,"about_ca_system_score_codex":0.0008533457,"about_ca_system_score_gemma":0.0009073828,"threshold_uncertainty_score":0.034463465},"labels":[],"label_agreement":null},{"id":"W52626029","doi":"10.1007/978-3-319-06483-3_2","title":"Rhetorical Figuration as a Metric in Text Summarization","year":2014,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Automatic summarization; Rhetorical question; Computer science; Metric (unit); Natural language processing; Artificial intelligence; Linguistics; Philosophy; Engineering","score_opus":0.013562769440007325,"score_gpt":0.2671132741178015,"score_spread":0.25355050467779416,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W52626029","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.33926213,0.013947947,0.59932095,0.0023236708,0.0009933999,0.0005672294,0.0054383576,0.006315425,0.031830836],"genre_scores_gemma":[0.8095704,0.0011211148,0.18036808,0.00010816223,0.00043147212,0.00032852322,0.004144663,0.0005304298,0.003397185],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9943311,0.0028642125,0.00057524216,0.00066482363,0.0013967573,0.00016775151],"domain_scores_gemma":[0.9726653,0.019067533,0.0025408126,0.0014770987,0.003713314,0.0005359166],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004007407,0.00066155934,0.00084415526,0.0061742095,0.0009516911,0.0037476593,0.0009634693,0.0010885913,0.004250799],"category_scores_gemma":[0.032285165,0.00033386264,0.0004606801,0.0052727736,0.0010331603,0.005654501,0.0016645732,0.0012011697,0.0017012578],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017347224,0.00021578603,0.012295318,0.0019950408,0.00030127724,0.00024990164,0.003767494,0.017192945,0.04306119,0.048150305,0.016418094,0.854618],"study_design_scores_gemma":[0.00028746136,0.0040902887,0.0682221,0.0010051737,0.000995215,0.0010897865,0.005889558,0.5709309,0.07946839,0.18866433,0.078904524,0.00045225656],"about_ca_topic_score_codex":0.0005388008,"about_ca_topic_score_gemma":0.00082337833,"teacher_disagreement_score":0.0061742095,"about_ca_system_score_codex":0.0008879658,"about_ca_system_score_gemma":0.0005709164,"threshold_uncertainty_score":0.021193504},"labels":[],"label_agreement":null},{"id":"W54100278","doi":"10.63317/42vnn94pj3ga","title":"A Trainable Tokenizer, solution for multilingual texts and compound expression tokenization","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Lexical analysis; Computer science; Task (project management); Natural language processing; Security token; Artificial intelligence; Character (mathematics); Expression (computer science); Alphanumeric; Representation (politics); Regular expression; Lemmatisation; Programming language","score_opus":0.02290123785870335,"score_gpt":0.2812507490096366,"score_spread":0.25834951115093324,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W54100278","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004010186,0.00011059292,0.98072654,0.0001395432,0.00012186471,0.000102867205,0.0007699788,0.012953202,0.0010652823],"genre_scores_gemma":[0.05554784,0.00017792208,0.9260871,0.00012596906,0.0000878019,0.00026392806,0.004026647,0.0021265352,0.011556239],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99852645,0.00026214347,0.00015616561,0.00057937077,0.0002947319,0.00018113939],"domain_scores_gemma":[0.9984793,0.00058658776,0.00009053135,0.00031705282,0.00044236102,0.00008421166],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014104767,0.0015329352,0.0013830866,0.0013428772,0.0014196848,0.0018624364,0.0028275643,0.001864375,0.017546995],"category_scores_gemma":[0.0041244817,0.0012021598,0.0011821757,0.0019096695,0.00089660136,0.003792931,0.0032389883,0.0031521493,0.011226174],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00065177865,0.00014993425,0.0006804483,0.000514331,0.00010687283,0.0005385252,0.0003383293,0.020969307,0.046953604,0.023184929,0.037030198,0.8688817],"study_design_scores_gemma":[0.00016494092,0.00018717724,0.0007667266,0.00006812202,0.00018512859,0.0009067098,0.0004505982,0.7318541,0.15827584,0.047311362,0.059717514,0.00011182995],"about_ca_topic_score_codex":0.004045208,"about_ca_topic_score_gemma":0.008532515,"teacher_disagreement_score":0.017546995,"about_ca_system_score_codex":0.0010665394,"about_ca_system_score_gemma":0.0032521728,"threshold_uncertainty_score":0.058700502},"labels":[],"label_agreement":null},{"id":"W54898593","doi":"","title":"Leveraging supplemental representations for sequential transduction","year":2012,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta; University of Toronto","funders":"","keywords":"Grapheme; Transliteration; Computer science; Transduction (biophysics); Artificial intelligence; Word error rate; Natural language processing; Word (group theory); Variety (cybernetics); Speech recognition; Machine learning; Mathematics; Engineering","score_opus":0.03825997597957435,"score_gpt":0.33805498247153726,"score_spread":0.2997950064919629,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W54898593","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021672186,0.00016909334,0.9531544,0.0002695109,0.0001663263,0.00013947747,0.0016690538,0.019232323,0.0035276231],"genre_scores_gemma":[0.28847292,0.0002177977,0.6942564,0.00019161665,0.00012325266,0.00046853116,0.009593592,0.0013417263,0.005334213],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.998672,0.0004890812,0.000106111554,0.00034851683,0.0003089199,0.00007536946],"domain_scores_gemma":[0.9954457,0.0016036829,0.00023548008,0.0017604791,0.000854512,0.00010016086],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012707623,0.0016704402,0.00080793153,0.0014868184,0.00052225945,0.0016891089,0.0015585166,0.0016503495,0.010487805],"category_scores_gemma":[0.007966934,0.000507451,0.000930211,0.0014229395,0.00056398136,0.003783534,0.002333075,0.0018866936,0.0061655403],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005265196,0.00043271304,0.0015177444,0.000618307,0.00010639203,0.00058680255,0.0005167405,0.061073195,0.09972494,0.027708253,0.02195334,0.7852351],"study_design_scores_gemma":[0.00005827321,0.00040352918,0.0007926218,0.00007027724,0.00008729827,0.00041687334,0.00015888897,0.85428727,0.054670937,0.06390624,0.025050474,0.00009731955],"about_ca_topic_score_codex":0.0011596306,"about_ca_topic_score_gemma":0.0029468013,"teacher_disagreement_score":0.010487805,"about_ca_system_score_codex":0.00043273624,"about_ca_system_score_gemma":0.0011714285,"threshold_uncertainty_score":0.03508514},"labels":[],"label_agreement":null},{"id":"W54968529","doi":"10.3115/1117822.1455625","title":"Automatic verb classification using multilingual resources","year":2001,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Classifier (UML); Verb; Linguistics","score_opus":0.04035144476570724,"score_gpt":0.32208635156483123,"score_spread":0.281734906799124,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W54968529","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.28504342,0.0016418839,0.66998005,0.0012715519,0.00023371665,0.0005654059,0.0052971332,0.012330375,0.02363645],"genre_scores_gemma":[0.58376,0.0005870653,0.38756594,0.0003509374,0.00018102735,0.00064190757,0.022300571,0.0010133673,0.0035990858],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99652797,0.001208333,0.00036600872,0.0009987141,0.0006686215,0.00023044598],"domain_scores_gemma":[0.9924253,0.0036559803,0.0005241919,0.0012175652,0.001961251,0.00021567682],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003181818,0.000981014,0.0013000594,0.0067345197,0.0015415575,0.0030590526,0.0014141614,0.0009028453,0.0038973505],"category_scores_gemma":[0.0124989245,0.0006082571,0.00070962444,0.0037486192,0.0007113995,0.006502347,0.0021518911,0.0011132042,0.0026349314],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00053354114,0.0007218708,0.01979837,0.00050237816,0.00017917484,0.00067240786,0.0010878448,0.006990339,0.050151464,0.012109208,0.01577313,0.89148045],"study_design_scores_gemma":[0.00031496354,0.00046422402,0.03663399,0.00037395675,0.00039094416,0.0015095364,0.0024267135,0.6771716,0.1009935,0.07764589,0.10173583,0.00033892525],"about_ca_topic_score_codex":0.0036214497,"about_ca_topic_score_gemma":0.007829383,"teacher_disagreement_score":0.0067345197,"about_ca_system_score_codex":0.00093780644,"about_ca_system_score_gemma":0.0014439876,"threshold_uncertainty_score":0.016827226},"labels":[],"label_agreement":null},{"id":"W55337924","doi":"10.7202/1030025ar","title":"SVG contre Flash : théorie et pratique","year":2015,"lang":"fr","type":"article","venue":"Documentation et bibliothèques","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Humanities; Scalable Vector Graphics; Philosophy; Art; Computer science","score_opus":0.02975808799410103,"score_gpt":0.3526500492058922,"score_spread":0.3228919612117912,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W55337924","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025506146,0.019224074,0.6159446,0.033010058,0.0013842604,0.00008778716,0.00056171155,0.001787338,0.30249402],"genre_scores_gemma":[0.629148,0.02035276,0.24555033,0.007982399,0.0019314018,0.00052553567,0.0010278543,0.0019256801,0.09155599],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9944667,0.0024284308,0.0003216128,0.00096195674,0.0015476639,0.00027359062],"domain_scores_gemma":[0.9874834,0.009064974,0.00031434928,0.0019615367,0.0010005938,0.00017518579],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0068904404,0.0006307777,0.0008616668,0.0027571975,0.0021724962,0.010103707,0.002070809,0.002674476,0.016434481],"category_scores_gemma":[0.016485963,0.0006874504,0.0010130154,0.003326038,0.014608955,0.02030441,0.004125189,0.0064641726,0.0031647438],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000012517281,0.0000040160276,0.00010641008,0.000065485045,0.000006061574,0.00004176866,0.0012482508,0.00023136343,0.00018215465,0.9822029,0.0021148382,0.013784283],"study_design_scores_gemma":[0.000010871019,0.000016116506,0.00019431794,0.00021627037,0.000011704015,0.00015980333,0.0014108297,0.0022013385,0.0008788362,0.8734966,0.12138217,0.000021084012],"about_ca_topic_score_codex":0.0065280083,"about_ca_topic_score_gemma":0.0038624958,"teacher_disagreement_score":0.016434481,"about_ca_system_score_codex":0.0046546613,"about_ca_system_score_gemma":0.0025633604,"threshold_uncertainty_score":0.054978848},"labels":[],"label_agreement":null},{"id":"W56023180","doi":"","title":"Proceedings of the Fourteenth Conference on Computational Natural Language Learning","year":2010,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":92,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Session (web analytics); Artificial intelligence; Task (project management); Natural language processing; Natural language; Scope (computer science); Perspective (graphical); Natural (archaeology); World Wide Web; Programming language; History","score_opus":0.007853802270773744,"score_gpt":0.2563531898917275,"score_spread":0.24849938762095375,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W56023180","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011980387,0.106051765,0.53236145,0.07107523,0.08805332,0.0019795522,0.018916732,0.012006331,0.1575753],"genre_scores_gemma":[0.076005,0.066229254,0.39057416,0.018158477,0.021793118,0.0028838841,0.093209215,0.006042762,0.3251041],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9936133,0.0025899117,0.00064617384,0.0012143918,0.0016197303,0.00031650282],"domain_scores_gemma":[0.986615,0.006385562,0.00034536058,0.002589856,0.0027118146,0.0013524599],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008408821,0.0016360538,0.002425536,0.0021703558,0.0013266747,0.007991309,0.003669532,0.0024348185,0.08923993],"category_scores_gemma":[0.020070968,0.00062889815,0.0015813201,0.0018063843,0.0027857623,0.008732988,0.004728726,0.0065084095,0.03156616],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025488113,0.00018025588,0.0008747997,0.0008099376,0.00022343341,0.00017066041,0.000266482,0.0015975096,0.0015829069,0.022160249,0.69028586,0.28159302],"study_design_scores_gemma":[0.000051482053,0.00007658881,0.00094021804,0.00038410025,0.00006473056,0.00023645838,0.00016916978,0.0079054,0.0010826985,0.052553907,0.9364835,0.00005189083],"about_ca_topic_score_codex":0.0024582762,"about_ca_topic_score_gemma":0.00438335,"teacher_disagreement_score":0.08923993,"about_ca_system_score_codex":0.0018533831,"about_ca_system_score_gemma":0.0039165514,"threshold_uncertainty_score":0.2985373},"labels":[],"label_agreement":null},{"id":"W562246665","doi":"","title":"HLT-NAACL 2003 : Human Language Technology conference of the North American Chapter of the Association for Computational Linguistics : proceedings of the main conference : May 27 to June 1, 2003, Edmonton, Alberta, Canada","year":2003,"lang":"en","type":"book","venue":"Association for Computational Linguistics eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":65,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Library science; Association (psychology); History; Media studies; Linguistics; Sociology; Computer science; Philosophy; Epistemology","score_opus":0.010121846143812641,"score_gpt":0.2450450150691456,"score_spread":0.23492316892533296,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W562246665","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0025156324,0.041180972,0.24342969,0.031303454,0.030191569,0.0011596528,0.04337791,0.046804123,0.56003696],"genre_scores_gemma":[0.004379882,0.016600115,0.05766368,0.0027550692,0.0016038361,0.00052560586,0.042670887,0.0075238813,0.8662769],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9986259,0.0002484969,0.00008442313,0.00023149316,0.0007358922,0.00007380418],"domain_scores_gemma":[0.9936799,0.0012358804,0.00012805458,0.00045453056,0.0039010758,0.000600582],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035106507,0.0015572506,0.0022307204,0.0023839308,0.0019006206,0.009379723,0.002580304,0.0018670086,0.13572603],"category_scores_gemma":[0.006021361,0.0010786025,0.0006289551,0.0038569116,0.0011568018,0.0064590783,0.0023722202,0.0037809124,0.14714655],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000019254183,0.000017380082,0.000048760376,0.00014375083,0.0000032776638,0.000013340672,0.00007494606,0.00011007388,0.00032758797,0.0027926245,0.9258975,0.070551485],"study_design_scores_gemma":[0.000011851979,0.000010152309,0.00022630215,0.00014836204,0.000010522576,0.00007428456,0.00009619033,0.0009871089,0.00057666196,0.003615305,0.9942252,0.000018129336],"about_ca_topic_score_codex":0.03638345,"about_ca_topic_score_gemma":0.07550711,"teacher_disagreement_score":0.96361655,"about_ca_system_score_codex":0.002926676,"about_ca_system_score_gemma":0.009094826,"threshold_uncertainty_score":0.4540488},"labels":[],"label_agreement":null},{"id":"W56834323","doi":"","title":"An architectural framework for natural language interfaces to agent systems.","year":2006,"lang":"en","type":"article","venue":"Computational intelligence","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Manitoba","funders":"","keywords":"Computer science; Natural language; Natural language user interface; Human–computer interaction; Artificial intelligence; Ambiguity; Universal Networking Language; Natural language processing; Programming language","score_opus":0.01855455023083028,"score_gpt":0.3387013832656849,"score_spread":0.3201468330348546,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W56834323","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0012752956,0.00079208316,0.98783445,0.0019421296,0.000091313574,0.00023299953,0.000084581654,0.0011540656,0.006593118],"genre_scores_gemma":[0.037672557,0.0009798664,0.95435286,0.0003626851,0.00008783394,0.00070015003,0.00040305493,0.00025211697,0.0051888074],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9949195,0.0021413008,0.00080819475,0.00063606066,0.0011428664,0.00035217087],"domain_scores_gemma":[0.9964057,0.0014671523,0.0002558831,0.000698925,0.0008095576,0.00036272896],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009626962,0.0013066778,0.00090390793,0.0029633585,0.0023880769,0.008011589,0.0041317306,0.0035498547,0.0063089514],"category_scores_gemma":[0.008195677,0.0016851921,0.0034083542,0.0021862618,0.0069849775,0.009656043,0.0046286704,0.0050816373,0.0020096686],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000020535004,0.000037908292,0.00017693781,0.00019448856,0.000033697725,0.00019290936,0.0011450354,0.0073531424,0.0012249798,0.9615982,0.003040877,0.024981251],"study_design_scores_gemma":[0.00006194391,0.00006858658,0.00019651969,0.00033034183,0.000078275916,0.00043464606,0.0006331126,0.091542594,0.0021529903,0.72973716,0.17470117,0.00006279925],"about_ca_topic_score_codex":0.010323221,"about_ca_topic_score_gemma":0.010676887,"teacher_disagreement_score":0.010323221,"about_ca_system_score_codex":0.004219934,"about_ca_system_score_gemma":0.0051930593,"threshold_uncertainty_score":0.050912917},"labels":[],"label_agreement":null},{"id":"W576119080","doi":"10.5206/entrehojas.v3i2.6142","title":"Spanish future of probability: teaching and learning","year":2014,"lang":"en","type":"article","venue":"Entrehojas Revista de Estudios Hispánicos","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":true,"ca_institutions":"University of Ottawa","funders":"","keywords":"Grammaticality; Meaning (existential); Computer science; Second-language acquisition; Process (computing); Control (management); Linguistics; Psychology; Feature (linguistics); Mathematics education; Language acquisition; Artificial intelligence; Grammar","score_opus":0.007036821164714807,"score_gpt":0.23823951795118456,"score_spread":0.23120269678646976,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W576119080","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8159287,0.0022723302,0.039680686,0.004762041,0.00011765752,0.00008305483,0.00008672122,0.00023589913,0.13683279],"genre_scores_gemma":[0.96479833,0.0019987156,0.0152503215,0.00013001697,0.000041033647,0.000041839852,0.00007386508,0.000025570018,0.017640168],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99953246,0.00019764579,0.000010187792,0.00007741292,0.00013745142,0.000044787015],"domain_scores_gemma":[0.9982374,0.0009042388,0.0001446739,0.00014309263,0.0002378564,0.00033273367],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011024536,0.00019560661,0.00018041997,0.00035727452,0.0005318297,0.0023511783,0.00048341043,0.0003941275,0.0049169012],"category_scores_gemma":[0.003944584,0.00010506226,0.00018042789,0.00028913192,0.0009155217,0.0017316343,0.00088606024,0.0006391287,0.0007904544],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002107068,0.0013487788,0.06207619,0.00039175298,0.000020344074,0.0009972339,0.02701612,0.0025315906,0.011880674,0.114834875,0.012055564,0.7666361],"study_design_scores_gemma":[0.00013190927,0.0015814907,0.13727482,0.0005922012,0.000078847654,0.003185045,0.04493326,0.012896072,0.026400149,0.18177131,0.5910629,0.000092046175],"about_ca_topic_score_codex":0.0024898967,"about_ca_topic_score_gemma":0.003664636,"teacher_disagreement_score":0.0049169012,"about_ca_system_score_codex":0.001060459,"about_ca_system_score_gemma":0.00221837,"threshold_uncertainty_score":0.016448677},"labels":[],"label_agreement":null},{"id":"W576925655","doi":"","title":"Annual Workshop on Formal Approaches to Slavic Languages : The Ottawa Meeting 2003","year":2004,"lang":"en","type":"book","venue":"Michigan Slavic Publications eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Slavic languages; Library science; History; Computer science; Linguistics; Classics; Philosophy","score_opus":0.03316693391524434,"score_gpt":0.26823101431095464,"score_spread":0.2350640803957103,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W576925655","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023339448,0.03850824,0.28997865,0.039174862,0.0105455145,0.00053693936,0.0064013703,0.007914258,0.5836007],"genre_scores_gemma":[0.064792156,0.014572784,0.08527541,0.001160347,0.0008517388,0.000256418,0.0056725885,0.0016781411,0.82574034],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9994019,0.0001372518,0.0000346118,0.00011367488,0.00019188412,0.000120566154],"domain_scores_gemma":[0.9987637,0.00025661927,0.000027909193,0.00018156589,0.00058491144,0.00018538743],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015877171,0.0010079157,0.0012639042,0.000979124,0.0027652148,0.005616432,0.0016933518,0.0012263515,0.07718855],"category_scores_gemma":[0.0025068305,0.000807497,0.00072398316,0.0018764373,0.0018558132,0.0048847054,0.001483599,0.0019284328,0.015773306],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018469495,0.000099887664,0.0006897864,0.0002619243,0.000021789907,0.00015338772,0.0017758432,0.001713869,0.0024748265,0.09237212,0.698899,0.20135292],"study_design_scores_gemma":[0.000029877734,0.000031072726,0.0011885487,0.00028289625,0.00003910558,0.00014464163,0.0010478063,0.0050552506,0.0017143044,0.046551652,0.9438699,0.00004499664],"about_ca_topic_score_codex":0.15752521,"about_ca_topic_score_gemma":0.38645974,"teacher_disagreement_score":0.15752521,"about_ca_system_score_codex":0.0055515333,"about_ca_system_score_gemma":0.008000591,"threshold_uncertainty_score":0.31321663},"labels":[],"label_agreement":null},{"id":"W58494562","doi":"","title":"UConcordia: CLaC Negation Focus Detection at *Sem 2012","year":2012,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":44,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Negation; Focus (optics); Scope (computer science); Task (project management); Computer science; Semantics (computer science); Annotation; Baseline (sea); Natural language processing; Artificial intelligence; Programming language; Political science; Engineering","score_opus":0.009836040982434329,"score_gpt":0.2517476975992264,"score_spread":0.24191165661679206,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W58494562","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.054223977,0.0033132466,0.046038114,0.0024104565,0.0024135746,0.0020710062,0.49778423,0.32366794,0.068077445],"genre_scores_gemma":[0.056779426,0.00026169265,0.073075056,0.0009285741,0.00019555251,0.0013090804,0.845123,0.0086486405,0.013678925],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99651873,0.0007564637,0.00024633444,0.0011126088,0.0009035122,0.0004622838],"domain_scores_gemma":[0.9953506,0.00076985685,0.00020439866,0.0016718343,0.0014212163,0.0005820294],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003451166,0.0052347626,0.0020020125,0.0052086734,0.0031216803,0.0036119388,0.0044813678,0.0038598455,0.030684492],"category_scores_gemma":[0.008043624,0.0012259731,0.0014713829,0.0027977445,0.001026788,0.0041042673,0.0051501426,0.002963894,0.03966158],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007537023,0.00029177385,0.0017257971,0.00058498996,0.00007985832,0.00035542855,0.00018212167,0.0008488621,0.004026045,0.0014340499,0.94055027,0.04916716],"study_design_scores_gemma":[0.0013211394,0.0004550647,0.018482273,0.0004855785,0.0001707466,0.0012761077,0.00078345265,0.076575376,0.044428308,0.012592072,0.8431044,0.00032548115],"about_ca_topic_score_codex":0.04129291,"about_ca_topic_score_gemma":0.09257456,"teacher_disagreement_score":0.04129291,"about_ca_system_score_codex":0.0025552681,"about_ca_system_score_gemma":0.0031650807,"threshold_uncertainty_score":0.10264987},"labels":[],"label_agreement":null},{"id":"W59585178","doi":"","title":"Proceedings of the conference on Human Language Technology and Empirical Methods in Natural Language Processing","year":2005,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":231,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Language technology; Computer science; Human language; Computational linguistics; Linguistics; Natural language; Artificial intelligence; Natural language processing; Comprehension approach","score_opus":0.03212528760022702,"score_gpt":0.40638603987494587,"score_spread":0.37426075227471883,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W59585178","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008638581,0.16322787,0.40125442,0.1488839,0.07094289,0.0012029363,0.0046962597,0.0024459322,0.19870727],"genre_scores_gemma":[0.08745792,0.09682527,0.38141644,0.023166286,0.04185366,0.0024765567,0.010682838,0.002992604,0.35312843],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.98484623,0.009132113,0.0008828907,0.0017514594,0.0030395584,0.00034775984],"domain_scores_gemma":[0.96125114,0.024901124,0.0007457624,0.0047686864,0.006037326,0.002296054],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.027058104,0.001017946,0.0016204347,0.0028009748,0.0016425889,0.009266915,0.0018489592,0.0027392467,0.051955875],"category_scores_gemma":[0.046010748,0.00074135006,0.0011775392,0.0024453902,0.005367933,0.0075113275,0.0040816064,0.006003789,0.012077498],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024475684,0.00014798394,0.001745458,0.00071706314,0.00018838115,0.00018388871,0.0011841928,0.00066514616,0.0013366622,0.09778519,0.5751151,0.32068625],"study_design_scores_gemma":[0.000041449177,0.000045293156,0.0016344218,0.0004478576,0.000044828146,0.00023838323,0.0003708107,0.0016350212,0.0006063615,0.052437253,0.9424552,0.000043009622],"about_ca_topic_score_codex":0.0045910366,"about_ca_topic_score_gemma":0.005836438,"teacher_disagreement_score":0.051955875,"about_ca_system_score_codex":0.0032915666,"about_ca_system_score_gemma":0.0062863533,"threshold_uncertainty_score":0.1738097},"labels":[],"label_agreement":null},{"id":"W600316455","doi":"","title":"A language modeling approach to the Text Retrieval Conference","year":2005,"lang":"de","type":"book-chapter","venue":"TNO Repository","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Université de Montréal","keywords":"Computer science; Information retrieval","score_opus":0.025517094562324556,"score_gpt":0.2542816972577927,"score_spread":0.22876460269546814,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W600316455","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005425699,0.010287206,0.8892408,0.0057552718,0.0012133754,0.00023835471,0.001235206,0.0024892238,0.08411485],"genre_scores_gemma":[0.21523486,0.011280282,0.53823644,0.0023275218,0.0019141809,0.00090170687,0.0044937977,0.0017716258,0.22383972],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9976981,0.0010086803,0.00016567032,0.00044168875,0.0005737354,0.00011202153],"domain_scores_gemma":[0.9970842,0.0018428736,0.00013248983,0.00034873546,0.0004880133,0.00010364178],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002549383,0.0009388788,0.0010248398,0.0026633157,0.0017749836,0.007102126,0.0027394707,0.002389295,0.018692099],"category_scores_gemma":[0.007985577,0.00074980856,0.0018118463,0.0039504976,0.0012290417,0.008993844,0.0015325485,0.0022134949,0.008745441],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016121008,0.0001519892,0.0004530602,0.00056876685,0.00015258057,0.00044349374,0.001136269,0.06045739,0.0030748094,0.66456825,0.06339845,0.20543371],"study_design_scores_gemma":[0.000042590324,0.00005777867,0.0003007338,0.00014931275,0.00007984413,0.00039349005,0.00038716182,0.3933053,0.003086438,0.41127735,0.19083968,0.00008037626],"about_ca_topic_score_codex":0.010342775,"about_ca_topic_score_gemma":0.009816651,"teacher_disagreement_score":0.018692099,"about_ca_system_score_codex":0.0025924833,"about_ca_system_score_gemma":0.0023425606,"threshold_uncertainty_score":0.06253135},"labels":[],"label_agreement":null},{"id":"W606180550","doi":"","title":"HLT-NAACL 2003 : Human Language Technology conference of the North American Chapter of the Association for Computational Linguistics: companion volume : short parers, student research workshop, demonstrations, tutorial abstracts : May 27 to June 1, 2003, Edmonton, Alberta, Canada","year":2003,"lang":"en","type":"book","venue":"Association for Computational Linguistics eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Volume (thermodynamics); Association (psychology); Library science; Computational linguistics; Computer science; Linguistics; Psychology; Artificial intelligence; Philosophy; Physics","score_opus":0.029312329665898845,"score_gpt":0.32405026646022594,"score_spread":0.2947379367943271,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W606180550","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00700731,0.05795234,0.36690444,0.049543977,0.063518435,0.0021277133,0.046770852,0.06422198,0.341953],"genre_scores_gemma":[0.012426467,0.028446188,0.12295247,0.0077363816,0.0043708556,0.0017006507,0.07001567,0.0124419825,0.7399094],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9978898,0.0006134768,0.00017763943,0.00038480034,0.00080385315,0.00013048017],"domain_scores_gemma":[0.99213153,0.0019226375,0.00017496743,0.0007787725,0.003978941,0.0010132554],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0069124093,0.0017689636,0.0028593673,0.0016899889,0.0020861072,0.009845867,0.0031719091,0.0023729987,0.11055403],"category_scores_gemma":[0.008419193,0.0011597644,0.0008656812,0.00223225,0.001684281,0.007998725,0.0033959893,0.004842316,0.107878864],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000053226857,0.000040919138,0.00009273806,0.00019810814,0.000008172979,0.000027432749,0.00012379722,0.0001416762,0.0006781546,0.0023766756,0.9287482,0.06751084],"study_design_scores_gemma":[0.000047409823,0.000036736023,0.00052069064,0.0003041232,0.000030155728,0.00014348922,0.00022907941,0.0024229537,0.0012885724,0.005765367,0.9891712,0.00004023067],"about_ca_topic_score_codex":0.026129197,"about_ca_topic_score_gemma":0.039822936,"teacher_disagreement_score":0.9738708,"about_ca_system_score_codex":0.0025399122,"about_ca_system_score_gemma":0.009963827,"threshold_uncertainty_score":0.3698401},"labels":[],"label_agreement":null},{"id":"W6149320","doi":"10.1016/s0022-5347(17)49961-7","title":"An extension of recursive descent parsing for Boolean grammars","year":2004,"lang":"en","type":"article","venue":"The Journal of Urology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Parsing expression grammar; Computer science; Programming language; Phrase structure grammar; Context-free grammar; Rule-based machine translation; Parsing; Correctness; Indexed grammar; Theoretical computer science; L-attributed grammar; Algorithm; Artificial intelligence","score_opus":0.017166421544003736,"score_gpt":0.2928868631224541,"score_spread":0.2757204415784504,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6149320","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0084516695,0.00015779636,0.9761771,0.0002499115,0.000061545805,0.00013473323,0.0005285405,0.011031599,0.0032071911],"genre_scores_gemma":[0.1636385,0.00026616702,0.8286107,0.00027915047,0.00008623815,0.00015384289,0.0018871196,0.001882053,0.0031962239],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99839574,0.0005472977,0.00017455107,0.0003701856,0.00034847396,0.0001637021],"domain_scores_gemma":[0.9961306,0.0026086648,0.00012733146,0.00066855823,0.0004061381,0.000058793892],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021128654,0.00083478994,0.0009140261,0.0011606452,0.00065708824,0.001615652,0.0020396584,0.0010918347,0.005816661],"category_scores_gemma":[0.0075158393,0.0007625613,0.0024988817,0.001155995,0.0012975576,0.0038212293,0.0021051862,0.001494224,0.0020775483],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00032741268,0.00022330949,0.0023821339,0.00074330234,0.00019598998,0.0009714601,0.0009609613,0.0951014,0.014424881,0.3238003,0.016983826,0.5438851],"study_design_scores_gemma":[0.00006948401,0.00009841015,0.0005631714,0.00008597381,0.00014338133,0.00039741865,0.0000778648,0.6591854,0.01021534,0.30463907,0.024444949,0.00007940517],"about_ca_topic_score_codex":0.007203623,"about_ca_topic_score_gemma":0.009089812,"teacher_disagreement_score":0.007203623,"about_ca_system_score_codex":0.0009684053,"about_ca_system_score_gemma":0.0012844433,"threshold_uncertainty_score":0.019458711},"labels":[],"label_agreement":null},{"id":"W622962318","doi":"10.71781/11078","title":"Recherche d'information translinguistique sur les documents en arabe","year":2008,"lang":"fr","type":"dissertation","venue":"Open MIND","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Natural language processing; Cross-language information retrieval; Artificial intelligence; Arabic; Context (archaeology); Machine translation; Information retrieval; Word (group theory); Linguistics","score_opus":0.1255534000594949,"score_gpt":0.3917661871829478,"score_spread":0.2662127871234529,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W622962318","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.29018968,0.04120298,0.49688014,0.024714075,0.0040259915,0.00058640906,0.022820061,0.008623565,0.11095704],"genre_scores_gemma":[0.48819962,0.023292243,0.29106382,0.0011240434,0.0012021217,0.00028162572,0.023936488,0.002103131,0.1687969],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9975356,0.00076024927,0.00012606554,0.0005538324,0.00083390815,0.00019034401],"domain_scores_gemma":[0.9934563,0.002762518,0.00018844502,0.00048401003,0.0028864252,0.00022226157],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037418015,0.0011735444,0.0016129345,0.004088008,0.0020328735,0.0072576604,0.0009231881,0.0014066879,0.024348455],"category_scores_gemma":[0.010891309,0.0006446682,0.0013324071,0.0034638925,0.0016428143,0.0041622776,0.0009240272,0.0020681364,0.0059055714],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001339402,0.00037798923,0.0065908195,0.0017632943,0.00029806033,0.0013900787,0.006823453,0.00992621,0.10533622,0.08824173,0.084279016,0.6936337],"study_design_scores_gemma":[0.00032762907,0.0005159178,0.038227558,0.0006845309,0.0004525761,0.0024739008,0.006882037,0.099794164,0.2011111,0.030074736,0.6191165,0.0003394248],"about_ca_topic_score_codex":0.2182933,"about_ca_topic_score_gemma":0.12832206,"teacher_disagreement_score":0.2182933,"about_ca_system_score_codex":0.004374803,"about_ca_system_score_gemma":0.00621524,"threshold_uncertainty_score":0.43404537},"labels":[],"label_agreement":null},{"id":"W627382014","doi":"10.29173/cais443","title":"Characteristics of Materials Used by English-Speaking Linguists in their Publications: A Citation Study of Literature Requirements , Citing Functions, and Citing Trends","year":2013,"lang":"en","type":"article","venue":"Proceedings of the Annual Conference of CAIS / Actes du congrès annuel de l ACSI","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Citation; Subject (documents); Computer science; Span (engineering); Linguistics; Citation analysis; Information retrieval; Library science; Engineering; Philosophy","score_opus":0.02336270329955263,"score_gpt":0.26477685788122496,"score_spread":0.2414141545816723,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W627382014","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99140066,0.0028547228,0.00044870292,0.00024055068,0.000024176996,0.000027074446,0.00094549905,0.000017719587,0.0040407977],"genre_scores_gemma":[0.9932273,0.0030261746,0.0012259753,0.000032945223,0.00007160431,0.000048215006,0.0012905701,0.000017271852,0.0010599081],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99386436,0.0012067478,0.0015407993,0.00037686262,0.0026787694,0.0003324239],"domain_scores_gemma":[0.8823705,0.061218947,0.02828799,0.0018771113,0.02469239,0.0015530406],"candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.0066731684,0.00018745734,0.0005968397,0.052362494,0.0018484929,0.0036430503,0.00056361384,0.0005608869,0.001341296],"category_scores_gemma":[0.071792014,0.00015208029,0.00038615306,0.07957393,0.0007594277,0.003202433,0.0012920457,0.00030553647,0.00058960065],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021591951,0.000080703394,0.8536556,0.0008480318,0.000113320275,0.0008612444,0.014418323,0.00022214133,0.0025490609,0.0015326977,0.0008754721,0.12462751],"study_design_scores_gemma":[0.000009886175,0.00012650032,0.96919334,0.00037448187,0.00016048338,0.0014200159,0.014255785,0.00054202985,0.0021973783,0.0007281861,0.010940437,0.000051544735],"about_ca_topic_score_codex":0.0053286757,"about_ca_topic_score_gemma":0.010411458,"teacher_disagreement_score":0.99332684,"about_ca_system_score_codex":0.0015574389,"about_ca_system_score_gemma":0.002620617,"threshold_uncertainty_score":0.035291553},"labels":[],"label_agreement":null},{"id":"W639870154","doi":"","title":"Proceedings of the 4th World Congress of African Linguistics New Brunswick 2003","year":2004,"lang":"en","type":"book","venue":"R. Köppe eBooks","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Political science; Applied linguistics; Linguistics; Philosophy","score_opus":0.01267238190085313,"score_gpt":0.24262469748706864,"score_spread":0.2299523155862155,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W639870154","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006039046,0.086608134,0.012110543,0.04962804,0.0339952,0.00032982216,0.005141422,0.00093194045,0.8052159],"genre_scores_gemma":[0.0071142614,0.024339626,0.004044,0.0013675002,0.0007722995,0.000120325676,0.002095657,0.00030833288,0.95983803],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99971205,0.000063191896,0.00002269547,0.0000640006,0.000078639685,0.00005938328],"domain_scores_gemma":[0.99915457,0.00018806536,0.000036252877,0.00011447061,0.0003374387,0.00016921],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015268051,0.00064350344,0.00069107837,0.0015579863,0.0021547535,0.004269718,0.0010335437,0.0010002246,0.19645804],"category_scores_gemma":[0.0020308797,0.00037169058,0.00031476968,0.0027246173,0.0011727092,0.0038078788,0.0019022243,0.0018174007,0.056639813],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000090702895,0.000035830588,0.0004092772,0.00025607907,0.000007546787,0.00015413699,0.0006391202,0.00011905026,0.0010961334,0.014087798,0.80622387,0.17688051],"study_design_scores_gemma":[0.0000028660968,0.0000026220168,0.00036662206,0.000123448,0.000003050243,0.000033727665,0.00026074567,0.000034498924,0.00015903109,0.0016073032,0.99740136,0.0000048577635],"about_ca_topic_score_codex":0.0744995,"about_ca_topic_score_gemma":0.25911155,"teacher_disagreement_score":0.19645804,"about_ca_system_score_codex":0.0034146444,"about_ca_system_score_gemma":0.0065187956,"threshold_uncertainty_score":0.6572176},"labels":[],"label_agreement":null},{"id":"W65082245","doi":"10.63317/5q2qm24pqyx4","title":"METIS-II: Machine Translation for Low Resource Languages","year":2006,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Machine translation; Natural language processing; Transfer-based machine translation; Translation (biology); Universal Networking Language; Artificial intelligence; Metis; Resource (disambiguation); Example-based machine translation; Machine translation software usability; Computer-assisted translation; Programming language; Natural language; Comprehension approach; World Wide Web","score_opus":0.008559174039101211,"score_gpt":0.26577983296532687,"score_spread":0.25722065892622564,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W65082245","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01239652,0.00075916527,0.9198223,0.00043778052,0.00037754554,0.00052472774,0.002435439,0.050483707,0.012762869],"genre_scores_gemma":[0.08359829,0.0004960629,0.88174343,0.0002968729,0.00018437495,0.000627736,0.009165519,0.005913972,0.01797372],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989975,0.00028020778,0.00006876565,0.00023746477,0.00032925516,0.00008688441],"domain_scores_gemma":[0.9991578,0.00028756657,0.00006182483,0.0002610079,0.00018355582,0.000048299793],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010796321,0.0011145275,0.0011650277,0.00117458,0.00081004295,0.002139957,0.0016294663,0.0008639873,0.014942086],"category_scores_gemma":[0.002184515,0.0006072425,0.0006727146,0.000872161,0.0006191759,0.0026237036,0.0015033645,0.0016291381,0.009796639],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001351306,0.0002416499,0.0011673624,0.0019589097,0.0002687903,0.0009619629,0.00089674897,0.0102168415,0.15936776,0.083295114,0.076517105,0.66375643],"study_design_scores_gemma":[0.00061847334,0.00090367714,0.0022809294,0.0003449406,0.00023999081,0.0024333105,0.0005739735,0.2574657,0.27161565,0.07118953,0.39213842,0.00019538747],"about_ca_topic_score_codex":0.0006737065,"about_ca_topic_score_gemma":0.0010528548,"teacher_disagreement_score":0.014942086,"about_ca_system_score_codex":0.00057654636,"about_ca_system_score_gemma":0.0008704806,"threshold_uncertainty_score":0.049986243},"labels":[],"label_agreement":null},{"id":"W658717802","doi":"10.20692/toyotakosenkiyo.kj00005389764","title":"古英語に現れる小節・結果構文 : York-Toronto-Helsinki Parsed Corpus of Old English Proseを検索して","year":2008,"lang":"ja","type":"article","venue":"豊田工業高等専門学校研究紀要","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Linguistics; History; Parsing; Corpus linguistics; Art; Artificial intelligence; Computer science; Philosophy","score_opus":0.020495999084574116,"score_gpt":0.24547486803793,"score_spread":0.2249788689533559,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W658717802","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7683164,0.008088084,0.003334593,0.0008291045,0.00021967426,0.00015470304,0.17568433,0.00039879928,0.042974353],"genre_scores_gemma":[0.8797569,0.0017599633,0.0034093354,0.00006238713,0.000038864153,0.00016180138,0.09654476,0.0002976684,0.017968269],"study_design_codex":"not_applicable","study_design_gemma":"observational","domain_scores_codex":[0.99961346,0.000075556156,0.000043755077,0.00011231572,0.00007681079,0.0000780541],"domain_scores_gemma":[0.9984332,0.00073818024,0.00013573056,0.00020024047,0.00033031488,0.00016243116],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006731536,0.00066808384,0.0003490602,0.002675757,0.0021891606,0.0018675347,0.00061868655,0.00051457586,0.009151285],"category_scores_gemma":[0.0022019374,0.0005281599,0.00016510228,0.004394051,0.0018586463,0.0009238423,0.0012842207,0.00080594275,0.0020324325],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.003986452,0.00031676792,0.08374831,0.006373306,0.00041820147,0.009077778,0.20616917,0.003776214,0.094975196,0.07039843,0.30256683,0.21819335],"study_design_scores_gemma":[0.00011742213,0.000068707486,0.61898434,0.00049324904,0.00019091547,0.0011638203,0.023574831,0.0012898949,0.01784122,0.0019653472,0.33414668,0.0001635173],"about_ca_topic_score_codex":0.55918556,"about_ca_topic_score_gemma":0.8125339,"teacher_disagreement_score":0.55918556,"about_ca_system_score_codex":0.0056781103,"about_ca_system_score_gemma":0.0056344764,"threshold_uncertainty_score":0.886821},"labels":[],"label_agreement":null},{"id":"W6891414117","doi":"10.3929/ethz-a-010137112","title":"Comparing ICP variants on real-world data sets: Open-source library and experimental protocol","year":2013,"lang":"en","type":"article","venue":"reroDoc Digital Library","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Fonds Québécois de la Recherche sur la Nature et les Technologies","keywords":"Protocol (science); Data collection; Identification (biology); Matching (statistics); Feature (linguistics)","score_opus":0.04447859576449631,"score_gpt":0.31075462819149846,"score_spread":0.26627603242700215,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6891414117","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.076871745,0.0005063416,0.78164566,0.00053658243,0.00043249925,0.027614485,0.04931238,0.048606407,0.014473843],"genre_scores_gemma":[0.074409716,0.00046658402,0.7296787,0.00034774301,0.000083843675,0.09133737,0.09240329,0.006331902,0.004940952],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9883279,0.00312567,0.0017846341,0.0017957404,0.0043576197,0.0006085181],"domain_scores_gemma":[0.9759854,0.0064499336,0.0009985666,0.007267606,0.008777014,0.00052143465],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011489998,0.0021450173,0.0014459195,0.0046834005,0.0018983126,0.0018155046,0.0046876743,0.0021515358,0.018012695],"category_scores_gemma":[0.028568929,0.00121327,0.0010454466,0.0059722676,0.0021552816,0.002909008,0.005282925,0.0020153662,0.009969069],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.006692003,0.0069086454,0.009908133,0.00646064,0.00071402127,0.0013230474,0.0022650168,0.07611382,0.08521169,0.018042598,0.16160876,0.6247516],"study_design_scores_gemma":[0.0027240487,0.004398015,0.038667187,0.0010036828,0.0004396731,0.0028873445,0.003021851,0.21069339,0.35376018,0.033427373,0.34761378,0.001363465],"about_ca_topic_score_codex":0.003311848,"about_ca_topic_score_gemma":0.0038508193,"teacher_disagreement_score":0.018012695,"about_ca_system_score_codex":0.0014003048,"about_ca_system_score_gemma":0.0034588103,"threshold_uncertainty_score":0.060765684},"labels":[],"label_agreement":null},{"id":"W6891761579","doi":"10.48448/gyjr-wh13","title":"Dolphin: A Challenging and Diverse Benchmark for Arabic NLG","year":2023,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Benchmark (surveying); Arabic; Feature (linguistics); Matching (statistics)","score_opus":0.023308095335309472,"score_gpt":0.30258791529591755,"score_spread":0.2792798199606081,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6891761579","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.22436363,0.011805807,0.12500879,0.009393361,0.004667368,0.00272868,0.216003,0.17358525,0.23244412],"genre_scores_gemma":[0.26264447,0.0023683992,0.21853845,0.0023307803,0.00046275335,0.0011567819,0.46641555,0.01053693,0.03554588],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99621296,0.0013518422,0.00047958968,0.0008344668,0.00085091824,0.0002702979],"domain_scores_gemma":[0.99226063,0.0037030324,0.00017941951,0.0013226966,0.0019494283,0.00058476464],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032620556,0.002720085,0.0012156247,0.0052550635,0.002892916,0.0035385836,0.0029187154,0.0037886,0.028437588],"category_scores_gemma":[0.016946357,0.00038133425,0.0010566028,0.0037331989,0.0015007028,0.005402143,0.004518057,0.0023179562,0.01881558],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012817568,0.0006109299,0.0023194004,0.0030063041,0.00019829311,0.0013768367,0.00090631616,0.02503001,0.008220386,0.014485191,0.6148122,0.32775238],"study_design_scores_gemma":[0.0011007898,0.00057180563,0.005203351,0.00076163316,0.00014760005,0.0024061657,0.0037239695,0.2865066,0.032960232,0.045561045,0.6208304,0.00022638342],"about_ca_topic_score_codex":0.018403534,"about_ca_topic_score_gemma":0.021692982,"teacher_disagreement_score":0.028437588,"about_ca_system_score_codex":0.002047062,"about_ca_system_score_gemma":0.0027161438,"threshold_uncertainty_score":0.095133185},"labels":[],"label_agreement":null},{"id":"W6891775876","doi":"10.48448/mcwj-g845","title":"DadmaTools: Natural Language Processing Toolkit for Persian Language","year":2022,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Optech (Canada)","funders":"","keywords":"Language identification; Natural language; Persian; Universal Networking Language; Natural (archaeology)","score_opus":0.011460592986319984,"score_gpt":0.30426533465323313,"score_spread":0.29280474166691317,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6891775876","genre_codex":"software","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":"software","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0031966765,0.0005817463,0.3361052,0.0004353531,0.00024377211,0.00026637712,0.05291011,0.5755676,0.030693166],"genre_scores_gemma":[0.067781925,0.0012626433,0.56918937,0.0012424835,0.00020632145,0.0011089706,0.18406072,0.09701438,0.078133136],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.999526,0.00009814345,0.00006453793,0.0001247507,0.0001424911,0.00004404094],"domain_scores_gemma":[0.9992291,0.00031251597,0.000051799776,0.00017138122,0.00018849758,0.00004677466],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006386166,0.0012491863,0.00051822467,0.002109491,0.0005354574,0.0019340503,0.0015509563,0.0005090228,0.09172918],"category_scores_gemma":[0.0025293303,0.00082230644,0.0009573874,0.0014086363,0.00038921926,0.003219095,0.0017053734,0.0016264074,0.057248544],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00041025755,0.000068008594,0.0009572697,0.001583973,0.000092584334,0.0005215243,0.0008010043,0.0014888559,0.012969926,0.030601243,0.6003187,0.35018674],"study_design_scores_gemma":[0.00013606506,0.00004807069,0.00085864245,0.00018778411,0.000050302806,0.0007304354,0.00032001777,0.017928937,0.021292945,0.03903182,0.9193254,0.00008956343],"about_ca_topic_score_codex":0.0032146946,"about_ca_topic_score_gemma":0.005978998,"teacher_disagreement_score":0.09172918,"about_ca_system_score_codex":0.00049487205,"about_ca_system_score_gemma":0.0012587437,"threshold_uncertainty_score":0.30686468},"labels":[],"label_agreement":null},{"id":"W6891826570","doi":"10.48448/2ahj-zr25","title":"SIDLR: Slot and Intent Detection Models for Low-Resource Language Varieties","year":2023,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Natural language; Variety (cybernetics); Feature (linguistics); Key (lock); Field (mathematics)","score_opus":0.01568839396889587,"score_gpt":0.27617723198237737,"score_spread":0.2604888380134815,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6891826570","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011908911,0.00056954677,0.86413413,0.0006368553,0.00031882027,0.0002293323,0.011671346,0.10412098,0.006410045],"genre_scores_gemma":[0.20785424,0.00054406864,0.71794784,0.00075428473,0.00027691244,0.0004648252,0.04490901,0.008718828,0.018530006],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99822456,0.0003464373,0.000122006895,0.0006559555,0.00050606736,0.0001449914],"domain_scores_gemma":[0.99666166,0.0014007147,0.00015727099,0.0009875982,0.00067114795,0.00012155997],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021661187,0.002010638,0.0013077155,0.0022723381,0.0010258529,0.003455459,0.003643393,0.0018497524,0.021000229],"category_scores_gemma":[0.007824828,0.0009936814,0.0022445708,0.0017548891,0.00090467476,0.006257926,0.0032675238,0.0031814794,0.015463617],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010818032,0.00050672836,0.0064427922,0.0008701681,0.00026075533,0.000499493,0.00054875296,0.041109875,0.018470569,0.049207456,0.2176338,0.6633678],"study_design_scores_gemma":[0.00007712679,0.00013440687,0.0011283641,0.000081540624,0.00007488156,0.00036702887,0.0002134498,0.8496053,0.016294714,0.09060289,0.041324258,0.00009604527],"about_ca_topic_score_codex":0.008695447,"about_ca_topic_score_gemma":0.011037832,"teacher_disagreement_score":0.021000229,"about_ca_system_score_codex":0.0012119971,"about_ca_system_score_gemma":0.0015832832,"threshold_uncertainty_score":0.070252776},"labels":[],"label_agreement":null},{"id":"W6892676523","doi":"10.5281/zenodo.11989964","title":"canadian fundamentals of nursing 6th edition pdf","year":2024,"lang":"en","type":"other","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Nurse education; Nursing research; Citation; Scope (computer science); Team nursing; Health care; Download; Foundation (evidence)","score_opus":0.024414400134044435,"score_gpt":0.2726210219527874,"score_spread":0.24820662181874298,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6892676523","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00018404576,0.0016683831,0.00056062656,0.0031021135,0.0053611244,0.00021236006,0.004247981,0.0009511665,0.9837123],"genre_scores_gemma":[0.0007261482,0.0017852302,0.00074196735,0.00064538344,0.00025879836,0.00004959067,0.001617398,0.00028960695,0.9938858],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9965743,0.00010498391,0.00010442811,0.00016638525,0.0026077994,0.00044212438],"domain_scores_gemma":[0.9891695,0.00022675675,0.00011068814,0.00032031426,0.007813319,0.0023594706],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.0016952576,0.0015534312,0.0012683112,0.004006547,0.0055310293,0.008275068,0.0026887308,0.002325842,0.73074883],"category_scores_gemma":[0.0075932285,0.00088991254,0.0009820795,0.0055097626,0.0009743417,0.0032746957,0.0037563555,0.002828235,0.6172351],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000039890238,0.0000071048717,0.000026498366,0.0000468053,3.565214e-7,0.000009881652,0.000020250227,0.00001983022,0.00003396813,0.0006409371,0.9757,0.023490384],"study_design_scores_gemma":[0.0000014829552,0.0000026855512,0.00020918228,0.00005673324,5.564287e-7,0.000016165824,0.00005159933,0.0000173674,0.00002524463,0.00017271658,0.99944025,0.0000059545378],"about_ca_topic_score_codex":0.5142884,"about_ca_topic_score_gemma":0.7431175,"teacher_disagreement_score":0.5142884,"about_ca_system_score_codex":0.018505856,"about_ca_system_score_gemma":0.048396423,"threshold_uncertainty_score":0.97714406},"labels":[],"label_agreement":null},{"id":"W6893716360","doi":"10.5281/zenodo.4290719","title":"SSHOC Considerations for the Vocabulary Platforms - CLARIN: main requirements and best practices","year":2020,"lang":"en","type":"article","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canarie","funders":"European Commission","keywords":"Best practice; Vocabulary","score_opus":0.1291573733617981,"score_gpt":0.3159389149387414,"score_spread":0.18678154157694327,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6893716360","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015469523,0.004180712,0.32627895,0.20134313,0.015417299,0.0059327153,0.009598533,0.03290184,0.38887727],"genre_scores_gemma":[0.07982478,0.0032127008,0.47006312,0.02780612,0.0065131565,0.0056029214,0.027820943,0.031945094,0.34721115],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9608039,0.011936632,0.005202903,0.003299313,0.014830443,0.003926721],"domain_scores_gemma":[0.79512006,0.046365693,0.0049360823,0.050166383,0.08960272,0.013809144],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.05659317,0.0014686141,0.0020418814,0.004705687,0.006239145,0.026793836,0.0077038007,0.010722069,0.11728686],"category_scores_gemma":[0.12860575,0.0022216465,0.002034173,0.0036593385,0.0049464083,0.04717555,0.012867156,0.00878634,0.11505853],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008597472,0.00035889685,0.0019489154,0.0014257113,0.00003899327,0.00046733682,0.003676739,0.0008269764,0.014141003,0.17116661,0.6218398,0.18324922],"study_design_scores_gemma":[0.00009929674,0.00015321889,0.0009906145,0.0013377985,0.00003261388,0.00041319503,0.0022391905,0.002330589,0.004519557,0.03176986,0.95599735,0.000116879484],"about_ca_topic_score_codex":0.019385885,"about_ca_topic_score_gemma":0.019467644,"teacher_disagreement_score":0.11728686,"about_ca_system_score_codex":0.005025297,"about_ca_system_score_gemma":0.015911961,"threshold_uncertainty_score":0.39236367},"labels":[],"label_agreement":null},{"id":"W6894274261","doi":"10.5281/zenodo.8100411","title":"Short answer assessment: Establishing links between research strands","year":2012,"lang":"en","type":"article","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Association (psychology); Focus (optics); Key (lock)","score_opus":0.07268730975677437,"score_gpt":0.339287530965699,"score_spread":0.2666002212089247,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6894274261","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1414937,0.011176676,0.6543441,0.03081606,0.004871421,0.014185949,0.023083309,0.005489254,0.11453954],"genre_scores_gemma":[0.33964434,0.0030485888,0.5951358,0.0027911225,0.0010197167,0.022322115,0.022816556,0.001007376,0.012214386],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9081978,0.050501537,0.014951446,0.007584246,0.01738786,0.0013770704],"domain_scores_gemma":[0.36658686,0.3939484,0.04544771,0.025424244,0.15845203,0.010140725],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.10079321,0.0013655007,0.0017535716,0.033958055,0.0052637127,0.013153536,0.0036048037,0.0037340594,0.026550828],"category_scores_gemma":[0.5324027,0.0010965588,0.0013365084,0.018982716,0.0031140447,0.024345951,0.022633547,0.003646737,0.007736951],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016526306,0.00039507,0.05004611,0.012601602,0.0006100644,0.00065987854,0.07261697,0.0012079702,0.004396546,0.09126214,0.108245574,0.6563054],"study_design_scores_gemma":[0.0008320678,0.0006731267,0.035013728,0.010305146,0.0011222407,0.0005523796,0.06684332,0.015012521,0.005272366,0.44485408,0.4189708,0.0005481598],"about_ca_topic_score_codex":0.0022534183,"about_ca_topic_score_gemma":0.003742143,"teacher_disagreement_score":0.10079321,"about_ca_system_score_codex":0.0035007712,"about_ca_system_score_gemma":0.009941946,"threshold_uncertainty_score":0.53305185},"labels":[],"label_agreement":null},{"id":"W6901649024","doi":"10.60692/2zfdb-ba138","title":"On Orthogonality Constraints for Transformers","year":2021,"lang":"en","type":"article","venue":"Greater South Information System","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Orthogonality; Transformer; Joint (building); Natural language; Algebra over a field; Matrix algebra","score_opus":0.02878086650117915,"score_gpt":0.245701246738939,"score_spread":0.21692038023775986,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6901649024","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.030805573,0.002243678,0.8176918,0.006846021,0.001233166,0.00031881203,0.0022214362,0.0013595618,0.13728],"genre_scores_gemma":[0.55018294,0.0038738707,0.38888216,0.003832871,0.0018046426,0.0005358,0.0056434786,0.0019399978,0.043304287],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.994779,0.0016515276,0.00052295055,0.0010014361,0.0012905459,0.0007544822],"domain_scores_gemma":[0.99014133,0.0059607243,0.00040308683,0.0014685938,0.0016856695,0.00034068138],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0052867304,0.0010872916,0.001180015,0.0021408969,0.002594284,0.0040363157,0.0019201967,0.001576332,0.025815139],"category_scores_gemma":[0.017717991,0.0013753932,0.0020033675,0.0030592112,0.0033197275,0.022176243,0.0065168412,0.0046742493,0.003682186],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000049265487,0.000026383248,0.00017635735,0.000066808876,0.000014332317,0.00007966511,0.000117484364,0.001394864,0.00038400872,0.97232413,0.008699218,0.016667524],"study_design_scores_gemma":[0.000021754684,0.0000103303955,0.00006019128,0.000034820063,0.000011792225,0.00007318927,0.00007243724,0.004689405,0.000496025,0.9797881,0.0147292325,0.000012687593],"about_ca_topic_score_codex":0.0041451883,"about_ca_topic_score_gemma":0.0070135854,"teacher_disagreement_score":0.025815139,"about_ca_system_score_codex":0.002019038,"about_ca_system_score_gemma":0.0021675827,"threshold_uncertainty_score":0.086360216},"labels":[],"label_agreement":null},{"id":"W6901677292","doi":"10.60692/0mms5-bz672","title":"Unsupervised Dependency Graph Network","year":2022,"lang":"en","type":"article","venue":"Greater South Information System","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Dependency (UML); Dependency grammar; Parsing; Dependency graph; Sentence; Graph","score_opus":0.018807826920340016,"score_gpt":0.20681007101294566,"score_spread":0.18800224409260563,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6901677292","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.044257317,0.000364359,0.9425566,0.0005097232,0.00006256424,0.00016013163,0.0021204348,0.003029067,0.00693971],"genre_scores_gemma":[0.57140696,0.0007237212,0.39833218,0.0006446702,0.00012706651,0.0005191302,0.011272076,0.0009049976,0.016069245],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992385,0.00022740227,0.00003099673,0.00032220574,0.0001229899,0.00005795067],"domain_scores_gemma":[0.9981292,0.0010034657,0.0001571254,0.00030961976,0.00033787076,0.00006269173],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007087407,0.00077302754,0.0005146038,0.0015155496,0.00068777706,0.0005682681,0.0014288899,0.0009759375,0.0048593264],"category_scores_gemma":[0.003821734,0.00050877396,0.0009204171,0.0014137924,0.0006493745,0.0021783072,0.0010093782,0.0010129528,0.0011409156],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037490434,0.0002319837,0.0068496447,0.0004888087,0.00028833954,0.00056005997,0.000500353,0.4346743,0.02462971,0.12824339,0.035516653,0.3676419],"study_design_scores_gemma":[0.000011669029,0.000030483026,0.0014210393,0.000016098133,0.000035875677,0.00011365062,0.000023570963,0.9394469,0.0042140055,0.04783287,0.006836459,0.000017403192],"about_ca_topic_score_codex":0.0070899944,"about_ca_topic_score_gemma":0.013979426,"teacher_disagreement_score":0.0070899944,"about_ca_system_score_codex":0.0012742238,"about_ca_system_score_gemma":0.0012764491,"threshold_uncertainty_score":0.016256094},"labels":[],"label_agreement":null},{"id":"W6901753181","doi":"10.60692/r5cjj-b8505","title":"Proceedings of the Fifth Named Entity Workshop","year":2015,"lang":"en","type":"article","venue":"Greater South Information System","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Transliteration; Named-entity recognition; Named entity; Task (project management); Conditional random field; Entity linking; Scope (computer science); Deep learning","score_opus":0.03273616498747871,"score_gpt":0.22746386287599712,"score_spread":0.1947276978885184,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6901753181","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024964802,0.027096083,0.17032275,0.089762025,0.27237397,0.002105025,0.053404354,0.01623629,0.34373462],"genre_scores_gemma":[0.0552422,0.009276357,0.08048779,0.012231229,0.026071224,0.0015830793,0.134134,0.0061065927,0.67486745],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99413246,0.0019154425,0.0005630456,0.0011321191,0.0016336141,0.00062331196],"domain_scores_gemma":[0.9860372,0.0031314562,0.00040750517,0.0026351882,0.005103591,0.0026850197],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012040056,0.0016032036,0.0014228595,0.0020646427,0.002832663,0.009295194,0.003543662,0.0030383868,0.09070351],"category_scores_gemma":[0.011843368,0.0006274816,0.0021315708,0.0019332129,0.0009768254,0.009597824,0.006186933,0.0038606636,0.06571032],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028224196,0.00013008504,0.0003155806,0.000300713,0.000039559633,0.00025992468,0.00023249515,0.00043297347,0.0013638667,0.0053783916,0.91791844,0.07334563],"study_design_scores_gemma":[0.000026842494,0.00003647064,0.00042928456,0.00010378249,0.000020902542,0.00012653893,0.00023535395,0.00072972523,0.0015710096,0.00277896,0.99391586,0.000025425281],"about_ca_topic_score_codex":0.0034950054,"about_ca_topic_score_gemma":0.00641228,"teacher_disagreement_score":0.09070351,"about_ca_system_score_codex":0.002433899,"about_ca_system_score_gemma":0.0033960312,"threshold_uncertainty_score":0.30343348},"labels":[],"label_agreement":null},{"id":"W6906437178","doi":"10.17613/0qqep-nes88","title":"Denotation Ambiguity Scoring for Panlingual Lexical Translation Inference","year":2017,"lang":"en","type":"article","venue":"Knowledge Commons (Lakehead University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Ambiguity; Inference; Machine translation; Semantic feature; Lexicographical order; Ranking (information retrieval); Translation (biology); Probabilistic logic; Denotation (semiotics)","score_opus":0.07762471666124136,"score_gpt":0.33051030585583807,"score_spread":0.2528855891945967,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6906437178","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02713707,0.00013527043,0.9683176,0.00016638672,0.000037938593,0.00009751194,0.00019374893,0.001193512,0.002720906],"genre_scores_gemma":[0.39092302,0.00010130985,0.60664225,0.000079366204,0.000047823134,0.00018840891,0.0006309795,0.00029740055,0.0010894428],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99248815,0.003975538,0.00073336397,0.0008551033,0.0017191465,0.0002287581],"domain_scores_gemma":[0.9785791,0.014009585,0.0012455774,0.0034434006,0.0024435467,0.0002787592],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009354313,0.00083065033,0.00096030615,0.0039646253,0.0013818839,0.0038089463,0.0016317018,0.0011512685,0.0055402457],"category_scores_gemma":[0.055739913,0.00054818054,0.00076006044,0.0032196841,0.0018944534,0.0063832863,0.0040542264,0.002186057,0.0013249335],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00059727737,0.00026573523,0.014305041,0.00046419137,0.00015906016,0.00027582148,0.0016654998,0.048313674,0.010933391,0.25646442,0.0051880158,0.6613678],"study_design_scores_gemma":[0.00005694974,0.00015129987,0.004571639,0.00008522469,0.00005325854,0.00029130068,0.0006013193,0.64896774,0.011283875,0.3277442,0.0060816323,0.00011149623],"about_ca_topic_score_codex":0.0017988392,"about_ca_topic_score_gemma":0.0031399406,"teacher_disagreement_score":0.009354313,"about_ca_system_score_codex":0.0016129061,"about_ca_system_score_gemma":0.001627429,"threshold_uncertainty_score":0.0494709},"labels":[],"label_agreement":null},{"id":"W6907762920","doi":"10.25316/ir-11438","title":"The Nanaimo Free Press [Wednesday, June 20, 1877]","year":2019,"lang":"en","type":"other","venue":"VIUSpace (Vancouver Island University Library)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"","score_opus":0.005734560369384432,"score_gpt":0.18962420357236323,"score_spread":0.1838896432029788,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6907762920","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0007450314,0.0149958925,0.00021034044,0.003110404,0.0046672276,0.00002576281,0.0029710846,0.00016276249,0.97311157],"genre_scores_gemma":[0.0013912625,0.0021274076,0.000074105265,0.00012530627,0.00015247246,0.000008828087,0.00031581073,0.00006139226,0.99574345],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99963653,0.000027792803,0.000014377783,0.0000671062,0.00018667724,0.00006739699],"domain_scores_gemma":[0.9997614,0.000034477882,0.000014460813,0.000021967579,0.000119161516,0.00004856173],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00030835246,0.0009254849,0.0006000575,0.002462993,0.004371139,0.0054824897,0.00078353845,0.001384703,0.27437428],"category_scores_gemma":[0.0013517097,0.00044238966,0.00031496512,0.004258534,0.00094116933,0.0018794127,0.0013711988,0.0022106909,0.10373396],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000026214642,0.0000063988905,0.00018817442,0.00009828471,0.0000042841266,0.000085751046,0.00017042916,0.00005143348,0.00009214689,0.016939553,0.94169325,0.040644098],"study_design_scores_gemma":[0.0000010763663,0.0000014615188,0.00031946323,0.00004368038,7.947754e-7,0.000014398519,0.00004336456,0.000009487913,0.000030091796,0.00042553907,0.9991086,0.0000020959503],"about_ca_topic_score_codex":0.14677005,"about_ca_topic_score_gemma":0.52676016,"teacher_disagreement_score":0.85322994,"about_ca_system_score_codex":0.005028819,"about_ca_system_score_gemma":0.0032889817,"threshold_uncertainty_score":0.9178734},"labels":[],"label_agreement":null},{"id":"W6910112932","doi":"10.3897/bdj.10.e76050.figure11","title":"Figure 11 from: Simon ADF, Adamczyk EM, Basman A, Chu JWF, Gartner HN, Fletcher K, Gibbs CJ, Gibbs DM, Gilmore SR, Harbo RM, Harris LH, Humphrey E, Lamb A, Lambert P, McDaniel N, Scott J, Starzomski BM (2022) Toward an atlas of Salish Sea biodiversity: the flora and fauna of Galiano Island, British Columbia, Canada. Part I. Marine zoology. Biodiversity Data Journal 10: e76050. https://doi.org/10.3897/BDJ.10.e76050","year":2022,"lang":"en","type":"other","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Nucleofection; Gestational period; TSG101; Liquation; Diafiltration; Dysgeusia; Emperipolesis; Triacetin; Hyporeflexia","score_opus":0.026938408809048974,"score_gpt":0.22391289023928965,"score_spread":0.19697448143024068,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6910112932","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00047485088,0.00028569248,0.0013778864,0.0005715073,0.0011788155,0.000113124515,0.05496945,0.003728884,0.9372998],"genre_scores_gemma":[0.0035706423,0.00048700254,0.0024444119,0.00023418185,0.00015483673,0.00007385413,0.042728007,0.0037846935,0.9465224],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9998117,0.000008660717,0.0000070732904,0.000031525396,0.00011472607,0.000026358499],"domain_scores_gemma":[0.9995764,0.000054833614,0.0000142484905,0.000057130714,0.00019539775,0.0001018673],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.00016301409,0.0010148968,0.0005210452,0.0017780681,0.0012363561,0.0024402752,0.000880303,0.00075343036,0.8894352],"category_scores_gemma":[0.0011574776,0.0003774693,0.0006064799,0.0027648795,0.00049056456,0.0016591595,0.001537346,0.0011038366,0.77114147],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000009664502,0.0000049719565,0.000090004956,0.00004736483,0.0000010058557,0.000031482192,0.000026819174,0.00003179669,0.00009421543,0.00031668638,0.9845993,0.014746695],"study_design_scores_gemma":[0.0000048138304,0.0000021810451,0.00079280406,0.000047590023,0.0000014342354,0.000053639076,0.000040142524,0.00004113749,0.00009598754,0.00020760656,0.99870884,0.0000038201006],"about_ca_topic_score_codex":0.033097476,"about_ca_topic_score_gemma":0.083817735,"teacher_disagreement_score":0.9669025,"about_ca_system_score_codex":0.0010604868,"about_ca_system_score_gemma":0.0011898845,"threshold_uncertainty_score":0.15770721},"labels":[],"label_agreement":null},{"id":"W6911856830","doi":"10.5281/zenodo.14597581","title":"Lodgepole pine flowering phenology brms model objects","year":2025,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Phenology; Pinus contorta; Object (grammar); Pollen; Code (set theory); Pinus <genus>","score_opus":0.021331979494401833,"score_gpt":0.26586028486649166,"score_spread":0.24452830537208983,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6911856830","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0007282572,0.00008786473,0.00020371712,0.0000359303,0.00002766712,0.000013005786,0.99716896,0.0012339834,0.00050049974],"genre_scores_gemma":[0.0009790874,0.000030395868,0.00037353608,0.000020743842,0.0000040556156,0.00004538405,0.9980995,0.00007894011,0.00036834937],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9995395,0.000042114676,0.000033022596,0.00023266819,0.000085611144,0.000067188565],"domain_scores_gemma":[0.99942774,0.00011454582,0.00006453435,0.00019089266,0.00013458852,0.00006763806],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00048763744,0.002864812,0.0011907818,0.0015781876,0.00055905245,0.0012811334,0.0026085838,0.0016067099,0.03007515],"category_scores_gemma":[0.0015921975,0.0007540948,0.0019610997,0.00240552,0.0002752269,0.000782857,0.0009481677,0.0016576461,0.05320647],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013610891,0.00005185678,0.0026851667,0.00071270106,0.00009066272,0.000050544095,0.000028168992,0.0024348877,0.00074888393,0.00037527294,0.98778486,0.004900853],"study_design_scores_gemma":[0.0005996786,0.00007189458,0.02525164,0.00027001265,0.00012520187,0.00021471498,0.000113583075,0.0072903144,0.0024837991,0.0021205891,0.961381,0.00007752034],"about_ca_topic_score_codex":0.026855454,"about_ca_topic_score_gemma":0.04819945,"teacher_disagreement_score":0.03007515,"about_ca_system_score_codex":0.001537797,"about_ca_system_score_gemma":0.0012239966,"threshold_uncertainty_score":0.10061145},"labels":[],"label_agreement":null},{"id":"W6912843967","doi":"10.5281/zenodo.6472988","title":"SSHOC'n Tell Challenge - How to use VCR and the Switchboard in teaching and training","year":2022,"lang":"en","type":"article","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canarie","funders":"European Commission","keywords":"Interoperability; Training (meteorology); Workflow; Mode (computer interface); Resource (disambiguation); Ontology","score_opus":0.042602280336717833,"score_gpt":0.2532450939475584,"score_spread":0.21064281361084056,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6912843967","genre_codex":"commentary","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013543602,0.005891452,0.20582724,0.50728834,0.023867832,0.00087745936,0.0019993663,0.015367516,0.22533713],"genre_scores_gemma":[0.20959957,0.0053315386,0.29841894,0.0655178,0.0045886943,0.0015744863,0.003433307,0.006876814,0.40465882],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9861829,0.006723797,0.000489304,0.0016724202,0.003263787,0.0016677631],"domain_scores_gemma":[0.9638573,0.010101102,0.0008339877,0.0051381746,0.007477925,0.012591531],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.019741654,0.0010208819,0.000867537,0.001395685,0.00404255,0.014811143,0.0038997468,0.0077134045,0.07185748],"category_scores_gemma":[0.02942475,0.0005400847,0.00095462846,0.0012785724,0.004107537,0.019408619,0.008961169,0.0071385703,0.04916854],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025220079,0.00043174232,0.0017237688,0.00078855635,0.000031983167,0.0003289974,0.0038826081,0.0010647755,0.0036083518,0.095554344,0.5203545,0.37197822],"study_design_scores_gemma":[0.00005935186,0.00025587375,0.00093379413,0.00052034674,0.00001148398,0.00028661118,0.005013223,0.0014650287,0.0023057414,0.04601018,0.94305384,0.00008465079],"about_ca_topic_score_codex":0.0036581636,"about_ca_topic_score_gemma":0.0052151484,"teacher_disagreement_score":0.07185748,"about_ca_system_score_codex":0.002750133,"about_ca_system_score_gemma":0.0077084755,"threshold_uncertainty_score":0.2403872},"labels":[],"label_agreement":null},{"id":"W6920413301","doi":"10.60692/49ea2-rcy41","title":"Recursive Top-Down Production for Sentence Generation with Latent Trees","year":2020,"lang":"en","type":"article","venue":"Greater South Information System","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; Mila - Quebec Artificial Intelligence Institute; McGill University; Canadian Institute for Advanced Research","funders":"","keywords":"Tree (set theory); Translation (biology); Tree structure; Rule-based machine translation; Binary tree; Sentence; Sequence (biology); Property (philosophy)","score_opus":0.037606479297597,"score_gpt":0.22209010464308054,"score_spread":0.18448362534548354,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6920413301","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009220605,0.0001821909,0.9859313,0.00021598351,0.00003332638,0.000052310595,0.00031646554,0.0025634873,0.0014842524],"genre_scores_gemma":[0.48070017,0.0002754865,0.50982493,0.0002815744,0.00007710108,0.00030913667,0.0018629226,0.00097551296,0.005693216],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99949694,0.00019785941,0.00002346556,0.00016159723,0.00007757662,0.000042552078],"domain_scores_gemma":[0.9977591,0.0017381869,0.00008507549,0.00022553804,0.0001446762,0.000047501082],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010327003,0.00068470626,0.0005875438,0.00052782317,0.00039075973,0.0010210561,0.0016093395,0.0009491375,0.0061352914],"category_scores_gemma":[0.0048700324,0.00057960505,0.0009421437,0.00064914115,0.0008484914,0.001577016,0.0011425611,0.0019470387,0.0022884184],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001781024,0.00014387933,0.0018272379,0.0003227676,0.000106245076,0.00039137618,0.000505957,0.5990061,0.013491271,0.086671874,0.008800566,0.28855464],"study_design_scores_gemma":[0.000011029267,0.00001433716,0.00007283754,0.000009191949,0.00000957399,0.000047554757,0.000008325057,0.9622051,0.00225594,0.033984613,0.0013741712,0.000007329453],"about_ca_topic_score_codex":0.003091067,"about_ca_topic_score_gemma":0.008449777,"teacher_disagreement_score":0.0061352914,"about_ca_system_score_codex":0.00092214684,"about_ca_system_score_gemma":0.0011885909,"threshold_uncertainty_score":0.020524561},"labels":[],"label_agreement":null},{"id":"W6920512498","doi":"10.60692/f0rsc-1n406","title":"Unsupervised Dependency Graph Network","year":2022,"lang":"en","type":"article","venue":"Greater South Information System","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Dependency (UML); Dependency grammar; Parsing; Dependency graph; Sentence; Graph","score_opus":0.018807826920340016,"score_gpt":0.20681007101294566,"score_spread":0.18800224409260563,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6920512498","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.044257317,0.000364359,0.9425566,0.0005097232,0.00006256424,0.00016013163,0.0021204348,0.003029067,0.00693971],"genre_scores_gemma":[0.57140696,0.0007237212,0.39833218,0.0006446702,0.00012706651,0.0005191302,0.011272076,0.0009049976,0.016069245],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992385,0.00022740227,0.00003099673,0.00032220574,0.0001229899,0.00005795067],"domain_scores_gemma":[0.9981292,0.0010034657,0.0001571254,0.00030961976,0.00033787076,0.00006269173],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007087407,0.00077302754,0.0005146038,0.0015155496,0.00068777706,0.0005682681,0.0014288899,0.0009759375,0.0048593264],"category_scores_gemma":[0.003821734,0.00050877396,0.0009204171,0.0014137924,0.0006493745,0.0021783072,0.0010093782,0.0010129528,0.0011409156],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037490434,0.0002319837,0.0068496447,0.0004888087,0.00028833954,0.00056005997,0.000500353,0.4346743,0.02462971,0.12824339,0.035516653,0.3676419],"study_design_scores_gemma":[0.000011669029,0.000030483026,0.0014210393,0.000016098133,0.000035875677,0.00011365062,0.000023570963,0.9394469,0.0042140055,0.04783287,0.006836459,0.000017403192],"about_ca_topic_score_codex":0.0070899944,"about_ca_topic_score_gemma":0.013979426,"teacher_disagreement_score":0.0070899944,"about_ca_system_score_codex":0.0012742238,"about_ca_system_score_gemma":0.0012764491,"threshold_uncertainty_score":0.016256094},"labels":[],"label_agreement":null},{"id":"W6920946916","doi":"10.6084/m9.figshare.26621485.v1","title":"Additional file 4 of Reported adverse events related to use of hepatitis C virus direct-acting antivirals with opioids: 2017–2021","year":2024,"lang":"en","type":"article","venue":"Figshare","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vancouver Infectious Diseases Centre; Simon Fraser University","funders":"","keywords":"Adverse effect; Hepatitis C virus; Hepacivirus; Hepatitis C; Virus; MEDLINE","score_opus":0.030195769197300555,"score_gpt":0.2765135409019264,"score_spread":0.24631777170462588,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6920946916","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00018700316,0.00002431082,0.000075966345,0.000124232,0.00002052421,0.00008983032,0.9980259,0.00007842803,0.0013738129],"genre_scores_gemma":[0.013006393,0.0003349238,0.0013310151,0.0010282644,0.00017382555,0.0021529591,0.9624053,0.00031510278,0.019252105],"study_design_codex":"not_applicable","study_design_gemma":"observational","domain_scores_codex":[0.99893814,0.00019113754,0.00034582426,0.00018276933,0.00022046632,0.000121697805],"domain_scores_gemma":[0.97776675,0.014655656,0.0028852646,0.00085140835,0.0033768448,0.0004640851],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0012849217,0.00073061057,0.0010279327,0.0021029457,0.0006793267,0.0009853223,0.0010081284,0.0010103403,0.8390304],"category_scores_gemma":[0.025149224,0.00037764225,0.00091093965,0.003356939,0.0001708521,0.001473555,0.00082856737,0.0008631772,0.11575836],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005269763,0.000085450396,0.0027081699,0.002407222,0.00004880215,0.00006158577,0.0000386999,0.00026325145,0.000040951756,0.00047470382,0.98531854,0.008025672],"study_design_scores_gemma":[0.008072566,0.0005862198,0.055440307,0.0076332483,0.00028818526,0.00060256955,0.00055960956,0.0015712183,0.00063095766,0.008100448,0.91634953,0.00016526645],"about_ca_topic_score_codex":0.009383331,"about_ca_topic_score_gemma":0.01513217,"teacher_disagreement_score":0.8390304,"about_ca_system_score_codex":0.0010935027,"about_ca_system_score_gemma":0.001490253,"threshold_uncertainty_score":0.22960353},"labels":[],"label_agreement":null},{"id":"W6925066475","doi":"10.17605/osf.io/y9bnw","title":"IRIS - Time To Treatment Individual Participant Meta-Analysis - Update SAP","year":2023,"lang":"en","type":"other","venue":"Open Science Framework","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Randomization; Atrial fibrillation; Covariate; Thrombolysis; Clinical trial; Occlusion","score_opus":0.1214476471224806,"score_gpt":0.3964148733684847,"score_spread":0.2749672262460041,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6925066475","genre_codex":"dataset","genre_gemma":"protocol","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"protocol","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009426024,0.2781632,0.13396478,0.016498292,0.012562705,0.018774875,0.47806072,0.01444731,0.038101997],"genre_scores_gemma":[0.17923301,0.12663853,0.25348347,0.025717167,0.008235752,0.107897274,0.22687922,0.013395909,0.058519635],"study_design_codex":"not_applicable","study_design_gemma":"meta_analysis","domain_scores_codex":[0.9806943,0.013534095,0.0020730302,0.0014745054,0.0018974572,0.00032671914],"domain_scores_gemma":[0.9683741,0.022114372,0.002510474,0.0041973162,0.0023775876,0.00042618348],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.026309092,0.002615583,0.0068100756,0.0027336043,0.0004893165,0.002714696,0.002788332,0.0019139623,0.07652774],"category_scores_gemma":[0.09276532,0.0015642287,0.023523316,0.003950344,0.00037349178,0.0019857509,0.0027654793,0.0037266125,0.009378159],"study_design_candidate":"meta_analysis","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.019925093,0.000092400194,0.003044044,0.2618958,0.26088732,0.00026014427,0.00023688978,0.00392802,0.0009509726,0.010225752,0.32279447,0.115759104],"study_design_scores_gemma":[0.039975427,0.0011426046,0.0103995465,0.024267437,0.4976952,0.0005174477,0.000074283125,0.0048188446,0.0012428272,0.023247307,0.39633456,0.00028456328],"about_ca_topic_score_codex":0.0038890305,"about_ca_topic_score_gemma":0.009682723,"teacher_disagreement_score":0.07652774,"about_ca_system_score_codex":0.0013180984,"about_ca_system_score_gemma":0.0031817907,"threshold_uncertainty_score":0.25601083},"labels":[],"label_agreement":null},{"id":"W6926050135","doi":"10.20381/ruor-31092","title":"Gifted children: What and how to care?","year":2025,"lang":"en","type":"other","venue":"University of Ottawa - Library","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Association (psychology); Mental health; Mediation; Sleep (system call); Mindset; Meaning (existential); Strengths and Difficulties Questionnaire","score_opus":0.0029870348928009703,"score_gpt":0.17640429849074107,"score_spread":0.1734172635979401,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6926050135","genre_codex":"empirical","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.48303455,0.05859943,0.0026247583,0.4291057,0.0021540686,0.000108445456,0.00090820447,0.00014442188,0.023320327],"genre_scores_gemma":[0.9143031,0.04781508,0.009287321,0.021860417,0.0004614216,0.00018459477,0.00041568856,0.000042130534,0.005630231],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9994247,0.00023590872,0.000042580945,0.000055791257,0.00009246216,0.00014860058],"domain_scores_gemma":[0.99808097,0.00023339775,0.00026588317,0.000051898794,0.00015685386,0.0012110207],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011580272,0.00024780893,0.00042340052,0.00032760086,0.0017789892,0.0017762234,0.00057303987,0.0012695162,0.0041933837],"category_scores_gemma":[0.0057275603,0.00013154383,0.00039892909,0.00031825918,0.0023471976,0.0030489436,0.0017757888,0.0023535674,0.0005924648],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020858787,0.0004364399,0.29095712,0.001386131,0.00011718218,0.007414633,0.051669046,0.00041255777,0.0018003319,0.022152867,0.15048674,0.47295842],"study_design_scores_gemma":[0.00014523337,0.00067126803,0.33733353,0.01193386,0.00016493927,0.014368112,0.25237507,0.0007390983,0.0011757248,0.057970196,0.32276732,0.00035563708],"about_ca_topic_score_codex":0.02773957,"about_ca_topic_score_gemma":0.036063757,"teacher_disagreement_score":0.02773957,"about_ca_system_score_codex":0.0023934473,"about_ca_system_score_gemma":0.004215862,"threshold_uncertainty_score":0.05515623},"labels":[],"label_agreement":null},{"id":"W6926684399","doi":"10.24433/co.0736044.v1","title":"IDS-ML: Intrusion Detection System Development Using Machine Learning","year":2022,"lang":"en","type":"other","venue":"Code Ocean","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Intrusion detection system; Naive Bayes classifier; Bayesian probability; Random forest; Anomaly-based intrusion detection system; Bayesian inference; Intrusion; Support vector machine","score_opus":0.013513693115899726,"score_gpt":0.24549293413123716,"score_spread":0.23197924101533743,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6926684399","genre_codex":"software","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0020082858,0.00027972576,0.22339292,0.00071696576,0.0002896066,0.00043287312,0.02232734,0.69927585,0.051276457],"genre_scores_gemma":[0.056695778,0.0010007706,0.3346006,0.0019032682,0.00035549034,0.001390916,0.14238524,0.20244412,0.2592238],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99877673,0.000119982986,0.000089612775,0.00020888857,0.00070035673,0.00010444772],"domain_scores_gemma":[0.9976572,0.0007554703,0.00013639164,0.000401689,0.00086886465,0.00018027131],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00092322595,0.0016069663,0.0005921837,0.0019161307,0.00045040078,0.0014868269,0.0020247425,0.00095887727,0.14556931],"category_scores_gemma":[0.005568837,0.00096549874,0.00074382266,0.001331074,0.0005604973,0.0024398004,0.0017786025,0.0019288531,0.11364034],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021121693,0.00010098323,0.0008067091,0.00035838454,0.000026753682,0.00013984587,0.0000700876,0.0036318481,0.003853383,0.0062974254,0.8362765,0.14822693],"study_design_scores_gemma":[0.0003525719,0.00014310496,0.0023961754,0.0002147037,0.00003697264,0.000516141,0.00003501778,0.11433282,0.040098026,0.013883053,0.82786965,0.000121723875],"about_ca_topic_score_codex":0.0052932063,"about_ca_topic_score_gemma":0.00495795,"teacher_disagreement_score":0.14556931,"about_ca_system_score_codex":0.0010492386,"about_ca_system_score_gemma":0.0014307615,"threshold_uncertainty_score":0.48697788},"labels":[],"label_agreement":null},{"id":"W6930087413","doi":"10.5281/zenodo.12064342","title":"Iso 8098 pdf","year":2024,"lang":"en","type":"other","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Task (project management); Test (biology); International standard; National standard; Work (physics)","score_opus":0.02039903230179648,"score_gpt":0.2583906047957742,"score_spread":0.2379915724939777,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6930087413","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00052792754,0.0012490187,0.0101133585,0.0007331883,0.0027644548,0.0006191576,0.008962644,0.007037309,0.9679929],"genre_scores_gemma":[0.00318417,0.0018640832,0.005993232,0.00067859486,0.00051481527,0.00036116538,0.0130829895,0.003915351,0.9704057],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.994323,0.00027491912,0.00035390098,0.00033817306,0.004388813,0.000321201],"domain_scores_gemma":[0.98889834,0.00050440105,0.00029288098,0.0010406503,0.008871839,0.00039195994],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.00291477,0.0023600683,0.001245927,0.006478186,0.0016963514,0.008731324,0.0046599004,0.0034623637,0.59105736],"category_scores_gemma":[0.0104663,0.0012274402,0.0013286208,0.0051062843,0.0012163538,0.0066802367,0.0027419783,0.0025074505,0.6757027],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001167823,0.00009055816,0.00011265288,0.00062031014,0.000007332486,0.00011570862,0.00007644074,0.00033190905,0.0030654662,0.00735916,0.80972016,0.17838348],"study_design_scores_gemma":[0.000009261439,0.000022478367,0.00016519727,0.00010937184,0.000003962971,0.000070885326,0.00003123775,0.000072255636,0.0009942785,0.0007368346,0.9977748,0.000009402218],"about_ca_topic_score_codex":0.006866616,"about_ca_topic_score_gemma":0.0055773277,"teacher_disagreement_score":0.40894264,"about_ca_system_score_codex":0.0022893243,"about_ca_system_score_gemma":0.0039713155,"threshold_uncertainty_score":0.583307},"labels":[],"label_agreement":null},{"id":"W6930123465","doi":"10.5281/zenodo.12429354","title":"imm 5669 form 2022 pdf","year":2024,"lang":"en","type":"other","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Citizenship; Waiver; Principal (computer security); Refugee; Schedule; Immigration","score_opus":0.017203036059120783,"score_gpt":0.25314037303546333,"score_spread":0.23593733697634256,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6930123465","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0002534008,0.00035335857,0.000942953,0.0005621068,0.001726191,0.00019222054,0.005823712,0.004140036,0.9860059],"genre_scores_gemma":[0.00083366735,0.00017476799,0.00030276098,0.00025488395,0.00016063973,0.000047554575,0.002192019,0.0010025321,0.99503124],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9990532,0.000061787585,0.000042115626,0.00012738319,0.0005938346,0.00012171612],"domain_scores_gemma":[0.9973431,0.00026481008,0.00007888488,0.0002784677,0.0014128021,0.00062194484],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.0008910695,0.001357773,0.0011377566,0.0023672846,0.0019805725,0.0077366726,0.0023135517,0.0027228273,0.9536662],"category_scores_gemma":[0.004090398,0.0009833159,0.0008847066,0.0019224938,0.00062306103,0.004574251,0.0027652318,0.0020879922,0.9336898],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000055011908,0.000045554225,0.000029782397,0.0001231322,0.0000018384475,0.000031281772,0.000020253468,0.0000328456,0.00052740285,0.0012234182,0.9659516,0.031957712],"study_design_scores_gemma":[0.00001450235,0.000026050437,0.00012249475,0.000037664515,0.0000017437632,0.000046421104,0.000029039726,0.000038236893,0.00035283994,0.00036627162,0.99895847,0.0000062092154],"about_ca_topic_score_codex":0.003584524,"about_ca_topic_score_gemma":0.005400114,"teacher_disagreement_score":0.04633379,"about_ca_system_score_codex":0.0014680289,"about_ca_system_score_gemma":0.0017416651,"threshold_uncertainty_score":0.06608945},"labels":[],"label_agreement":null},{"id":"W6930468936","doi":"10.5281/zenodo.13818283","title":"ships of the expanse pdf","year":2024,"lang":"en","type":"other","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Navy; Space (punctuation); Point (geometry); On board; Quarter (Canadian coin); Character (mathematics); Natural (archaeology)","score_opus":0.023462547786371704,"score_gpt":0.25093438990549216,"score_spread":0.22747184211912047,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6930468936","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0003522476,0.00072220643,0.0008298897,0.0012050034,0.004195815,0.0002177218,0.0033515727,0.004298123,0.9848275],"genre_scores_gemma":[0.0015265655,0.00070015417,0.000600337,0.00068841607,0.0010935373,0.00005869762,0.0019007857,0.0020862387,0.9913453],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99927074,0.00004250299,0.000033505417,0.000074465,0.00047712965,0.000101722006],"domain_scores_gemma":[0.9964881,0.00047456115,0.00010885819,0.00044920947,0.0016355668,0.0008438863],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0006485772,0.0013000711,0.00091946253,0.0027404348,0.0018269741,0.007993112,0.0015420208,0.0018481507,0.9027074],"category_scores_gemma":[0.0064478214,0.0007158792,0.0011371537,0.0022449552,0.0007561067,0.007656355,0.0037204481,0.002422122,0.8375771],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000016324435,0.000016175847,0.00003591727,0.00010149639,0.0000013277715,0.000062190295,0.000031562282,0.00002945772,0.00015358404,0.0011226475,0.9726433,0.025785996],"study_design_scores_gemma":[0.000006485496,0.000010970091,0.00022578814,0.000058088146,0.00000163274,0.00009192251,0.000053839594,0.000029591436,0.00012474986,0.0004977758,0.9988913,0.00000784689],"about_ca_topic_score_codex":0.0027565698,"about_ca_topic_score_gemma":0.005677349,"teacher_disagreement_score":0.9027074,"about_ca_system_score_codex":0.0010804938,"about_ca_system_score_gemma":0.0012108053,"threshold_uncertainty_score":0.13877612},"labels":[],"label_agreement":null},{"id":"W6930597274","doi":"10.5281/zenodo.14235288","title":"REHOUSE public report: Definition of the long-term performance tests of the components of the RPs","year":2024,"lang":"en","type":"article","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Anesthesia Research Foundation","funders":"European Commission","keywords":"Deliverable; Task (project management); Listing (finance); Acceptance testing; Order (exchange); European union","score_opus":0.04821118015236916,"score_gpt":0.2574594997346185,"score_spread":0.20924831958224935,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6930597274","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.026045095,0.0067983135,0.23183155,0.0060075806,0.0028889854,0.01305799,0.09300215,0.0125101,0.60785824],"genre_scores_gemma":[0.12927969,0.008271882,0.20825084,0.004415747,0.0013779398,0.03171757,0.17917095,0.009877942,0.42763746],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.97826517,0.0042357845,0.0016264832,0.0016465848,0.0125367455,0.0016892984],"domain_scores_gemma":[0.9815207,0.0021503116,0.0018949278,0.0027824417,0.0110050645,0.00064657285],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.017076751,0.0027628215,0.0013300027,0.004333016,0.0018700048,0.006691656,0.005701821,0.004745664,0.044658206],"category_scores_gemma":[0.016700968,0.0011812904,0.0014935547,0.002383618,0.0024981475,0.004754329,0.0045849434,0.00423933,0.061016083],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001078812,0.0013541114,0.004719814,0.004691982,0.000083068,0.001062807,0.0020410158,0.0100746015,0.03065935,0.11156536,0.5846929,0.24797614],"study_design_scores_gemma":[0.00007552487,0.00081538316,0.0052517806,0.00096159364,0.000038994396,0.0004907395,0.00062169315,0.0007465509,0.014260817,0.005522823,0.9710933,0.000120703815],"about_ca_topic_score_codex":0.009915828,"about_ca_topic_score_gemma":0.0055452553,"teacher_disagreement_score":0.044658206,"about_ca_system_score_codex":0.004736054,"about_ca_system_score_gemma":0.011676121,"threshold_uncertainty_score":0.1493966},"labels":[],"label_agreement":null},{"id":"W6930605642","doi":"10.5281/zenodo.15293594","title":"Building, securing, sharing containers: Paths towards a sustainable ecosystem for admins, users, applications","year":2024,"lang":"en","type":"article","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Xenon Pharmaceuticals (Canada)","funders":"","keywords":"Container (type theory); Software; Session (web analytics); Exploit; Data sharing; Application lifecycle management; Software development","score_opus":0.020231793284071084,"score_gpt":0.2784601396284359,"score_spread":0.2582283463443648,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6930605642","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.05497439,0.009892548,0.36100245,0.3488159,0.006944934,0.00052302825,0.0004952628,0.012659898,0.20469162],"genre_scores_gemma":[0.36193675,0.020016154,0.4016097,0.032728992,0.002005894,0.00056160917,0.0013440709,0.007115473,0.17268139],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9954112,0.0018187471,0.00015231334,0.00033859725,0.0012255254,0.0010535793],"domain_scores_gemma":[0.9935675,0.00068447046,0.0003963531,0.0013902327,0.0009928992,0.0029686342],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010639749,0.0006185354,0.0005507914,0.0008159159,0.008065707,0.01822623,0.0021687557,0.004146096,0.014480793],"category_scores_gemma":[0.006555319,0.00079442223,0.0008418314,0.0014191017,0.0066269315,0.040009703,0.025133843,0.006046241,0.010274778],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012449575,0.00017993203,0.0029270041,0.00065526395,0.000047305155,0.00047912804,0.025312822,0.0017406959,0.005432254,0.3091349,0.41139713,0.2425691],"study_design_scores_gemma":[0.0000067596325,0.000047290778,0.0004497354,0.00025426174,0.00000825539,0.0001474715,0.007995029,0.00094975886,0.0007405241,0.047974043,0.941387,0.000039827195],"about_ca_topic_score_codex":0.0048311497,"about_ca_topic_score_gemma":0.0066986834,"teacher_disagreement_score":0.01822623,"about_ca_system_score_codex":0.0035689368,"about_ca_system_score_gemma":0.012295295,"threshold_uncertainty_score":0.05626899},"labels":[],"label_agreement":null},{"id":"W6930883462","doi":"10.5281/zenodo.270006","title":"Figure 5 in A new species of Oeneis from Alaska, United States, with notes on the Oeneis chryxus complex (Lepidoptera: Nymphalidae: Satyrinae)","year":2016,"lang":"en","type":"other","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Aedeagus; Dorsum; Male genitalia; Scale (ratio); Adult male","score_opus":0.03382600545303695,"score_gpt":0.2480519755044771,"score_spread":0.21422597005144017,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6930883462","genre_codex":"empirical","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.54026574,0.0147897005,0.0125903785,0.0012068864,0.0055618784,0.0007552833,0.012145725,0.0010250851,0.4116593],"genre_scores_gemma":[0.85136247,0.005990822,0.024627224,0.0006328879,0.00044106576,0.00025306136,0.009748029,0.00013249904,0.10681196],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9999223,0.000004895382,0.000007812881,0.00003125567,0.000020976142,0.0000127584935],"domain_scores_gemma":[0.9999286,0.000010984364,0.000012080219,0.000007774564,0.000023010616,0.000017626302],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.00008192971,0.00042037282,0.0002061654,0.0015230486,0.0017261106,0.00035296485,0.00030391305,0.0005817697,0.02259552],"category_scores_gemma":[0.00017338154,0.00018120735,0.00019116665,0.0011893394,0.00056735845,0.00076269393,0.00064659666,0.00079942663,0.0032944228],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005371466,0.00028396954,0.19266973,0.001990596,0.00015119679,0.019154303,0.013432849,0.0008451497,0.039588444,0.004296789,0.07554496,0.6515048],"study_design_scores_gemma":[0.000024386549,0.00013466486,0.37982056,0.000508044,0.00017427611,0.012696471,0.009246886,0.00027813012,0.0021831037,0.0008712384,0.5940076,0.00005474837],"about_ca_topic_score_codex":0.04414873,"about_ca_topic_score_gemma":0.1444081,"teacher_disagreement_score":0.9774045,"about_ca_system_score_codex":0.00034958153,"about_ca_system_score_gemma":0.00047026196,"threshold_uncertainty_score":0.087783515},"labels":[],"label_agreement":null},{"id":"W6930893362","doi":"10.5281/zenodo.15218318","title":"Empis (Enoplempis) paraerobatica Sinclair, Brooks & Cumming, 2025, sp. nov.","year":2025,"lang":"en","type":"article","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Food Inspection Agency","funders":"","keywords":"Holotype; Butte; Placer mining; Conglomerate","score_opus":0.02358803383911245,"score_gpt":0.2778823409524366,"score_spread":0.2542943071133241,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6930893362","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.126299,0.033398002,0.014964272,0.0029820248,0.0034482873,0.0023495224,0.017667137,0.00216235,0.79672945],"genre_scores_gemma":[0.58826864,0.03029311,0.040253464,0.0061277156,0.0015324996,0.0031961652,0.021059817,0.00054146594,0.3087271],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9996774,0.000025923608,0.000026702115,0.00012005008,0.00011278887,0.000037222384],"domain_scores_gemma":[0.99962044,0.00005459787,0.00009454688,0.000043828506,0.00012803687,0.000058646903],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00037579704,0.0015039416,0.00066707,0.0020847258,0.0028622688,0.0007988464,0.0011328673,0.0011717879,0.019288026],"category_scores_gemma":[0.0008316324,0.00044055106,0.00025549292,0.0018941496,0.0009598635,0.0024559807,0.0010999768,0.0020321028,0.015430274],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00031022175,0.0001442156,0.016160324,0.0009778729,0.00010821758,0.0026524048,0.0022702436,0.0007609503,0.0102507565,0.003050261,0.09874111,0.86457336],"study_design_scores_gemma":[0.000119536766,0.00020855997,0.19504538,0.0017405616,0.00026092157,0.007583763,0.0031001046,0.000891635,0.0028662232,0.0016341491,0.78649354,0.000055629433],"about_ca_topic_score_codex":0.0360334,"about_ca_topic_score_gemma":0.08900926,"teacher_disagreement_score":0.0360334,"about_ca_system_score_codex":0.0015542537,"about_ca_system_score_gemma":0.0010795787,"threshold_uncertainty_score":0.07164729},"labels":[],"label_agreement":null},{"id":"W6931403194","doi":"10.5281/zenodo.5450540","title":"Aegidinus howdenorum Colby 2009, new species","year":2009,"lang":"en","type":"article","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Holotype; Paratype; Allotype; Cannibalism; Genus","score_opus":0.028036938897786436,"score_gpt":0.24821372691050603,"score_spread":0.2201767880127196,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6931403194","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2504274,0.018330855,0.0039717555,0.002494429,0.0018683505,0.0005844308,0.008211809,0.00044921754,0.71366173],"genre_scores_gemma":[0.8571173,0.013056283,0.011472788,0.0019328948,0.00066330575,0.00039647814,0.0069538476,0.00010973249,0.10829731],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9998472,0.000014907774,0.000012194891,0.00005951128,0.000044579458,0.000021593614],"domain_scores_gemma":[0.9998369,0.000030255418,0.00003710568,0.000020538915,0.000048843674,0.000026342803],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00012685625,0.0005496259,0.00029515036,0.0011239483,0.0016598919,0.00055056514,0.00039639245,0.00046122537,0.015987445],"category_scores_gemma":[0.00026330433,0.00024774563,0.0001225911,0.0007060394,0.000641664,0.001310778,0.0005910768,0.001012668,0.0035580043],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00092822034,0.00017972822,0.04905693,0.0013718279,0.00009073591,0.0048720716,0.003287784,0.00052395486,0.041979246,0.007402861,0.120960176,0.7693465],"study_design_scores_gemma":[0.000081839666,0.00019736636,0.320328,0.0006839484,0.00013224431,0.010459252,0.0027354802,0.00031874265,0.0037960457,0.0015050711,0.6596977,0.000064355656],"about_ca_topic_score_codex":0.017815357,"about_ca_topic_score_gemma":0.08977606,"teacher_disagreement_score":0.017815357,"about_ca_system_score_codex":0.0011106605,"about_ca_system_score_gemma":0.00042802028,"threshold_uncertainty_score":0.053483367},"labels":[],"label_agreement":null},{"id":"W6931913916","doi":"10.5683/sp3/0uymwx","title":"Sondage panel sur l'élection canadienne de 2008","year":2022,"lang":"fr","type":"dataset","venue":"Borealis","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Waterloo; Université de Montréal","funders":"","keywords":"Panel data; Technical report; Panel discussion","score_opus":0.029442182011673645,"score_gpt":0.26041665352442717,"score_spread":0.23097447151275352,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6931913916","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0022951402,0.00006475413,0.00015561229,0.00019514226,0.000079469435,0.000050749477,0.9923436,0.00024501616,0.0045705345],"genre_scores_gemma":[0.0034629395,0.00006868486,0.00038937273,0.000059987295,0.000018141207,0.00019943924,0.9869299,0.000056634948,0.008814875],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99812645,0.0001949404,0.00013354314,0.00031785827,0.0008491844,0.00037801574],"domain_scores_gemma":[0.9923402,0.0008986741,0.0004235795,0.0008599728,0.00496904,0.0005085475],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017596454,0.0009572168,0.0009698633,0.004443983,0.0011628933,0.0015621404,0.0013638006,0.00088083826,0.031975694],"category_scores_gemma":[0.007384124,0.0005330957,0.00048752775,0.008039133,0.0003391878,0.0006916447,0.0010938784,0.0016440502,0.03137908],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005449903,0.000019330624,0.0034117084,0.000088548775,0.000011504894,0.000018612822,0.00007441033,0.00014745082,0.00008805824,0.0003229341,0.9913496,0.0044133402],"study_design_scores_gemma":[0.00008272473,0.000020754445,0.08425626,0.00013049215,0.000017963566,0.00003140628,0.00038321267,0.0006640267,0.00049898366,0.00024487937,0.9136253,0.000043947723],"about_ca_topic_score_codex":0.74321085,"about_ca_topic_score_gemma":0.8577554,"teacher_disagreement_score":0.25678915,"about_ca_system_score_codex":0.0061183893,"about_ca_system_score_gemma":0.01008237,"threshold_uncertainty_score":0.5166029},"labels":[],"label_agreement":null},{"id":"W6939068694","doi":"10.60692/v5y4c-kkk74","title":"MasakhaPOS: Part-of-Speech Tagging for Typologically Diverse African languages","year":2023,"lang":"en","type":"article","venue":"Greater South Information System","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Mila - Quebec Artificial Intelligence Institute","funders":"","keywords":"Conditional random field; Transfer (computing); Baseline (sea); Transfer of learning; Training set; Field (mathematics); Languages of Africa","score_opus":0.045390498958156875,"score_gpt":0.26460863083939634,"score_spread":0.21921813188123945,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6939068694","genre_codex":"dataset","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.44527173,0.0016559524,0.038989596,0.0010373854,0.0006691884,0.0005290085,0.47425225,0.013456191,0.02413856],"genre_scores_gemma":[0.27636167,0.000463555,0.05952176,0.00038188646,0.0000895823,0.0010223851,0.65431035,0.0011092224,0.0067396415],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99930155,0.00017965412,0.00007891919,0.00024093466,0.000102159705,0.00009671863],"domain_scores_gemma":[0.99882895,0.00038169327,0.00014849499,0.0003359894,0.00018063853,0.00012425218],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001080448,0.00082092383,0.00041052225,0.0026979784,0.0014346801,0.0008486579,0.0008507756,0.0007260899,0.008193879],"category_scores_gemma":[0.002403236,0.0003471692,0.0005083658,0.00218944,0.00048068905,0.001832771,0.002497413,0.00095433695,0.005331371],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.003817364,0.0006431042,0.11386803,0.004110424,0.00039465263,0.0028337457,0.007027846,0.0077943434,0.09571309,0.013821741,0.361603,0.38837263],"study_design_scores_gemma":[0.00043972698,0.0004009078,0.18215083,0.00055065245,0.00020979409,0.0028157544,0.008063222,0.028348608,0.051273298,0.01654936,0.7089414,0.00025643324],"about_ca_topic_score_codex":0.0065862644,"about_ca_topic_score_gemma":0.014526887,"teacher_disagreement_score":0.008193879,"about_ca_system_score_codex":0.00051948486,"about_ca_system_score_gemma":0.0011849011,"threshold_uncertainty_score":0.027411282},"labels":[],"label_agreement":null},{"id":"W6939217260","doi":"10.60692/e3qnm-hc398","title":"On Orthogonality Constraints for Transformers","year":2021,"lang":"en","type":"article","venue":"Greater South Information System","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Orthogonality; Transformer; Joint (building); Natural language; Algebra over a field; Matrix algebra","score_opus":0.02878086650117915,"score_gpt":0.245701246738939,"score_spread":0.21692038023775986,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6939217260","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.030805573,0.002243678,0.8176918,0.006846021,0.001233166,0.00031881203,0.0022214362,0.0013595618,0.13728],"genre_scores_gemma":[0.55018294,0.0038738707,0.38888216,0.003832871,0.0018046426,0.0005358,0.0056434786,0.0019399978,0.043304287],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.994779,0.0016515276,0.00052295055,0.0010014361,0.0012905459,0.0007544822],"domain_scores_gemma":[0.99014133,0.0059607243,0.00040308683,0.0014685938,0.0016856695,0.00034068138],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0052867304,0.0010872916,0.001180015,0.0021408969,0.002594284,0.0040363157,0.0019201967,0.001576332,0.025815139],"category_scores_gemma":[0.017717991,0.0013753932,0.0020033675,0.0030592112,0.0033197275,0.022176243,0.0065168412,0.0046742493,0.003682186],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000049265487,0.000026383248,0.00017635735,0.000066808876,0.000014332317,0.00007966511,0.000117484364,0.001394864,0.00038400872,0.97232413,0.008699218,0.016667524],"study_design_scores_gemma":[0.000021754684,0.0000103303955,0.00006019128,0.000034820063,0.000011792225,0.00007318927,0.00007243724,0.004689405,0.000496025,0.9797881,0.0147292325,0.000012687593],"about_ca_topic_score_codex":0.0041451883,"about_ca_topic_score_gemma":0.0070135854,"teacher_disagreement_score":0.025815139,"about_ca_system_score_codex":0.002019038,"about_ca_system_score_gemma":0.0021675827,"threshold_uncertainty_score":0.086360216},"labels":[],"label_agreement":null},{"id":"W6939270573","doi":"10.6084/m9.figshare.12679767.v1","title":"Additional file 3 of Examining the relationship between maternal body size, gestational glucose tolerance status, mode of delivery and ethnicity on human milk microbiota at three months post-partum","year":2020,"lang":"en","type":"article","venue":"Figshare","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"SickKids Foundation; University of Ottawa; Hospital for Sick Children; University of Toronto","funders":"","keywords":"Gestational diabetes; Ethnic group; Gestation; Pregnancy; Race (biology); Table (database)","score_opus":0.05990217257932343,"score_gpt":0.2850957906498606,"score_spread":0.2251936180705372,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6939270573","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0003126047,0.000017969833,0.00016759626,0.00007958518,0.000016233942,0.000055828325,0.9987638,0.00012376513,0.0004626027],"genre_scores_gemma":[0.022140704,0.00023934872,0.005043361,0.0009523863,0.00015026118,0.0037513585,0.95121324,0.000865435,0.01564382],"study_design_codex":"not_applicable","study_design_gemma":"observational","domain_scores_codex":[0.9992669,0.0001370959,0.00013521418,0.00020278124,0.00014446999,0.00011352954],"domain_scores_gemma":[0.9813648,0.014903993,0.001026645,0.00088980055,0.0014322515,0.00038247922],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0015289176,0.0010799279,0.0013073357,0.0017640239,0.001033851,0.0014329446,0.0016623908,0.0012499647,0.82789534],"category_scores_gemma":[0.03326904,0.0004761007,0.0011003106,0.0026484532,0.00023916525,0.0014954808,0.0010343244,0.00096849987,0.08375135],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006496315,0.0001305658,0.008357262,0.0034086625,0.00012969127,0.00011581518,0.00013041162,0.00041105162,0.00013631991,0.0006701731,0.9751149,0.010745532],"study_design_scores_gemma":[0.010967628,0.0007702184,0.133968,0.009354036,0.0011412172,0.00157991,0.0019898491,0.004444597,0.0017075677,0.019462707,0.81423783,0.00037643308],"about_ca_topic_score_codex":0.0135689005,"about_ca_topic_score_gemma":0.020941757,"teacher_disagreement_score":0.82789534,"about_ca_system_score_codex":0.00082732836,"about_ca_system_score_gemma":0.0017261788,"threshold_uncertainty_score":0.24548644},"labels":[],"label_agreement":null},{"id":"W6939515603","doi":"10.6084/m9.figshare.13364169.v1","title":"Additional file 8 of Evaluation of the health and healthcare system burden due to antimicrobial-resistant Escherichia coli infections in humans: a systematic review and meta-analysis","year":2020,"lang":"en","type":"article","venue":"Figshare","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Guelph; Public Health Agency of Canada","funders":"","keywords":"Escherichia coli; Antibiotic resistance; Healthcare system; Antimicrobial; Health care; Systematic review","score_opus":0.06871351118725767,"score_gpt":0.3201041003704403,"score_spread":0.25139058918318263,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6939515603","genre_codex":"dataset","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00083642773,0.0012153495,0.00053336687,0.00030813168,0.00010408013,0.0018343679,0.99370533,0.0003400008,0.0011228719],"genre_scores_gemma":[0.08709529,0.007904745,0.03472815,0.005973088,0.0011094764,0.10816941,0.70847815,0.0022658834,0.044275824],"study_design_codex":"systematic_review","study_design_gemma":"meta_analysis","domain_scores_codex":[0.9956891,0.0012177757,0.0016569671,0.00059548224,0.0005644438,0.00027627655],"domain_scores_gemma":[0.92872036,0.056943823,0.008268815,0.0017034854,0.0037021416,0.00066136353],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.006346435,0.0019306302,0.0053617647,0.0063845506,0.0006750276,0.0024850967,0.001963253,0.0017571527,0.78842723],"category_scores_gemma":[0.07909845,0.0015378812,0.0077142,0.009104633,0.000518178,0.0034250503,0.0020549002,0.0011660162,0.027075684],"study_design_candidate":"meta_analysis","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.003897214,0.00018113395,0.0043601943,0.6939124,0.008825723,0.00021748808,0.00026169425,0.0011569761,0.0004079072,0.0021674668,0.26549754,0.019114299],"study_design_scores_gemma":[0.15940505,0.0035092728,0.078244634,0.20335805,0.052038684,0.0015977678,0.0012351692,0.0059254062,0.0019849907,0.021242693,0.47046497,0.0009932226],"about_ca_topic_score_codex":0.005774782,"about_ca_topic_score_gemma":0.015383725,"teacher_disagreement_score":0.78842723,"about_ca_system_score_codex":0.0019752344,"about_ca_system_score_gemma":0.003564363,"threshold_uncertainty_score":0.30178285},"labels":[],"label_agreement":null},{"id":"W6939535173","doi":"10.6084/m9.figshare.25464907.v1","title":"Additional file 1 of Translation and validation of menopause quick 6 (MQ6) into the Malay language","year":2024,"lang":"en","type":"article","venue":"Figshare","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Malay; Translation (biology); Machine translation; Semantics (computer science)","score_opus":0.017466241913144248,"score_gpt":0.26653329015843075,"score_spread":0.2490670482452865,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6939535173","genre_codex":"dataset","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0006844366,0.000023349035,0.00092080014,0.00012286307,0.0000811107,0.0002543974,0.99381495,0.00078623777,0.003311858],"genre_scores_gemma":[0.011414576,0.00012554156,0.009979194,0.00062706065,0.000105292376,0.0039769197,0.9500771,0.0029916416,0.02070271],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99898046,0.00030457083,0.00022255332,0.00021764432,0.00016666646,0.00010797102],"domain_scores_gemma":[0.97382826,0.018854382,0.0008119546,0.0018407902,0.004126654,0.0005380482],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0021981376,0.0010899585,0.0009344305,0.0027466856,0.0010663824,0.0014709515,0.0013161692,0.00094850967,0.88120395],"category_scores_gemma":[0.039089,0.0005822659,0.0007943094,0.0025846593,0.00045810867,0.0015007564,0.0015360563,0.00088878087,0.37066728],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005455832,0.0001409858,0.001332693,0.002797386,0.000034227414,0.00011151873,0.0003320817,0.00020428155,0.00038190105,0.0007785908,0.9744461,0.01889451],"study_design_scores_gemma":[0.0023522433,0.00040166973,0.023609105,0.0026513287,0.00016542288,0.0006082525,0.0012321706,0.0008193906,0.002834552,0.007230872,0.95788574,0.00020916498],"about_ca_topic_score_codex":0.0070576747,"about_ca_topic_score_gemma":0.01241772,"teacher_disagreement_score":0.88120395,"about_ca_system_score_codex":0.0008652478,"about_ca_system_score_gemma":0.0027388858,"threshold_uncertainty_score":0.16944814},"labels":[],"label_agreement":null},{"id":"W6950021013","doi":"10.5281/zenodo.48718","title":"stackr: vcf2dadi for ALF ;)","year":2016,"lang":"en","type":"other","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Function (biology); Component (thermodynamics); Set (abstract data type); Process (computing); Identification (biology)","score_opus":0.0237835670884203,"score_gpt":0.2716508388044106,"score_spread":0.2478672717159903,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6950021013","genre_codex":"software","genre_gemma":"software","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"software","genre_consensus":"software","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.000566643,0.00018536708,0.06289625,0.00029715768,0.0004507525,0.0000807004,0.08136603,0.8305403,0.023616927],"genre_scores_gemma":[0.0124318525,0.00027364882,0.040641677,0.0007416262,0.0002869213,0.0005106795,0.16679654,0.7321823,0.046134666],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9984505,0.0001909304,0.00013385994,0.00043971758,0.00049651164,0.00028838238],"domain_scores_gemma":[0.9972646,0.00064255577,0.0001431229,0.00095586356,0.0008111983,0.00018263794],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0026109659,0.005340041,0.0019930778,0.003045891,0.0012241481,0.0055963085,0.006305939,0.0030338129,0.38572475],"category_scores_gemma":[0.00721972,0.0030261816,0.002943091,0.0019209216,0.0007192067,0.0063240505,0.005698111,0.0037558686,0.430966],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019107891,0.000018939467,0.00032174826,0.00037866473,0.00004579557,0.000074684634,0.00011501558,0.0002541343,0.0012945567,0.004020382,0.96860427,0.024680702],"study_design_scores_gemma":[0.00013546934,0.000016857348,0.00048699687,0.000103235616,0.000034017594,0.00018564431,0.0000475159,0.0020391296,0.012246901,0.0055388077,0.9790461,0.00011932073],"about_ca_topic_score_codex":0.0066012633,"about_ca_topic_score_gemma":0.0045953006,"teacher_disagreement_score":0.38572475,"about_ca_system_score_codex":0.0022763957,"about_ca_system_score_gemma":0.0015948496,"threshold_uncertainty_score":0.8761891},"labels":[],"label_agreement":null},{"id":"W6955366362","doi":"10.58066/tdr5-6t46","title":"Letter from Kathleen Raine to Robin Skelton, [between 1950 and 1959], [01]","year":2023,"lang":"en","type":"article","venue":"UVic’s Research and Learning Repository (University of Victoria)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Wife; Syllabus; Creative writing; Adventure; English literature; Poetry","score_opus":0.023630783594921004,"score_gpt":0.27720650146378395,"score_spread":0.25357571786886296,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6955366362","genre_codex":"commentary","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0023642688,0.021110449,0.0007121069,0.78263825,0.11232861,0.000083845815,0.00079149555,0.0002719757,0.07969903],"genre_scores_gemma":[0.015465192,0.007473414,0.0004714131,0.46676725,0.020863628,0.00010548705,0.0003297703,0.00020444594,0.4883195],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99889755,0.00020910379,0.00009072886,0.00029851045,0.00034451575,0.0001595652],"domain_scores_gemma":[0.9972778,0.0007804964,0.00016495885,0.00008770863,0.0011457242,0.00054333964],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012584626,0.0006061664,0.00053073105,0.0007877879,0.0055439514,0.0036072533,0.000780704,0.0055792984,0.043514773],"category_scores_gemma":[0.014096432,0.0005355089,0.00032252044,0.00065361056,0.001079809,0.003000325,0.001493945,0.008237609,0.019287646],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000009905606,0.000002583953,0.00009645854,0.000011294169,8.354266e-7,0.0001131946,0.00031429992,0.0000025000265,0.00002818475,0.00048576287,0.99465936,0.004275706],"study_design_scores_gemma":[0.000002463993,0.0000060472107,0.00037116275,0.00006431494,0.0000011423543,0.0002660343,0.00058042014,0.000009109931,0.000043533895,0.00019559833,0.99845207,0.000008200522],"about_ca_topic_score_codex":0.024494136,"about_ca_topic_score_gemma":0.037250184,"teacher_disagreement_score":0.043514773,"about_ca_system_score_codex":0.0024492142,"about_ca_system_score_gemma":0.0036421677,"threshold_uncertainty_score":0.14557135},"labels":[],"label_agreement":null},{"id":"W6957842666","doi":"10.60692/cnz1s-f6c92","title":"Soil micronutrients & grassland productivity (NutNet dataset)","year":2021,"lang":"en","type":"article","venue":"Greater South Information System","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Guelph","funders":"","keywords":"Grassland; Biomass (ecology); Productivity; Nutrient; Soil nutrients; Soil water; Arid; Soil classification","score_opus":0.019782033589891074,"score_gpt":0.2262886352939414,"score_spread":0.2065066017040503,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6957842666","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0046917754,0.00008992771,0.000073033916,0.0000667663,0.000014320929,0.000013974851,0.9944765,0.00017700648,0.00039662886],"genre_scores_gemma":[0.004761875,0.000046643596,0.00038732289,0.00004155687,0.000004955787,0.00006124242,0.99440706,0.000016034654,0.00027341978],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9996742,0.000049227903,0.00004456426,0.00010531418,0.00008017529,0.000046460216],"domain_scores_gemma":[0.9991848,0.00021094157,0.00011683947,0.00014975897,0.0002336737,0.00010399432],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00045740465,0.0010894799,0.0007898855,0.001442877,0.00035774263,0.00069045566,0.0014595344,0.0014193126,0.010187106],"category_scores_gemma":[0.0022653393,0.0003232531,0.0008790344,0.0025966985,0.00022263403,0.00054434343,0.00095222576,0.00072341354,0.008890259],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000424133,0.00022351752,0.033312183,0.002181648,0.000258411,0.00022853915,0.00017712408,0.0036950537,0.0013902263,0.0008708373,0.9446556,0.012582722],"study_design_scores_gemma":[0.0008753889,0.00022190214,0.22797488,0.00057364616,0.0002494016,0.00039515842,0.0005777264,0.008168604,0.0016837106,0.0019402654,0.75723255,0.000106628744],"about_ca_topic_score_codex":0.047572196,"about_ca_topic_score_gemma":0.06947502,"teacher_disagreement_score":0.047572196,"about_ca_system_score_codex":0.0009443586,"about_ca_system_score_gemma":0.0010354335,"threshold_uncertainty_score":0.094590604},"labels":[],"label_agreement":null},{"id":"W6957966860","doi":"10.60692/1cgeg-66n77","title":"Proceedings of the Fifth Named Entity Workshop","year":2015,"lang":"en","type":"article","venue":"Greater South Information System","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Transliteration; Named-entity recognition; Named entity; Task (project management); Conditional random field; Entity linking; Scope (computer science); Deep learning","score_opus":0.03273616498747871,"score_gpt":0.22746386287599712,"score_spread":0.1947276978885184,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6957966860","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024964802,0.027096083,0.17032275,0.089762025,0.27237397,0.002105025,0.053404354,0.01623629,0.34373462],"genre_scores_gemma":[0.0552422,0.009276357,0.08048779,0.012231229,0.026071224,0.0015830793,0.134134,0.0061065927,0.67486745],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99413246,0.0019154425,0.0005630456,0.0011321191,0.0016336141,0.00062331196],"domain_scores_gemma":[0.9860372,0.0031314562,0.00040750517,0.0026351882,0.005103591,0.0026850197],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012040056,0.0016032036,0.0014228595,0.0020646427,0.002832663,0.009295194,0.003543662,0.0030383868,0.09070351],"category_scores_gemma":[0.011843368,0.0006274816,0.0021315708,0.0019332129,0.0009768254,0.009597824,0.006186933,0.0038606636,0.06571032],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028224196,0.00013008504,0.0003155806,0.000300713,0.000039559633,0.00025992468,0.00023249515,0.00043297347,0.0013638667,0.0053783916,0.91791844,0.07334563],"study_design_scores_gemma":[0.000026842494,0.00003647064,0.00042928456,0.00010378249,0.000020902542,0.00012653893,0.00023535395,0.00072972523,0.0015710096,0.00277896,0.99391586,0.000025425281],"about_ca_topic_score_codex":0.0034950054,"about_ca_topic_score_gemma":0.00641228,"teacher_disagreement_score":0.09070351,"about_ca_system_score_codex":0.002433899,"about_ca_system_score_gemma":0.0033960312,"threshold_uncertainty_score":0.30343348},"labels":[],"label_agreement":null},{"id":"W6958076519","doi":"10.6084/m9.figshare.20126087.v1","title":"Hamilton_allData.csv","year":2022,"lang":"en","type":"dataset","venue":"Figshare","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"","score_opus":0.02911780580013865,"score_gpt":0.28861404539105423,"score_spread":0.25949623959091556,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6958076519","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.000041709176,0.000017466864,0.000025729358,0.000028815157,0.000011031877,0.0000045938523,0.99905723,0.00034010803,0.00047327124],"genre_scores_gemma":[0.00024198738,0.000028724038,0.000156815,0.000022974586,0.000003534996,0.000028969487,0.99884003,0.00011260802,0.00056441664],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99904376,0.00009441693,0.00008617667,0.00026887518,0.00029521997,0.00021140375],"domain_scores_gemma":[0.99777764,0.00040705415,0.00016308592,0.00062469905,0.0007606869,0.0002668536],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007326685,0.0024527642,0.0015019443,0.0041660606,0.0010639147,0.0024187372,0.0032485658,0.001999709,0.10415338],"category_scores_gemma":[0.004944627,0.00088934455,0.0013601136,0.008732067,0.0006253878,0.0013414519,0.0021021592,0.0017123977,0.13820612],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000020813884,0.000008853159,0.00028294048,0.0002397461,0.000012943987,0.000007912667,0.000010932765,0.00020028728,0.00004050827,0.00029428393,0.99793065,0.00095019443],"study_design_scores_gemma":[0.00013958714,0.0000073451993,0.0018468912,0.00011672567,0.000016967286,0.00001648629,0.000050100636,0.00047736612,0.00028783508,0.0009875396,0.9960264,0.000026618043],"about_ca_topic_score_codex":0.16187018,"about_ca_topic_score_gemma":0.2490133,"teacher_disagreement_score":0.16187018,"about_ca_system_score_codex":0.002876493,"about_ca_system_score_gemma":0.0040220274,"threshold_uncertainty_score":0.34842777},"labels":[],"label_agreement":null},{"id":"W6958173769","doi":"10.60692/j695m-0rz06","title":"Phrase-aware Unsupervised Constituency Parsing","year":2022,"lang":"en","type":"article","venue":"Greater South Information System","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Parsing; Phrase; Dependency grammar; Generality; Grammar; Task (project management); Regularization (linguistics); Natural language; Benchmark (surveying)","score_opus":0.025218482103529706,"score_gpt":0.21792495482808363,"score_spread":0.19270647272455393,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6958173769","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018435964,0.00022764094,0.96658236,0.0001868645,0.000044284174,0.00008753303,0.00071180885,0.011345904,0.0023775538],"genre_scores_gemma":[0.2566359,0.00024293967,0.72643554,0.000342922,0.00009760789,0.0002563545,0.007988679,0.0027960653,0.005204027],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9989839,0.0003692384,0.000058725316,0.0003611636,0.00016033625,0.00006668134],"domain_scores_gemma":[0.9976979,0.0011788618,0.00013953711,0.00062460976,0.00030190562,0.000057285855],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012371636,0.0011850569,0.00091521646,0.0012763293,0.0007103423,0.00089990936,0.0018412907,0.001011339,0.004819952],"category_scores_gemma":[0.0032522795,0.0005760693,0.0012293139,0.001350295,0.0007447986,0.0024137248,0.0014153497,0.002135531,0.0037454665],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024476368,0.00023470503,0.0040870328,0.00055553124,0.0001762059,0.00038859554,0.0006202944,0.078120574,0.095107086,0.038019348,0.025597487,0.75684834],"study_design_scores_gemma":[0.00002753063,0.00008877526,0.0015555267,0.000024429151,0.00005023169,0.00020526032,0.00009471153,0.9151228,0.03335157,0.039251845,0.01019218,0.000035135596],"about_ca_topic_score_codex":0.002091013,"about_ca_topic_score_gemma":0.0052855136,"teacher_disagreement_score":0.004819952,"about_ca_system_score_codex":0.000711725,"about_ca_system_score_gemma":0.0016938015,"threshold_uncertainty_score":0.016124368},"labels":[],"label_agreement":null},{"id":"W6958274294","doi":"10.6084/m9.figshare.22603550","title":"Additional file 1 of Influence of previous experience with and beliefs regarding anal cancer screening on willingness to be screened among men living with HIV","year":2023,"lang":"en","type":"article","venue":"Open MIND","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ottawa Hospital; University of Ottawa; Women's College Hospital; University of British Columbia; BC Centre for Disease Control; University of Toronto; University Health Network","funders":"","keywords":"Anal cancer; Human immunodeficiency virus (HIV); Men who have sex with men; Cancer screening; HIV screening; MEDLINE; Qualitative research","score_opus":0.02156420005642946,"score_gpt":0.29095622458305,"score_spread":0.26939202452662053,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6958274294","genre_codex":"dataset","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00238021,0.000054840788,0.00018613912,0.00028565255,0.00003902823,0.00026987967,0.99469155,0.00006243519,0.0020302718],"genre_scores_gemma":[0.12407664,0.00069672166,0.005416775,0.0021538977,0.00039504713,0.020694362,0.80488205,0.00046178862,0.041222766],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9992235,0.00022343418,0.00014958462,0.00013243251,0.00013626448,0.00013482846],"domain_scores_gemma":[0.9777076,0.017128024,0.0017874655,0.0006942594,0.0022109533,0.00047165784],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0013765608,0.00090772175,0.0012370929,0.0015347499,0.0012593925,0.0010297495,0.0015260363,0.0013493283,0.76257205],"category_scores_gemma":[0.039414056,0.0004837764,0.0011455046,0.002587333,0.000216568,0.0022521324,0.000881057,0.0014083312,0.05496619],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010732462,0.00036362844,0.016532885,0.004582255,0.00016055985,0.00010514149,0.0003136892,0.00043409452,0.000055355387,0.0011535839,0.96477014,0.010455355],"study_design_scores_gemma":[0.028594466,0.0024055669,0.40008432,0.023589654,0.0019222506,0.0019413849,0.00857598,0.0074037346,0.0011876675,0.022090849,0.50163877,0.0005653045],"about_ca_topic_score_codex":0.016744193,"about_ca_topic_score_gemma":0.019499052,"teacher_disagreement_score":0.76257205,"about_ca_system_score_codex":0.00084744993,"about_ca_system_score_gemma":0.0018861262,"threshold_uncertainty_score":0.33866215},"labels":[],"label_agreement":null},{"id":"W6958364816","doi":"10.6084/m9.figshare.26554172.v1","title":"Additional file 3 of Physical activity promotion in chiropractic: a systematic review of clinician-based surveys","year":2024,"lang":"en","type":"article","venue":"Figshare","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of New Brunswick","funders":"","keywords":"Physical activity; Promotion (chess); Data collection; Qualitative research","score_opus":0.05065539933160696,"score_gpt":0.34129002096120675,"score_spread":0.2906346216295998,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6958364816","genre_codex":"dataset","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00046637902,0.00020267595,0.00031689083,0.00024025481,0.000048231526,0.0012238524,0.996268,0.00015877161,0.0010750036],"genre_scores_gemma":[0.04082243,0.0020449841,0.015216027,0.003229651,0.00040270758,0.124952085,0.78023523,0.0010301372,0.03206677],"study_design_codex":"not_applicable","study_design_gemma":"systematic_review","domain_scores_codex":[0.99692327,0.00067908526,0.0013144999,0.00036852807,0.00045537224,0.00025916824],"domain_scores_gemma":[0.9214588,0.061982658,0.008567309,0.0016096614,0.005644051,0.00073762785],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0050803744,0.0011749443,0.0024218915,0.005461696,0.00083257956,0.0015279717,0.0014773083,0.0013317353,0.830538],"category_scores_gemma":[0.07994884,0.00081278745,0.0021262395,0.008748571,0.0003524561,0.0027980797,0.0013805479,0.0008851906,0.042145498],"study_design_candidate":"systematic_review","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017335343,0.00013685759,0.0032405716,0.1707858,0.0005840031,0.000104519575,0.00034687785,0.0005511454,0.00015481944,0.0018943656,0.80132604,0.019141449],"study_design_scores_gemma":[0.048999686,0.001336455,0.080182634,0.14118719,0.003908091,0.0006613188,0.001975047,0.0030108434,0.001046779,0.016905313,0.70036197,0.00042471822],"about_ca_topic_score_codex":0.0059508206,"about_ca_topic_score_gemma":0.014351869,"teacher_disagreement_score":0.830538,"about_ca_system_score_codex":0.0020326646,"about_ca_system_score_gemma":0.0038778144,"threshold_uncertainty_score":0.24171704},"labels":[],"label_agreement":null},{"id":"W6960254084","doi":"10.13140/rg.2.2.34656.60169","title":"Jurisdiction and Scale in Ridehailing: Options for Ontario","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Jurisdiction; Scale (ratio); Work (physics); Legislation","score_opus":0.008835367739555927,"score_gpt":0.2767063823712288,"score_spread":0.26787101463167284,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6960254084","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.24565956,0.0013999189,0.0080074305,0.03804223,0.00033977808,0.0003191722,0.00456296,0.0005058229,0.7011632],"genre_scores_gemma":[0.8702798,0.00067631813,0.006815832,0.0017981189,0.000056224795,0.00016306325,0.0009422444,0.00015986677,0.119108476],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.996757,0.00046248516,0.00010220712,0.0002656006,0.00097054313,0.0014422362],"domain_scores_gemma":[0.994743,0.0011475973,0.0001847625,0.00047090853,0.0021261815,0.0013275222],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033458376,0.00022421233,0.00043431722,0.001242021,0.00559335,0.006262752,0.0022628575,0.0015436676,0.035494916],"category_scores_gemma":[0.012500873,0.00033996266,0.0007131854,0.0032243885,0.0026893357,0.004333633,0.003490536,0.0013692356,0.001642542],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00095349667,0.00019037933,0.06644309,0.00030047802,0.000096536336,0.0004099886,0.00826224,0.00494806,0.0010790981,0.55470276,0.16925429,0.1933596],"study_design_scores_gemma":[0.00036883,0.0001996347,0.098350786,0.00060520804,0.00032926595,0.00020422277,0.04855549,0.011633151,0.002002123,0.076836444,0.760608,0.0003068277],"about_ca_topic_score_codex":0.9519639,"about_ca_topic_score_gemma":0.98665684,"teacher_disagreement_score":0.0480361,"about_ca_system_score_codex":0.035219073,"about_ca_system_score_gemma":0.07395996,"threshold_uncertainty_score":0.25553346},"labels":[],"label_agreement":null},{"id":"W6966495929","doi":"10.48448/qf09-7q95","title":"X4D-SceneFormer: Enhanced Scene Understanding on 4D Point Cloud Videos through Cross-Modal Knowledge Transfer","year":2024,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Point cloud; Segmentation; Inference; Transformer; Cloud computing; Transfer of learning; Semantics (computer science); Knowledge transfer","score_opus":0.037916987602710936,"score_gpt":0.3456183589797978,"score_spread":0.30770137137708686,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6966495929","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015971273,0.00039978058,0.9730294,0.00019528861,0.000054507604,0.00010576273,0.0006723343,0.007920073,0.0016516463],"genre_scores_gemma":[0.33473366,0.0007744877,0.6507241,0.00066458527,0.000096846874,0.0002324652,0.0058021713,0.00074685924,0.0062248576],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99955887,0.000056437642,0.0000143637435,0.00018289914,0.000119062985,0.00006827939],"domain_scores_gemma":[0.99970144,0.00007908243,0.000033557088,0.000084525665,0.000063693646,0.000037635233],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00066979905,0.0016779007,0.001021151,0.0014954681,0.00033889187,0.0010394898,0.0023654238,0.0015580715,0.004138745],"category_scores_gemma":[0.0013922415,0.0005935919,0.0013885055,0.0010701587,0.00063092133,0.002172946,0.0026299162,0.0017751524,0.0015758331],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00039674118,0.0003110292,0.0017127591,0.00027601872,0.00025994205,0.00039932906,0.00033189723,0.21157561,0.047882207,0.0060151983,0.014397246,0.7164419],"study_design_scores_gemma":[0.000021388521,0.00007125959,0.00067593716,0.000016313918,0.000024006036,0.00013154667,0.0000647271,0.98170805,0.008950426,0.0052585336,0.00305478,0.000022965143],"about_ca_topic_score_codex":0.011603966,"about_ca_topic_score_gemma":0.014297147,"teacher_disagreement_score":0.011603966,"about_ca_system_score_codex":0.00075512927,"about_ca_system_score_gemma":0.00097111665,"threshold_uncertainty_score":0.023072839},"labels":[],"label_agreement":null},{"id":"W6968341211","doi":"10.5281/zenodo.2653758","title":"The CESSDA Vocabulary Service: A New State-of-the-Art Tool for Creating and Publishing Controlled Terms Lists","year":2019,"lang":"en","type":"article","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Canadian Institute for Public Safety Research and Treatment","funders":"","keywords":"Controlled vocabulary; Metadata; Vocabulary; Context (archaeology); Publishing; SNOMED CT","score_opus":0.013695885335107466,"score_gpt":0.2362870229799684,"score_spread":0.22259113764486094,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6968341211","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0025624481,0.0016053998,0.73691756,0.0013990825,0.00094119203,0.0017341012,0.05838502,0.16452299,0.031932227],"genre_scores_gemma":[0.024218237,0.0030503995,0.66988784,0.0012089679,0.00058322283,0.0026209662,0.20962363,0.0590957,0.02971107],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99006057,0.0021347832,0.002905522,0.0011155434,0.0033705644,0.0004130117],"domain_scores_gemma":[0.97794026,0.009410796,0.0017090153,0.00622085,0.0035432833,0.0011757686],"candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.013602302,0.0021606414,0.0022261834,0.017639406,0.002228462,0.010651616,0.0045495396,0.0024353955,0.05055207],"category_scores_gemma":[0.042217433,0.0021610549,0.0028488624,0.012418485,0.0017830645,0.018530635,0.009521015,0.0042893034,0.048338417],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00049970497,0.00015288609,0.002175169,0.0036872462,0.0002604465,0.00061819155,0.0021881696,0.0014241909,0.018807683,0.15087923,0.36418635,0.45512074],"study_design_scores_gemma":[0.000115117444,0.000057532576,0.0004793817,0.00046754902,0.00004718224,0.00039596637,0.00032661713,0.0040319446,0.0072933245,0.034037508,0.9525688,0.00017903229],"about_ca_topic_score_codex":0.011147057,"about_ca_topic_score_gemma":0.010374054,"teacher_disagreement_score":0.9893484,"about_ca_system_score_codex":0.0026014727,"about_ca_system_score_gemma":0.008056026,"threshold_uncertainty_score":0.16911352},"labels":[],"label_agreement":null},{"id":"W6976435142","doi":"10.60692/7vtrw-fg261","title":"Recursive Top-Down Production for Sentence Generation with Latent Trees","year":2020,"lang":"en","type":"article","venue":"Greater South Information System","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; Mila - Quebec Artificial Intelligence Institute; McGill University; Canadian Institute for Advanced Research","funders":"","keywords":"Tree (set theory); Translation (biology); Tree structure; Rule-based machine translation; Binary tree; Sentence; Sequence (biology); Property (philosophy)","score_opus":0.037606479297597,"score_gpt":0.22209010464308054,"score_spread":0.18448362534548354,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6976435142","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009220605,0.0001821909,0.9859313,0.00021598351,0.00003332638,0.000052310595,0.00031646554,0.0025634873,0.0014842524],"genre_scores_gemma":[0.48070017,0.0002754865,0.50982493,0.0002815744,0.00007710108,0.00030913667,0.0018629226,0.00097551296,0.005693216],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99949694,0.00019785941,0.00002346556,0.00016159723,0.00007757662,0.000042552078],"domain_scores_gemma":[0.9977591,0.0017381869,0.00008507549,0.00022553804,0.0001446762,0.000047501082],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010327003,0.00068470626,0.0005875438,0.00052782317,0.00039075973,0.0010210561,0.0016093395,0.0009491375,0.0061352914],"category_scores_gemma":[0.0048700324,0.00057960505,0.0009421437,0.00064914115,0.0008484914,0.001577016,0.0011425611,0.0019470387,0.0022884184],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001781024,0.00014387933,0.0018272379,0.0003227676,0.000106245076,0.00039137618,0.000505957,0.5990061,0.013491271,0.086671874,0.008800566,0.28855464],"study_design_scores_gemma":[0.000011029267,0.00001433716,0.00007283754,0.000009191949,0.00000957399,0.000047554757,0.000008325057,0.9622051,0.00225594,0.033984613,0.0013741712,0.000007329453],"about_ca_topic_score_codex":0.003091067,"about_ca_topic_score_gemma":0.008449777,"teacher_disagreement_score":0.0061352914,"about_ca_system_score_codex":0.00092214684,"about_ca_system_score_gemma":0.0011885909,"threshold_uncertainty_score":0.020524561},"labels":[],"label_agreement":null},{"id":"W6976619488","doi":"10.60692/6pxh0-f5314","title":"On the Role of Orthographic Variations in Building Multidialectal Arabic Word Embeddings","year":2021,"lang":"en","type":"article","venue":"Greater South Information System","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Sherbrooke","funders":"","keywords":"Word embedding; Lexicon; Word (group theory); Orthographic projection; Embedding; Canonical correlation; Arabic; Space (punctuation)","score_opus":0.01208779424678677,"score_gpt":0.21609902648169996,"score_spread":0.2040112322349132,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6976619488","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17155543,0.0011378229,0.818618,0.00034958703,0.0001631247,0.00010652827,0.00041103046,0.002437688,0.0052207517],"genre_scores_gemma":[0.60052186,0.00077652535,0.38990542,0.00016664401,0.000073941905,0.00011027687,0.0021456806,0.00044990753,0.0058497074],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992095,0.00031926617,0.000053815056,0.0002846352,0.00008756798,0.00004514613],"domain_scores_gemma":[0.9986343,0.0006491215,0.0001000762,0.00026682203,0.0002978239,0.00005179887],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00086960226,0.00087050966,0.00033021884,0.00097385497,0.00048412607,0.00094224105,0.00050016184,0.00041078057,0.0021428505],"category_scores_gemma":[0.0031516578,0.0002618511,0.0005586375,0.0010055274,0.00048775921,0.0022642338,0.0011858111,0.00093504187,0.0016261432],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002146709,0.00016527103,0.0068116533,0.00020827104,0.00013069015,0.0001689931,0.00059102406,0.028736526,0.035029296,0.008674745,0.0047440147,0.9145249],"study_design_scores_gemma":[0.000029645505,0.00025646374,0.009900419,0.00007896155,0.000114205184,0.0004528175,0.000869784,0.9214356,0.039401207,0.01566638,0.01172865,0.000065851396],"about_ca_topic_score_codex":0.003584418,"about_ca_topic_score_gemma":0.009849024,"teacher_disagreement_score":0.003584418,"about_ca_system_score_codex":0.00025228653,"about_ca_system_score_gemma":0.00072756805,"threshold_uncertainty_score":0.0071685314},"labels":[],"label_agreement":null},{"id":"W6976990631","doi":"10.6082/uchicago.7969","title":"One Language, Different Romanization Rules: The 20th Anniversary of the Pinyin Conversion Project","year":2023,"lang":"en","type":"article","venue":"Knowledge@UChicago (University of Chicago)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Pinyin; Romanization; Power (physics); Register (sociolinguistics)","score_opus":0.015367986025236908,"score_gpt":0.23891323325341882,"score_spread":0.22354524722818192,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6976990631","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.082837634,0.031001337,0.024764456,0.046854977,0.018526224,0.00010180468,0.001546433,0.00086394,0.7935033],"genre_scores_gemma":[0.4224704,0.018729815,0.017605267,0.004122675,0.0040560635,0.00012314714,0.0031242692,0.0019433284,0.527825],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.998629,0.00034926608,0.0000643774,0.00030477683,0.0004654144,0.00018708619],"domain_scores_gemma":[0.99903333,0.0002204726,0.00006767182,0.0002618659,0.0002950925,0.0001216283],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030069577,0.00036155473,0.00032817537,0.0011319785,0.002144257,0.0042717173,0.0006606297,0.0008974988,0.022782145],"category_scores_gemma":[0.004893809,0.0003210932,0.00033876364,0.0014302194,0.0023793594,0.005126068,0.0046519293,0.002787281,0.005575332],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00034161084,0.00009543593,0.0016256392,0.0001815016,0.000020375328,0.00038162505,0.0018775235,0.0006421041,0.0017303777,0.43740708,0.1371447,0.41855198],"study_design_scores_gemma":[0.0000103062175,0.000030661475,0.0018479618,0.00011044359,0.000008543291,0.0002395503,0.00036790856,0.00027585775,0.002075823,0.036622994,0.9583929,0.000017078053],"about_ca_topic_score_codex":0.002237636,"about_ca_topic_score_gemma":0.0027162058,"teacher_disagreement_score":0.022782145,"about_ca_system_score_codex":0.0017442511,"about_ca_system_score_gemma":0.0010692303,"threshold_uncertainty_score":0.076213896},"labels":[],"label_agreement":null},{"id":"W6977300777","doi":"10.6084/m9.figshare.12519011","title":"Additional file 1 of Sustainable by design: a systematic review of factors for health promotion program sustainability","year":2020,"lang":"en","type":"article","venue":"Figshare","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Public Health Ontario","funders":"","keywords":"Grey literature; Sustainability; MEDLINE; Promotion (chess); Public health","score_opus":0.03929564891891113,"score_gpt":0.3195193044678938,"score_spread":0.28022365554898265,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6977300777","genre_codex":"dataset","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00047998523,0.0013493395,0.00073117675,0.0006151702,0.000111064146,0.006563302,0.98739386,0.00020422993,0.0025517994],"genre_scores_gemma":[0.026932687,0.013696231,0.037870668,0.004620837,0.00068729493,0.29348594,0.5822507,0.000994056,0.039461505],"study_design_codex":"systematic_review","study_design_gemma":"systematic_review","domain_scores_codex":[0.9971481,0.0007195744,0.001171221,0.0003286591,0.00044550558,0.00018697736],"domain_scores_gemma":[0.9233845,0.0647667,0.0059098513,0.0010392013,0.0043298216,0.0005698843],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0060799364,0.0018574029,0.0038821043,0.0098228,0.0010008881,0.0021473018,0.002213792,0.0018410586,0.81954926],"category_scores_gemma":[0.07718637,0.0010837415,0.0030099198,0.013635861,0.0005601982,0.004458235,0.0020562774,0.0010796095,0.03300347],"study_design_candidate":"systematic_review","study_design_consensus":"systematic_review","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010323804,0.0000920434,0.000881312,0.6431871,0.0005342187,0.0000904446,0.00030128108,0.0005232917,0.00017103336,0.0026541133,0.32648972,0.024042988],"study_design_scores_gemma":[0.02364932,0.0010103334,0.020458916,0.37101728,0.005624363,0.0006086704,0.0013675072,0.0018582467,0.0007099833,0.016478831,0.55684394,0.00037255726],"about_ca_topic_score_codex":0.008186121,"about_ca_topic_score_gemma":0.020870306,"teacher_disagreement_score":0.81954926,"about_ca_system_score_codex":0.0039291284,"about_ca_system_score_gemma":0.007870151,"threshold_uncertainty_score":0.2573911},"labels":[],"label_agreement":null},{"id":"W6979286200","doi":"","title":"Towards a Probabilistic Framework for Analyzing and Improving LLM-Enabled Software","year":2025,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Agencia Nacional de Promoción Científica y Tecnológica; Consejo Nacional de Investigaciones Científicas y Técnicas; International Development Research Centre; Universidade Federal do Amazonas; Secretaría de Ciencia y Técnica, Universidad de Buenos Aires; Agencia Nacional de Investigación e Innovación","keywords":"Probabilistic logic; Identification (biology); Documentation; Reliability (semiconductor); Software; Strengths and weaknesses; Natural language; Software system","score_opus":0.026690736922143757,"score_gpt":0.21079064753268015,"score_spread":0.18409991061053638,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6979286200","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0015730921,0.00006556079,0.99746585,0.00022754092,0.000004179267,0.000028442737,0.000034948484,0.00034675404,0.00025369963],"genre_scores_gemma":[0.0830006,0.00023610925,0.9153812,0.00017579496,0.000048162077,0.00024023051,0.00019374439,0.00031292922,0.00041127595],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9825494,0.008237717,0.0011006669,0.002275256,0.0051956186,0.00064137287],"domain_scores_gemma":[0.953029,0.027914213,0.0047251205,0.00939913,0.0044113877,0.00052111375],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.022712052,0.0018852799,0.0014791052,0.006475952,0.0017619451,0.005128094,0.004302582,0.0020440628,0.0016529354],"category_scores_gemma":[0.06533794,0.0023638413,0.0039232057,0.003044095,0.007884376,0.010974929,0.00615887,0.0049853055,0.0006006729],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006555369,0.00011682444,0.0034682387,0.00031040044,0.00016382166,0.00022739897,0.00066673616,0.42287317,0.0055152443,0.4943254,0.0011908775,0.07107641],"study_design_scores_gemma":[0.000012909279,0.000042623928,0.00028775475,0.00006453722,0.000038233575,0.00007718579,0.00007769626,0.60260224,0.0032161272,0.39080232,0.0027402777,0.00003804226],"about_ca_topic_score_codex":0.007491486,"about_ca_topic_score_gemma":0.009684881,"teacher_disagreement_score":0.022712052,"about_ca_system_score_codex":0.0044872896,"about_ca_system_score_gemma":0.0071673715,"threshold_uncertainty_score":0.12011427},"labels":[],"label_agreement":null},{"id":"W6979380867","doi":"","title":"Neural Machine Translation for Coptic-French: Strategies for Low-Resource Ancient Languages","year":2025,"lang":"en","type":"article","venue":"ArXiv.org","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Machine translation; Robustness (evolution); Pipeline (software); Translation (biology); Training set","score_opus":0.02564002490769811,"score_gpt":0.30699418644563475,"score_spread":0.28135416153793663,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6979380867","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15805367,0.0026102953,0.80882204,0.0008731421,0.00025097735,0.00022667815,0.00086055335,0.012754035,0.015548671],"genre_scores_gemma":[0.58499026,0.0012548944,0.40074718,0.00026763827,0.00011912966,0.00022993851,0.0035888602,0.001542871,0.007259292],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992712,0.00029291998,0.00004299459,0.00023009781,0.00010289109,0.00005984016],"domain_scores_gemma":[0.998519,0.0007062172,0.000079243226,0.00033501498,0.00032232274,0.000038170165],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015023886,0.0010528885,0.00080535497,0.0011254696,0.0010602969,0.0016494422,0.0010432845,0.00084438367,0.004455709],"category_scores_gemma":[0.0055622864,0.0003923072,0.00050029537,0.0012738998,0.00073010527,0.0018962494,0.0013798986,0.0013082419,0.0030531597],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00033910325,0.00014289095,0.0021318088,0.0008345103,0.00014783848,0.00033566504,0.0009068079,0.059624493,0.040353898,0.018399,0.00917442,0.86760956],"study_design_scores_gemma":[0.000119453616,0.00033090392,0.00227636,0.00020309941,0.0002044516,0.00074171176,0.00085309753,0.8251082,0.07889994,0.048974484,0.042204536,0.00008379081],"about_ca_topic_score_codex":0.0036386617,"about_ca_topic_score_gemma":0.008420602,"teacher_disagreement_score":0.004455709,"about_ca_system_score_codex":0.0008376301,"about_ca_system_score_gemma":0.0013872029,"threshold_uncertainty_score":0.01490587},"labels":[],"label_agreement":null},{"id":"W6982579360","doi":"","title":"An Internally Replicated Quasi-experimental Comparison of Checklist Perspective-based Reading of Code Documents","year":2001,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Checklist; Reading (process); Code (set theory); Source code","score_opus":0.020424758806973912,"score_gpt":0.35959205205785044,"score_spread":0.3391672932508765,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6982579360","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.97510034,0.00019498031,0.005025248,0.00018917202,0.0005552449,0.01097647,0.00050887186,0.000254677,0.007194942],"genre_scores_gemma":[0.9232949,0.0001999182,0.020035591,0.00064789865,0.00029320538,0.045110885,0.0008592226,0.00025612026,0.009302252],"study_design_codex":"randomized_trial","study_design_gemma":"nonrandomized_trial","domain_scores_codex":[0.97104,0.01488871,0.0026949688,0.0059064077,0.004583028,0.0008868479],"domain_scores_gemma":[0.7994692,0.15276928,0.01560372,0.01801296,0.010562879,0.0035821055],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.016769899,0.0012220307,0.0020719485,0.0009780797,0.0019624121,0.003196633,0.0032075013,0.002861089,0.0131536415],"category_scores_gemma":[0.13538101,0.001977487,0.0010067679,0.0006689245,0.0027249258,0.00398402,0.0021466375,0.0038090239,0.0031604352],"study_design_candidate":"nonrandomized_trial","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.54348284,0.23967816,0.012776981,0.0036244055,0.0014622689,0.00019851599,0.029565051,0.0021654442,0.060614884,0.00513371,0.004349313,0.096948415],"study_design_scores_gemma":[0.24811883,0.55472213,0.13403822,0.0007793934,0.0021896926,0.00014654866,0.0059488984,0.009320843,0.024184069,0.009458998,0.0103831,0.0007091772],"about_ca_topic_score_codex":0.0029532206,"about_ca_topic_score_gemma":0.0029925387,"teacher_disagreement_score":0.016769899,"about_ca_system_score_codex":0.002125666,"about_ca_system_score_gemma":0.0034936208,"threshold_uncertainty_score":0.08868879},"labels":[],"label_agreement":null},{"id":"W6987191832","doi":"","title":"Self-training for Machine Translation","year":2006,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"","score_opus":0.019854267366581056,"score_gpt":0.27291077881635983,"score_spread":0.25305651144977875,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6987191832","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.026735809,0.0012455689,0.9288609,0.0008257473,0.001206425,0.00019715,0.0014956552,0.023980228,0.015452536],"genre_scores_gemma":[0.24367861,0.0006883838,0.70531195,0.0005498653,0.0005524714,0.00038335825,0.009425213,0.003139072,0.036271114],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9982589,0.00081820873,0.0001271355,0.00037962777,0.00027857616,0.0001376117],"domain_scores_gemma":[0.995349,0.0020276485,0.00014224304,0.0015213551,0.00086787593,0.000091921196],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020317596,0.0009252632,0.0008995956,0.0013547455,0.0011042993,0.0011110536,0.0011560968,0.0013537529,0.026434101],"category_scores_gemma":[0.0065686475,0.0007025379,0.0010442378,0.0014539154,0.00051884446,0.0028634365,0.0014625527,0.0017467096,0.016435267],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037467683,0.00020738956,0.0008413443,0.0003033693,0.00012509857,0.00019345328,0.000108502965,0.015591004,0.018136488,0.007856223,0.05990953,0.8963529],"study_design_scores_gemma":[0.00009254473,0.00026989583,0.0017514245,0.00010829216,0.000119054945,0.0004856309,0.0001119929,0.8498809,0.07251771,0.032391448,0.042225327,0.00004579667],"about_ca_topic_score_codex":0.0011791415,"about_ca_topic_score_gemma":0.0024246802,"teacher_disagreement_score":0.026434101,"about_ca_system_score_codex":0.0003337369,"about_ca_system_score_gemma":0.0008783646,"threshold_uncertainty_score":0.08843088},"labels":[],"label_agreement":null},{"id":"W6995474741","doi":"","title":"Optimising grammars using shape annealing","year":2003,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"","score_opus":0.026376754019042218,"score_gpt":0.28414669310592605,"score_spread":0.25776993908688384,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6995474741","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06145305,0.00030416183,0.9014843,0.0007884623,0.00037155967,0.00019202191,0.0002833119,0.0074227327,0.027700467],"genre_scores_gemma":[0.34429818,0.00016088362,0.63252425,0.00023162723,0.0001004557,0.00011650486,0.00084253587,0.0032381436,0.018487405],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9991757,0.00027879718,0.000037041384,0.00017837115,0.00025004594,0.00007999436],"domain_scores_gemma":[0.9982388,0.00087504653,0.00005420491,0.0004933237,0.0002877054,0.000050861978],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00086516445,0.00050548353,0.0009378958,0.00057499,0.0006038524,0.0012807065,0.0012530396,0.0012713633,0.01551387],"category_scores_gemma":[0.0051569757,0.0006913616,0.0014108832,0.0006256227,0.0010615143,0.0018124309,0.0014361126,0.0016060339,0.004145035],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005190103,0.00012901088,0.0015101159,0.0004976287,0.00012347153,0.0005103803,0.00052980386,0.30323964,0.039102662,0.09925292,0.022577152,0.5320082],"study_design_scores_gemma":[0.00008673311,0.00010604876,0.00045709038,0.000048374113,0.00006804301,0.0001682793,0.00012684568,0.79644555,0.022350833,0.1630273,0.017085562,0.000029298752],"about_ca_topic_score_codex":0.0013076738,"about_ca_topic_score_gemma":0.0024736654,"teacher_disagreement_score":0.01551387,"about_ca_system_score_codex":0.0005609382,"about_ca_system_score_gemma":0.00081376627,"threshold_uncertainty_score":0.051899076},"labels":[],"label_agreement":null},{"id":"W6996533661","doi":"","title":"A schema &amp; constraint-based representation to understanding natural language","year":2010,"lang":"en","type":"article","venue":"cIRcle (University of British Columbia)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Parsing; Ambiguity; Semantics (computer science); Natural language; Schema (genetic algorithms); Syntax; Consistency (knowledge bases); Interpretation (philosophy); Natural language understanding; Backtracking","score_opus":0.01609539966006225,"score_gpt":0.23325253219536202,"score_spread":0.21715713253529978,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6996533661","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0029797663,0.0006992612,0.9874895,0.0017983903,0.00005413328,0.00007943133,0.0003455065,0.0004143252,0.0061395704],"genre_scores_gemma":[0.046070844,0.0012213474,0.9482922,0.0003242622,0.00005624073,0.00023157979,0.0009978429,0.00012833302,0.0026773121],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99926966,0.00030093334,0.00006609311,0.00013935062,0.00018651636,0.000037389822],"domain_scores_gemma":[0.9989636,0.00047310797,0.00009256752,0.0002393821,0.0001890983,0.000042252086],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015920921,0.00051224494,0.000398538,0.0013482891,0.0008147443,0.0028643338,0.002393642,0.0010398914,0.0056586745],"category_scores_gemma":[0.0044806846,0.0005301362,0.0014182789,0.002911096,0.0022849818,0.0058819912,0.001313752,0.002160179,0.0010966543],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000020182757,0.000029613417,0.0002803984,0.00019406967,0.000037862956,0.00017190161,0.0009983865,0.017830081,0.0013521838,0.909341,0.005590883,0.06415354],"study_design_scores_gemma":[0.00002575484,0.000028682676,0.00031558154,0.00019981113,0.000050153358,0.00027406463,0.0005210886,0.21533868,0.0030759745,0.67978305,0.10034832,0.000038815102],"about_ca_topic_score_codex":0.012528156,"about_ca_topic_score_gemma":0.011404191,"teacher_disagreement_score":0.012528156,"about_ca_system_score_codex":0.0016019546,"about_ca_system_score_gemma":0.0020238862,"threshold_uncertainty_score":0.02491045},"labels":[],"label_agreement":null},{"id":"W6998601894","doi":"","title":"Automated crossword solver using GPT-4 and logic constraints","year":2024,"lang":"en","type":"dissertation","venue":"eScholarship@McGill (McGill)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Solver; Problem solver; Automation; Object (grammar)","score_opus":0.02073267663534066,"score_gpt":0.2924350120671077,"score_spread":0.27170233543176703,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6998601894","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.032283735,0.0010563759,0.87996966,0.0029096738,0.0005649877,0.0007577572,0.00339067,0.027466942,0.051600184],"genre_scores_gemma":[0.13111334,0.0004113008,0.84619725,0.0010995335,0.00010433237,0.0004084908,0.0049774526,0.0026582652,0.013030085],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99820876,0.00043154883,0.00015026786,0.00042404793,0.0005240864,0.000261274],"domain_scores_gemma":[0.9965004,0.0021820539,0.00016294955,0.00041921026,0.00059194287,0.00014347822],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016181513,0.0016394218,0.0011408721,0.0013822075,0.0012114091,0.0032623378,0.0026803303,0.0019719386,0.038213257],"category_scores_gemma":[0.008822789,0.00081106444,0.002521056,0.001566107,0.001356053,0.0033565846,0.0045385505,0.003195413,0.0063880086],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004975307,0.00054681004,0.003441898,0.0017344317,0.00020470345,0.0018059219,0.00062434626,0.1902159,0.008014805,0.17311202,0.08996842,0.5298332],"study_design_scores_gemma":[0.00033015228,0.00014218576,0.00041812006,0.00021040524,0.000089195404,0.0005635485,0.00038158448,0.76380837,0.008247792,0.16363114,0.062120378,0.000057123398],"about_ca_topic_score_codex":0.008938403,"about_ca_topic_score_gemma":0.017746286,"teacher_disagreement_score":0.038213257,"about_ca_system_score_codex":0.0020110244,"about_ca_system_score_gemma":0.0062704664,"threshold_uncertainty_score":0.12783605},"labels":[],"label_agreement":null},{"id":"W6998680185","doi":"","title":"Application des fonctions d'influence en traitement automatique du langage","year":2024,"lang":"fr","type":"other","venue":"Archipelago (University of Quebec in Montreal)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Government of Canada","keywords":"Context (archaeology); Democratic legitimacy; Information scientist","score_opus":0.00749396244531786,"score_gpt":0.21968412965547165,"score_spread":0.2121901672101538,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W6998680185","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.24527365,0.004811588,0.7141943,0.00075175724,0.0005300483,0.00020382124,0.0006575648,0.009913496,0.023663785],"genre_scores_gemma":[0.89732915,0.0009771726,0.091815576,0.00014540022,0.00012360593,0.00014620573,0.0004153889,0.0010887877,0.00795853],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99737036,0.0005473731,0.00013608793,0.0006535822,0.0011450188,0.00014756208],"domain_scores_gemma":[0.9901157,0.0062884726,0.0007264189,0.00075983454,0.0019132092,0.00019643869],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002432354,0.001327495,0.00072458433,0.0015682158,0.00060612796,0.0023666385,0.00066109136,0.00098181,0.0047414135],"category_scores_gemma":[0.014837077,0.00038660405,0.0011465641,0.0010487839,0.0008269306,0.0018145578,0.00080604944,0.0009568016,0.0015342523],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013822436,0.00020037904,0.02192957,0.0008891249,0.00023375468,0.0007310245,0.0013105466,0.15298,0.1586306,0.014144795,0.0043113525,0.6432566],"study_design_scores_gemma":[0.000059333674,0.00096367364,0.039478827,0.00018792435,0.0003215384,0.0012326549,0.0004872258,0.76208776,0.1462355,0.013019133,0.03567187,0.00025448142],"about_ca_topic_score_codex":0.0052344324,"about_ca_topic_score_gemma":0.0035655352,"teacher_disagreement_score":0.0052344324,"about_ca_system_score_codex":0.0011342247,"about_ca_system_score_gemma":0.0007969539,"threshold_uncertainty_score":0.01586157},"labels":[],"label_agreement":null},{"id":"W7001102283","doi":"","title":"An Indigenous Post-Election Post-Mortem (Pt. 1)","year":2019,"lang":"en","type":"other","venue":"Bulletin of Miscellaneous Information (Royal Gardens Kew)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Indigenous; Wonder; Government (linguistics); Theme (computing); Power (physics); The arts","score_opus":0.003695148618207751,"score_gpt":0.2060143513672918,"score_spread":0.20231920274908405,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7001102283","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02982671,0.0032721264,0.00085214106,0.03490464,0.02797451,0.00025221068,0.00677735,0.00059610227,0.8955442],"genre_scores_gemma":[0.016935186,0.0003828376,0.00013052826,0.0012007098,0.0009202423,0.000028295288,0.00045991427,0.00014275122,0.97979945],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99954385,0.00005583368,0.000011022271,0.00005396475,0.00019225996,0.00014290599],"domain_scores_gemma":[0.9994246,0.000102254366,0.000035483343,0.00006558399,0.00018618484,0.00018588555],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00068862253,0.00023391335,0.00023148613,0.0007386421,0.005534558,0.002989294,0.0006042571,0.0013535783,0.1965],"category_scores_gemma":[0.003329734,0.00021526158,0.0001860513,0.0009869759,0.00061574584,0.0014204767,0.0023927449,0.002693613,0.050428126],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000514423,0.000026952463,0.0006696732,0.00010103169,0.0000032211467,0.0007406021,0.0048517305,0.000017545095,0.00055687217,0.0057027033,0.961641,0.025637068],"study_design_scores_gemma":[0.000001285741,0.000009527078,0.0018342028,0.000032042248,0.0000011250789,0.00005381313,0.0014171228,0.00001041133,0.00015073293,0.000182649,0.99630404,0.0000030927504],"about_ca_topic_score_codex":0.020856421,"about_ca_topic_score_gemma":0.08973063,"teacher_disagreement_score":0.1965,"about_ca_system_score_codex":0.0014688874,"about_ca_system_score_gemma":0.0012241692,"threshold_uncertainty_score":0.657358},"labels":[],"label_agreement":null},{"id":"W7001544986","doi":"","title":"La gestión de procesos de pedidos en el sistema SAP y su influencia en la atención al cliente de la empresa La Viga S.A. en el último trimestre del año 2016 en Lima","year":2017,"lang":"en","type":"dissertation","venue":"renati","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Order (exchange); Process (computing); Customer service; Service (business); Quarter (Canadian coin); Management system","score_opus":0.006262090785762393,"score_gpt":0.3335627236594802,"score_spread":0.3273006328737178,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7001544986","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.82905984,0.0046176477,0.026425257,0.0072714556,0.00013485062,0.00010514524,0.00030283083,0.000630789,0.13145222],"genre_scores_gemma":[0.95500386,0.0037642193,0.011736276,0.0003329632,0.000063659805,0.000053957603,0.00025730662,0.00009115507,0.028696543],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99924785,0.00028897068,0.000021170812,0.00012074949,0.0002356408,0.000085528314],"domain_scores_gemma":[0.99776554,0.001374196,0.0002673593,0.000120013065,0.00032756708,0.00014534772],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014925927,0.00026720422,0.00013782695,0.00058092416,0.00072057155,0.0045687607,0.00030937837,0.0006916068,0.00383492],"category_scores_gemma":[0.0040962454,0.00020422728,0.00028075758,0.00074018916,0.0013943937,0.0019678148,0.0008752733,0.0010390423,0.00081763556],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005392491,0.0005841231,0.07036223,0.0008389141,0.000054692016,0.001226143,0.06979404,0.0074366536,0.02463355,0.17748179,0.018127495,0.62892115],"study_design_scores_gemma":[0.000082616156,0.00055522256,0.35823962,0.0014119422,0.00016101029,0.0009494508,0.07006036,0.04655677,0.023147367,0.12662295,0.37199587,0.00021671304],"about_ca_topic_score_codex":0.011632066,"about_ca_topic_score_gemma":0.011147603,"teacher_disagreement_score":0.011632066,"about_ca_system_score_codex":0.0019478108,"about_ca_system_score_gemma":0.002095125,"threshold_uncertainty_score":0.023128688},"labels":[],"label_agreement":null},{"id":"W7001590663","doi":"","title":"La lecture romanesque comme déclencheur de l'écriture poétique : développement, validation et essai d'un outil pédagogique et papier-crayon pour favoriser l'expression des émotions chez les adolescents","year":2006,"lang":"fr","type":"other","venue":"Constellation (Université du Québec à Chicoutimi)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Context (archaeology); Pre school; Child abuse","score_opus":0.018328438107443015,"score_gpt":0.24580536122645158,"score_spread":0.22747692311900858,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7001590663","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.34733257,0.025787162,0.09485991,0.08525507,0.014592677,0.0020176074,0.00023296119,0.000845844,0.42907625],"genre_scores_gemma":[0.6981364,0.016762495,0.04189436,0.008716328,0.0014330188,0.0018535007,0.0001695136,0.00052238296,0.230512],"study_design_codex":"qualitative","study_design_gemma":"not_applicable","domain_scores_codex":[0.9913795,0.0059513343,0.00020528559,0.00055328297,0.0013064591,0.0006041458],"domain_scores_gemma":[0.98957354,0.005499219,0.00061857124,0.00081253075,0.001613011,0.0018831807],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009911685,0.0008274391,0.0005452218,0.0007957742,0.0044511533,0.007389743,0.0014626816,0.0017360342,0.0093898885],"category_scores_gemma":[0.014451621,0.00034455088,0.00068160973,0.0005162275,0.008044044,0.0037595085,0.006131213,0.008759015,0.002492353],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019781408,0.0007507585,0.0035832212,0.0018002376,0.000022323773,0.0010329434,0.40127158,0.00044741132,0.007504647,0.20629595,0.056113444,0.32097974],"study_design_scores_gemma":[0.000051848427,0.0006254856,0.005727528,0.0024676067,0.00004757298,0.0011732824,0.112925075,0.00053809123,0.0070078676,0.010000276,0.85933834,0.00009717131],"about_ca_topic_score_codex":0.0035109362,"about_ca_topic_score_gemma":0.008087982,"teacher_disagreement_score":0.99648905,"about_ca_system_score_codex":0.003952121,"about_ca_system_score_gemma":0.009424472,"threshold_uncertainty_score":0.05241865},"labels":[],"label_agreement":null},{"id":"W7001760504","doi":"","title":"Le partage de la technologie, plus qu'une solution de rechange","year":2001,"lang":"fr","type":"other","venue":"Library and Archives Canada (Government of Canada)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Interpretation (philosophy); Action (physics); Focus (optics); Subject (documents); Context (archaeology)","score_opus":0.005344067551524349,"score_gpt":0.17342409465347702,"score_spread":0.16808002710195266,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7001760504","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015653364,0.017938446,0.42351514,0.0921174,0.004928348,0.00028949184,0.0013836685,0.006017505,0.4381567],"genre_scores_gemma":[0.09477826,0.015725601,0.19713292,0.006854398,0.0011842586,0.00019958844,0.0016701154,0.0014768447,0.68097806],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9949043,0.0010003962,0.00018312667,0.0008821619,0.0024574837,0.0005726752],"domain_scores_gemma":[0.99640286,0.0008204827,0.00013921877,0.0009863587,0.0014342725,0.00021680324],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037305893,0.0013469382,0.00078168965,0.002024257,0.004881879,0.015095489,0.0018914082,0.003838671,0.038718514],"category_scores_gemma":[0.0061705005,0.0008582924,0.0011233306,0.003156388,0.005344749,0.013177647,0.0039462373,0.0038318334,0.013539452],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013286524,0.00007894152,0.0008894187,0.0005666019,0.000043196527,0.00039398498,0.0033502874,0.0016389269,0.009913542,0.44648573,0.12571645,0.41079006],"study_design_scores_gemma":[0.00001439156,0.000023877124,0.00067103945,0.00021148002,0.000026615122,0.00043346515,0.0014058502,0.0025950824,0.0035331664,0.05250535,0.9385409,0.000038822986],"about_ca_topic_score_codex":0.12174661,"about_ca_topic_score_gemma":0.11172269,"teacher_disagreement_score":0.12174661,"about_ca_system_score_codex":0.0072093853,"about_ca_system_score_gemma":0.015545496,"threshold_uncertainty_score":0.24207592},"labels":[],"label_agreement":null},{"id":"W7001902188","doi":"","title":"Mackling &amp; Megarry - Friday, March 3rd, 2017","year":2017,"lang":"en","type":"other","venue":"Bulletin of Miscellaneous Information (Royal Gardens Kew)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Nightlife; Banquet; Happening; Yesterday; George (robot); Personality","score_opus":0.010518434348343927,"score_gpt":0.23049847442172297,"score_spread":0.21998004007337904,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7001902188","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0009153668,0.00064535864,0.00037265132,0.00346778,0.0030457687,0.00014506264,0.0021681066,0.0016688162,0.9875711],"genre_scores_gemma":[0.0010599351,0.0001832434,0.00009348784,0.00037539558,0.00008275528,0.000019337102,0.000282806,0.00030212136,0.99760085],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.999676,0.00002413222,0.000009587524,0.00006704385,0.00012448037,0.00009878832],"domain_scores_gemma":[0.9992188,0.000034813067,0.000024233033,0.000039991413,0.0002926109,0.00038955206],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.00040014044,0.00090986665,0.00045288083,0.0008553365,0.004404579,0.0042931503,0.00081869937,0.0015713255,0.8496984],"category_scores_gemma":[0.0014981063,0.0005581952,0.00038296598,0.00054916844,0.0005006691,0.0026661556,0.0030639688,0.0022083467,0.71035504],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000021285412,0.000010472966,0.00009276469,0.000035672954,0.0000010065296,0.00006585533,0.000113921036,0.0000072293687,0.00012725353,0.00047232897,0.98425967,0.014792411],"study_design_scores_gemma":[0.0000029618534,0.0000073735678,0.00044338708,0.00004462592,7.66703e-7,0.000037636393,0.0003313318,0.000012815766,0.000073699084,0.0000807206,0.99896014,0.0000045938655],"about_ca_topic_score_codex":0.020344201,"about_ca_topic_score_gemma":0.12020646,"teacher_disagreement_score":0.15030158,"about_ca_system_score_codex":0.0013581909,"about_ca_system_score_gemma":0.0012840098,"threshold_uncertainty_score":0.21438688},"labels":[],"label_agreement":null},{"id":"W7001906623","doi":"","title":"Local Sustainability Partnerships: Understanding the Relationship Between Partnership Structural Features and Partners’ Outcomes","year":2020,"lang":"en","type":"dissertation","venue":"UWSpace (University of Waterloo)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"Strong","keywords":"General partnership; Sustainability; Civil society; Value (mathematics); Qualitative research; Sustainable development; Qualitative property","score_opus":0.05681418010009724,"score_gpt":0.2949970003442257,"score_spread":0.23818282024412846,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7001906623","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9371689,0.000738411,0.00480323,0.006021714,0.00002087595,0.00009642334,0.00009026753,0.000029936577,0.051030207],"genre_scores_gemma":[0.99904245,0.00016057886,0.00029450646,0.00007265456,0.0000020667337,0.00003546947,0.00002047072,0.0000033446158,0.00036845342],"study_design_codex":"qualitative","study_design_gemma":"observational","domain_scores_codex":[0.99211615,0.005763361,0.00021097931,0.0003942319,0.0005940248,0.0009211612],"domain_scores_gemma":[0.9847511,0.0076800585,0.0028993622,0.00044443278,0.0015674762,0.002657508],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009923588,0.00027622926,0.00040181895,0.0021790017,0.0029700694,0.007257131,0.0012335071,0.00090913783,0.0052377684],"category_scores_gemma":[0.020910548,0.00024326115,0.00030798992,0.002436575,0.004539119,0.00937459,0.011216661,0.001692836,0.00031973963],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012455658,0.00091011386,0.35810408,0.00087069976,0.0001457947,0.0016797316,0.4273573,0.0024965757,0.0005928689,0.07768568,0.0025884472,0.12744415],"study_design_scores_gemma":[0.000018441953,0.00023598474,0.12183515,0.0005460531,0.00005810176,0.00024118548,0.82944065,0.002564796,0.0003165696,0.032515258,0.012187226,0.000040482726],"about_ca_topic_score_codex":0.006739983,"about_ca_topic_score_gemma":0.0110367695,"teacher_disagreement_score":0.009923588,"about_ca_system_score_codex":0.004896724,"about_ca_system_score_gemma":0.004233909,"threshold_uncertainty_score":0.05248159},"labels":[],"label_agreement":null},{"id":"W7002178666","doi":"","title":"Modelling and simulation of the ice accretion process on fixed or rotating cylindrical objects by the boundary element method = ModÃ©lisation et simulation des accrÃ©tions de glace atmosphÃ©rique sur des objets cylindriques fixes et tournants par la mÃ©thode des Ã©lÃ©ments finis de frontiÃ¨re","year":2004,"lang":"fr","type":"other","venue":"Library and Archives Canada (Government of Canada)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Gravitational force; Gravitational acceleration; Viscous flow; Boundary value problem","score_opus":0.016592057562599825,"score_gpt":0.2552041238972564,"score_spread":0.23861206633465656,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7002178666","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.62222666,0.00094608736,0.3527564,0.0005521643,0.00020566508,0.00017430432,0.00056492107,0.00097128237,0.021602564],"genre_scores_gemma":[0.9214096,0.00053897855,0.068149336,0.00007117208,0.000027227201,0.00019022022,0.00040500163,0.00015046552,0.009057844],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99988043,0.00003426099,0.000006175185,0.000019610292,0.00003588351,0.000023577973],"domain_scores_gemma":[0.99976224,0.00012301811,0.00003261876,0.000020318139,0.00003629147,0.000025399395],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003387706,0.00048472456,0.00055477937,0.00029415954,0.00048134194,0.0010998201,0.00086492585,0.0013100996,0.001767389],"category_scores_gemma":[0.0007175544,0.0003872307,0.0006969131,0.00035593638,0.00074788794,0.00063946133,0.0005572137,0.00049748726,0.00030077776],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00004364258,0.000023071747,0.0009105607,0.00003712293,0.000013911171,0.0000732257,0.000058231213,0.9909292,0.0031094495,0.0017492013,0.00013701685,0.0029153759],"study_design_scores_gemma":[0.000009087952,0.000013658021,0.00022170857,0.000003940828,0.000002409678,0.000010455981,0.000012106234,0.99844366,0.00059138355,0.00024493044,0.00044251842,0.0000041433736],"about_ca_topic_score_codex":0.013565014,"about_ca_topic_score_gemma":0.007238422,"teacher_disagreement_score":0.013565014,"about_ca_system_score_codex":0.00057568337,"about_ca_system_score_gemma":0.00092563144,"threshold_uncertainty_score":0.026972115},"labels":[],"label_agreement":null},{"id":"W7008088209","doi":"","title":"Beyond maximum-likelihood training: analysis and methods for building robust language generation models","year":2025,"lang":"en","type":"dissertation","venue":"eScholarship@McGill (McGill)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Natural language generation; Natural language; Key (lock); Matching (statistics); Abstraction","score_opus":0.032533974734453075,"score_gpt":0.32308123562433566,"score_spread":0.2905472608898826,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7008088209","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.003576829,0.00040144692,0.99388796,0.00027713855,0.000030462092,0.00003137004,0.00011795057,0.0012032209,0.0004735626],"genre_scores_gemma":[0.18098354,0.0007831333,0.80705905,0.0005904882,0.0002559987,0.00044451933,0.0019895574,0.0022991286,0.005594521],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9959111,0.0023517231,0.00019082459,0.0007755419,0.000609197,0.00016171613],"domain_scores_gemma":[0.95376647,0.04159915,0.00074783375,0.0019632066,0.0016289622,0.00029443647],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00847161,0.0013418512,0.0017954252,0.0016964596,0.0011190525,0.0026624785,0.0031240315,0.002742548,0.006066276],"category_scores_gemma":[0.046576425,0.0016699122,0.0017623696,0.0017571478,0.0014429917,0.0056106644,0.0026192141,0.004808231,0.0024720328],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036862394,0.0001667102,0.0025736126,0.00033343217,0.00023591211,0.0001572356,0.00034363067,0.4036215,0.0039007477,0.02483384,0.013841689,0.5496231],"study_design_scores_gemma":[0.000013563899,0.00001840924,0.0002496274,0.000018300818,0.000021857844,0.00002626248,0.000014016423,0.974019,0.0015855085,0.023053262,0.00096842163,0.000011766486],"about_ca_topic_score_codex":0.0058118757,"about_ca_topic_score_gemma":0.0052576642,"teacher_disagreement_score":0.00847161,"about_ca_system_score_codex":0.0010206335,"about_ca_system_score_gemma":0.00142445,"threshold_uncertainty_score":0.044802666},"labels":[],"label_agreement":null},{"id":"W7008486421","doi":"","title":"CESTA Evaluation Package","year":2007,"lang":"en","type":"other","venue":"Americanae (AECID Library)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Terminology; Arabic; Christian ministry; Section (typography); Machine translation; Domain (mathematical analysis)","score_opus":0.011912404685538605,"score_gpt":0.27397748056071514,"score_spread":0.2620650758751765,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7008486421","genre_codex":"dataset","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0058483817,0.001707946,0.11590348,0.0033715086,0.0023803858,0.01075527,0.43716246,0.12997505,0.29289547],"genre_scores_gemma":[0.025114587,0.00096781005,0.12797903,0.002371631,0.0007080079,0.026400471,0.6431215,0.054829024,0.118507884],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.985789,0.0060088737,0.0017483709,0.001572495,0.0042025237,0.00067868124],"domain_scores_gemma":[0.9551694,0.014591981,0.0010190964,0.0060249637,0.02234427,0.0008502964],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.015797203,0.0024715036,0.0021294684,0.0055405954,0.0017357747,0.0065496503,0.00434103,0.0022383311,0.29132718],"category_scores_gemma":[0.056585412,0.0014248413,0.0030316776,0.005151787,0.00085183623,0.005184767,0.004314231,0.003315609,0.24484193],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006422453,0.00012398322,0.0005112033,0.0014244526,0.0001243478,0.00009750524,0.00029345023,0.0014665544,0.0007714149,0.0058535184,0.92644036,0.06225094],"study_design_scores_gemma":[0.0004096264,0.000148125,0.0017029388,0.00030320033,0.000108700755,0.00016911163,0.00017362248,0.0035253877,0.0018268996,0.0067018364,0.98483014,0.000100411504],"about_ca_topic_score_codex":0.007541825,"about_ca_topic_score_gemma":0.0076211127,"teacher_disagreement_score":0.29132718,"about_ca_system_score_codex":0.002918202,"about_ca_system_score_gemma":0.005790864,"threshold_uncertainty_score":0.9745865},"labels":[],"label_agreement":null},{"id":"W7009013779","doi":"","title":"On the compute and parameter efficient fine-tuning of large language models","year":2023,"lang":"en","type":"dissertation","venue":"eScholarship@McGill (McGill)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Language model; Natural language; Identification (biology); Feature (linguistics); Set (abstract data type)","score_opus":0.0181016859083932,"score_gpt":0.26776125084559677,"score_spread":0.24965956493720357,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7009013779","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09919274,0.0020394789,0.85441005,0.0015019671,0.0003311607,0.00020252803,0.0009426938,0.03224731,0.009132089],"genre_scores_gemma":[0.47576252,0.0010915255,0.50943226,0.0010299379,0.00013309252,0.00037639518,0.0036330684,0.0025346463,0.0060065608],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990989,0.00030814053,0.000053958585,0.00024054466,0.0001991306,0.00009938274],"domain_scores_gemma":[0.9973909,0.0016768198,0.000073900555,0.00060579064,0.00018370821,0.000068846115],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016803072,0.0015301709,0.00088776933,0.00055833434,0.0006405673,0.00155008,0.0018152653,0.0012830232,0.005076404],"category_scores_gemma":[0.00809949,0.0007935847,0.001181026,0.00083672634,0.0008726735,0.0037098816,0.0020411876,0.0036903378,0.0030587777],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035020363,0.00017832412,0.0015304738,0.00021137594,0.00009343191,0.00014571765,0.00014184654,0.71054184,0.010700213,0.01099836,0.016387545,0.24872065],"study_design_scores_gemma":[0.00003027185,0.000039492104,0.00017356902,0.0000118187645,0.000011359579,0.000022391656,0.000023695162,0.9869757,0.0031748826,0.0076144957,0.0019110955,0.000011087718],"about_ca_topic_score_codex":0.010667316,"about_ca_topic_score_gemma":0.018810458,"teacher_disagreement_score":0.010667316,"about_ca_system_score_codex":0.0012758856,"about_ca_system_score_gemma":0.0021942954,"threshold_uncertainty_score":0.021210492},"labels":[],"label_agreement":null},{"id":"W7011771313","doi":"","title":"A new type of typology: improving the utility of linguistic typology in natural language processing via continuous data representations","year":2024,"lang":"en","type":"dissertation","venue":"eScholarship@McGill (McGill)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"McGill University","keywords":"Typology; Natural language; Linguistic typology; Deep linguistic processing; Type (biology); Language identification","score_opus":0.024513880204966833,"score_gpt":0.324075701020664,"score_spread":0.2995618208156972,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7011771313","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022898115,0.001038289,0.9624018,0.004081323,0.0005745623,0.00029599422,0.0015435586,0.001246326,0.005919995],"genre_scores_gemma":[0.1795963,0.0011343496,0.8107633,0.0013195595,0.0006618553,0.00072880636,0.0027986404,0.0007079649,0.002289241],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9814419,0.010244419,0.0016203189,0.003852133,0.002410801,0.0004303966],"domain_scores_gemma":[0.90249556,0.05749489,0.0040176944,0.029188966,0.0054203584,0.001382428],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.017738309,0.0011721088,0.0018517422,0.006639398,0.0022196749,0.0130345,0.0037236486,0.0028001172,0.0072835386],"category_scores_gemma":[0.09952279,0.0008301028,0.0023943002,0.011745681,0.006416534,0.030200828,0.009376844,0.007952227,0.0020698907],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00052000093,0.0004727294,0.022870624,0.0010163046,0.00033651065,0.00024658866,0.004693865,0.0139751285,0.004199794,0.3353394,0.019805774,0.59652334],"study_design_scores_gemma":[0.00010598453,0.00031029515,0.0046406034,0.00064429984,0.00015087017,0.00045301107,0.0032144992,0.17355293,0.005105399,0.76000184,0.051577035,0.00024316274],"about_ca_topic_score_codex":0.0018960336,"about_ca_topic_score_gemma":0.0017600498,"teacher_disagreement_score":0.017738309,"about_ca_system_score_codex":0.0017644379,"about_ca_system_score_gemma":0.0028529547,"threshold_uncertainty_score":0.09381026},"labels":[],"label_agreement":null},{"id":"W7015809594","doi":"","title":"Using Wordnet hierarchies to pinpoint differences in related texts","year":2009,"lang":"en","type":"article","venue":"Trinity's Access to Research Output (TARA) (Trinity College Dublin)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Trinity College","funders":"","keywords":"WordNet; Graph; Representation (politics); Knowledge representation and reasoning; Taxonomy (biology)","score_opus":0.18702378037880005,"score_gpt":0.43252420527941166,"score_spread":0.2455004249006116,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7015809594","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.31054592,0.0026585844,0.61856264,0.0017010828,0.00042714286,0.00110223,0.01182471,0.01693225,0.036245465],"genre_scores_gemma":[0.4481891,0.0008365571,0.53286934,0.00022545949,0.00009884845,0.0003477114,0.010374193,0.0010762068,0.0059825606],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99799687,0.00047172382,0.0002335299,0.0005448968,0.00064279186,0.00011015382],"domain_scores_gemma":[0.9919194,0.0044159945,0.001176287,0.000777559,0.0013239824,0.0003866986],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025450604,0.00069998496,0.000539275,0.016258437,0.0014848087,0.0032877456,0.00081211043,0.0007860295,0.0065641315],"category_scores_gemma":[0.011199819,0.00040411705,0.000498125,0.008133779,0.0010528102,0.0076483926,0.0022768194,0.00085897837,0.0018511685],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009746386,0.0003050388,0.030390522,0.0018833021,0.00024732898,0.0010994386,0.01357997,0.0042460347,0.08644125,0.08906663,0.016690465,0.7550753],"study_design_scores_gemma":[0.00024006721,0.00073618395,0.098594666,0.0014985254,0.0007617582,0.0029341006,0.018071255,0.14016758,0.07660868,0.39737642,0.26254797,0.00046292724],"about_ca_topic_score_codex":0.0050749015,"about_ca_topic_score_gemma":0.010859338,"teacher_disagreement_score":0.016258437,"about_ca_system_score_codex":0.001173114,"about_ca_system_score_gemma":0.0014830007,"threshold_uncertainty_score":0.021959245},"labels":[],"label_agreement":null},{"id":"W7016033350","doi":"","title":"USE ATYPICAL ASPHALT BINDERS FROM ALBERTA OILSAND SOURCES FOR THE EFFECTIVE RECYCLING OF ASPHALT PAVEMENT","year":2021,"lang":"en","type":"dissertation","venue":"QSpace (Queen's University Library)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Queen's University","funders":"","keywords":"Asphalt; Dynamic shear rheometer; Asphalt pavement; Rheology; Diesel fuel; Rheometer; Polyethylene terephthalate; Compaction","score_opus":0.009142226607916352,"score_gpt":0.22017990748365332,"score_spread":0.21103768087573696,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7016033350","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9985084,0.0001941347,0.00035830116,0.000008241653,0.0000035050716,0.0000141798855,0.000058061563,0.000020006495,0.00083525927],"genre_scores_gemma":[0.99473345,0.0003897966,0.0024938085,0.000011434703,0.0000017423315,0.000013004705,0.00014411275,0.000015791453,0.002196856],"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99971646,0.00001383881,0.000016362215,0.000041647156,0.00014889693,0.00006285644],"domain_scores_gemma":[0.9998987,0.000008468181,0.000030576193,0.000011599926,0.000034737557,0.000015919886],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00017786838,0.0004083856,0.00023909296,0.0004924189,0.00050777884,0.0004993561,0.00041741604,0.00020217492,0.0009878408],"category_scores_gemma":[0.0001727408,0.00013538994,0.0002502922,0.00043903256,0.00018421651,0.00025475686,0.00027878134,0.000292812,0.00021525868],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013544917,0.000063183325,0.0014951613,0.000072475836,0.000010885718,0.00009115648,0.00007157069,0.00037973482,0.99058586,0.00003787409,0.000033199896,0.0070234737],"study_design_scores_gemma":[0.0000053654253,0.00032175565,0.009535956,0.00000848679,0.000037936945,0.00007015906,0.00009110508,0.00039031584,0.98778,0.000010773903,0.0017388713,0.000009197533],"about_ca_topic_score_codex":0.029277349,"about_ca_topic_score_gemma":0.13952701,"teacher_disagreement_score":0.029277349,"about_ca_system_score_codex":0.0007950762,"about_ca_system_score_gemma":0.0006828828,"threshold_uncertainty_score":0.05821389},"labels":[],"label_agreement":null},{"id":"W7019241597","doi":"","title":"Exploring the limits of systematicity of natural language understanding models","year":2023,"lang":"en","type":"dissertation","venue":"eScholarship@McGill (McGill)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Natural language; Natural (archaeology); Normative; Identification (biology); Process (computing)","score_opus":0.11296467123599746,"score_gpt":0.29402376062490276,"score_spread":0.1810590893889053,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7019241597","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06722373,0.0019538752,0.91055876,0.009764699,0.00009210873,0.00017734672,0.0003899175,0.0020474044,0.007792217],"genre_scores_gemma":[0.68692505,0.001501004,0.30460665,0.0013760541,0.0002098671,0.00041505392,0.0012313954,0.001470357,0.0022646084],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9618566,0.02811571,0.001642402,0.0035385108,0.0039922646,0.00085452857],"domain_scores_gemma":[0.5667008,0.39167893,0.0044228234,0.028959531,0.0069046086,0.0013333213],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04099776,0.0009575325,0.0015845154,0.002234188,0.0020221833,0.008543842,0.0033122275,0.0024448691,0.004129326],"category_scores_gemma":[0.19869307,0.0031968346,0.0027668641,0.0019310308,0.0071445205,0.037397157,0.009961298,0.007938242,0.0007828082],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005404978,0.0003705543,0.011520139,0.0007215629,0.00045055684,0.00020994392,0.005779314,0.071763754,0.0028347932,0.7208302,0.0053965,0.17958231],"study_design_scores_gemma":[0.000050958366,0.00007632355,0.0005362671,0.000113733906,0.00012047606,0.000081688384,0.0005292274,0.29159036,0.0017405703,0.70048064,0.0046528596,0.000026943342],"about_ca_topic_score_codex":0.009662762,"about_ca_topic_score_gemma":0.0110134715,"teacher_disagreement_score":0.04099776,"about_ca_system_score_codex":0.0034544556,"about_ca_system_score_gemma":0.0059591024,"threshold_uncertainty_score":0.21681947},"labels":[],"label_agreement":null},{"id":"W7020917211","doi":"","title":"Minicentrale hydroélectrique de Val-Jalbert : influence de l’échelle de participation publique et acceptabilité sociale","year":2014,"lang":"fr","type":"other","venue":"Knowledge UdeS (Institutional Deposit of the University of Sherbrooke)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Public work; Conciliation; Skill mix","score_opus":0.010023054582166707,"score_gpt":0.23765819632826662,"score_spread":0.22763514174609992,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7020917211","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9758076,0.00035416728,0.00024306025,0.0014725417,0.0000144062615,0.000033772183,0.000095475865,0.000012026291,0.021966936],"genre_scores_gemma":[0.97912985,0.0002771996,0.00022293358,0.0003027672,0.0000069356,0.00003587269,0.00011172013,0.000008950664,0.01990375],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99930596,0.00016256182,0.000016629067,0.00010483726,0.00016945721,0.00024051881],"domain_scores_gemma":[0.9976205,0.00062755076,0.00030422234,0.00006804659,0.0004132238,0.0009664156],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010186965,0.00014910895,0.00021394342,0.0006786379,0.0033735633,0.0027000087,0.00062989426,0.00043445942,0.0090852985],"category_scores_gemma":[0.002848028,0.00020141088,0.00011516727,0.0007328602,0.0016020066,0.0007666324,0.0019644129,0.00072338665,0.0004807101],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000764118,0.0005983908,0.6103977,0.00032792124,0.00013073291,0.0023624746,0.19802497,0.0009938217,0.012699269,0.026383363,0.011901405,0.13541582],"study_design_scores_gemma":[0.000022836983,0.00012319189,0.85356796,0.00015880742,0.000023645036,0.00015304801,0.08618555,0.0005545483,0.0006774875,0.0006264666,0.057872232,0.000034219003],"about_ca_topic_score_codex":0.6654559,"about_ca_topic_score_gemma":0.8797036,"teacher_disagreement_score":0.33454412,"about_ca_system_score_codex":0.008336637,"about_ca_system_score_gemma":0.010304944,"threshold_uncertainty_score":0.6730286},"labels":[],"label_agreement":null},{"id":"W7021147362","doi":"","title":"New coconut opportunities","year":2000,"lang":"en","type":"other","venue":"CGSPace A Repository of Agricultural Research Outputs (Consultative Group for International Agricultural Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Process (computing); Exportation; Work (physics); Cold storage","score_opus":0.07349084348457322,"score_gpt":0.36225023288424996,"score_spread":0.2887593893996767,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7021147362","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0037887471,0.016206164,0.00422651,0.010631885,0.0035792508,0.0001458246,0.0017262694,0.0015932429,0.95810217],"genre_scores_gemma":[0.023037313,0.019316789,0.008038677,0.0045162393,0.0015897889,0.00018786937,0.004875762,0.0015886553,0.9368488],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.998762,0.00015601293,0.000028293609,0.00022366905,0.00055309176,0.00027697862],"domain_scores_gemma":[0.998359,0.00020805521,0.00007504881,0.00018424621,0.00037374493,0.00079989486],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018029166,0.0012122402,0.00038674672,0.0028484815,0.0020686197,0.007299806,0.001415204,0.0016922728,0.2610863],"category_scores_gemma":[0.0026140907,0.0003601419,0.00066613895,0.0036411122,0.0011465352,0.005877139,0.0046086293,0.0021354682,0.072269306],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002346871,0.00014878891,0.0005180145,0.0005671194,0.0000103871735,0.0004694228,0.00062005105,0.00008734095,0.0026507357,0.09841823,0.59910125,0.29717398],"study_design_scores_gemma":[0.0000045448423,0.000011388159,0.00014773643,0.0000592103,0.0000016151736,0.00007981017,0.000078795165,0.000027661974,0.00034894238,0.0020403801,0.9971955,0.000004410389],"about_ca_topic_score_codex":0.0053203143,"about_ca_topic_score_gemma":0.016570931,"teacher_disagreement_score":0.2610863,"about_ca_system_score_codex":0.0025748084,"about_ca_system_score_gemma":0.002079567,"threshold_uncertainty_score":0.87342066},"labels":[],"label_agreement":null},{"id":"W7021802400","doi":"","title":"Sérial Vénus","year":2009,"lang":"fr","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Lien; Relation (database); Order (exchange); Subject (documents)","score_opus":0.020915360766077692,"score_gpt":0.291186522093746,"score_spread":0.2702711613276683,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7021802400","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.036760483,0.0076029096,0.024791507,0.0045180903,0.0053028166,0.000081129714,0.002420397,0.0029051772,0.91561747],"genre_scores_gemma":[0.3597856,0.003977225,0.015151372,0.0016394552,0.0011797836,0.00007986052,0.0032065779,0.0017811543,0.613199],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99949527,0.00007040902,0.00002510219,0.00015302071,0.00017019828,0.00008598169],"domain_scores_gemma":[0.9992592,0.00016710247,0.000057367677,0.00016594947,0.00023599814,0.000114363764],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00036029075,0.00063695776,0.00043767472,0.0017298853,0.0023601276,0.002518466,0.000647509,0.0009988339,0.079420365],"category_scores_gemma":[0.0023126558,0.0003377013,0.00034446595,0.001347359,0.0010239249,0.0032506443,0.0022478986,0.0012602126,0.019209448],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007803378,0.000055025754,0.005254998,0.00054680783,0.0000622441,0.00394032,0.0031434137,0.0011037404,0.010385341,0.37451023,0.26800236,0.3322152],"study_design_scores_gemma":[0.000023758792,0.000047488025,0.0032059113,0.00020933167,0.000023232915,0.0031586187,0.0009179814,0.00081759156,0.0038846524,0.029659795,0.9580185,0.00003316517],"about_ca_topic_score_codex":0.0067571267,"about_ca_topic_score_gemma":0.012643757,"teacher_disagreement_score":0.079420365,"about_ca_system_score_codex":0.0013907586,"about_ca_system_score_gemma":0.0010816782,"threshold_uncertainty_score":0.26568758},"labels":[],"label_agreement":null},{"id":"W7022112305","doi":"","title":"Enlargement Policy","year":2018,"lang":"en","type":"book-chapter","venue":"Research Publications (Maastricht University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Montreal Council on Foreign Relations","funders":"","keywords":"Resizing; Government (linguistics); Margin (machine learning); Yield (engineering)","score_opus":0.06905925667259023,"score_gpt":0.33716698852731813,"score_spread":0.2681077318547279,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7022112305","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0016591804,0.00044040466,0.002917499,0.0028648116,0.0014407368,0.00012776381,0.0014170578,0.0009928972,0.9881396],"genre_scores_gemma":[0.02454379,0.00044417556,0.0028296958,0.00256992,0.0008718707,0.00030377592,0.0019712897,0.0011093883,0.9653562],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99623233,0.000595606,0.00015960971,0.00088595145,0.0014317,0.0006948944],"domain_scores_gemma":[0.9953459,0.0009316441,0.00017084337,0.0016438789,0.001447879,0.00045991628],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0033987632,0.0007485781,0.0010201358,0.0032627704,0.0035348043,0.0055007194,0.0025956787,0.00423666,0.4711581],"category_scores_gemma":[0.012211559,0.0006529273,0.0010576728,0.0026316275,0.0016994657,0.006722295,0.004281515,0.0038477273,0.20581828],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022334518,0.00008048965,0.00021445594,0.00016501501,0.000008873316,0.00013300884,0.00031825897,0.00013096357,0.0010884325,0.46831778,0.4570867,0.07223265],"study_design_scores_gemma":[0.000024146599,0.000017376658,0.00036861136,0.00005926531,0.000007995545,0.00008381175,0.000085984655,0.00011098933,0.0007462537,0.026673734,0.9718098,0.000012032396],"about_ca_topic_score_codex":0.004605297,"about_ca_topic_score_gemma":0.005155305,"teacher_disagreement_score":0.4711581,"about_ca_system_score_codex":0.0026901644,"about_ca_system_score_gemma":0.0034499397,"threshold_uncertainty_score":0.7543288},"labels":[],"label_agreement":null},{"id":"W7022363189","doi":"","title":"Canadian, Eh?","year":2002,"lang":"en","type":"article","venue":"DOAJ (DOAJ: Directory of Open Access Journals)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"","score_opus":0.20558419556333504,"score_gpt":0.5171250755955098,"score_spread":0.3115408800321747,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7022363189","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0011263404,0.0027770903,0.00028812033,0.0051196874,0.001316883,0.00004323109,0.0073758936,0.0004378478,0.981515],"genre_scores_gemma":[0.0030890645,0.0010788594,0.00026130196,0.0007312512,0.000049533814,0.000008841966,0.00082052726,0.00013822247,0.99382246],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99944323,0.000028137512,0.000014814224,0.00012017531,0.00022822434,0.00016537744],"domain_scores_gemma":[0.9991492,0.000051118728,0.000027586875,0.000043028143,0.00044609918,0.00028293364],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.00040133254,0.0012418163,0.00058844726,0.0021261787,0.008873141,0.005704701,0.0009419646,0.0018073639,0.7332999],"category_scores_gemma":[0.0015863752,0.00046369116,0.00048486897,0.0035157157,0.0011500486,0.0019136809,0.002096178,0.0017493172,0.29451612],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000077884455,0.000024665702,0.0008009286,0.00022433793,0.000009516234,0.00036817844,0.00027806693,0.00013378325,0.00043877008,0.017573532,0.8667818,0.11328872],"study_design_scores_gemma":[0.00000577198,0.0000025863221,0.0008991328,0.00006452001,0.0000036969918,0.00005520153,0.00032443836,0.00002038897,0.00010235139,0.00089338125,0.99762136,0.0000072287426],"about_ca_topic_score_codex":0.8168304,"about_ca_topic_score_gemma":0.9559516,"teacher_disagreement_score":0.7332999,"about_ca_system_score_codex":0.011808675,"about_ca_system_score_gemma":0.020413073,"threshold_uncertainty_score":0.38041526},"labels":[],"label_agreement":null},{"id":"W7023910871","doi":"","title":"Perceptions and Consequences of Confronting Sexism: A Multi-Method Examination of Context and Confrontier Identities","year":2023,"lang":"en","type":"other","venue":"York University Digital Library (York University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Perception; Context (archaeology); Competence (human resources); Interpersonal interaction; Interpersonal communication; Interpersonal relationship","score_opus":0.013671193261630372,"score_gpt":0.2092676285468188,"score_spread":0.1955964352851884,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7023910871","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9967571,0.000037523034,0.002278645,0.000013696791,0.000004366395,0.0002620568,0.000018271456,0.0000064995656,0.00062179845],"genre_scores_gemma":[0.98472285,0.00010822821,0.012716017,0.00006156345,0.000012702629,0.0014397267,0.000049181483,0.000010761783,0.0008788631],"study_design_codex":"observational","study_design_gemma":"qualitative","domain_scores_codex":[0.99265105,0.0056806477,0.00030619977,0.00066638755,0.00047734383,0.0002183485],"domain_scores_gemma":[0.9870863,0.008951265,0.0017588936,0.0013114351,0.0006467153,0.00024531916],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00641896,0.0005246941,0.00043361518,0.0011914116,0.0014934913,0.0011099243,0.0005042362,0.0006333813,0.0020373818],"category_scores_gemma":[0.012661569,0.00044833738,0.00038344163,0.0007354204,0.0012829993,0.0009538656,0.0024549353,0.0007799534,0.0001767192],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0037750362,0.01075836,0.32989097,0.0012759314,0.00025628976,0.0026244058,0.32687896,0.0009930584,0.1708954,0.002054984,0.00036162717,0.15023501],"study_design_scores_gemma":[0.00044708146,0.019486457,0.74383307,0.00021924268,0.00022741388,0.0020198228,0.17319706,0.004219797,0.047713175,0.002519111,0.0058887247,0.00022903517],"about_ca_topic_score_codex":0.00041395234,"about_ca_topic_score_gemma":0.0013268313,"teacher_disagreement_score":0.00641896,"about_ca_system_score_codex":0.0004354757,"about_ca_system_score_gemma":0.00037462043,"threshold_uncertainty_score":0.03394711},"labels":[],"label_agreement":null},{"id":"W7023913424","doi":"","title":"Radiotracer methods for understanding contaminant dynamics in aquatic environments","year":2017,"lang":"en","type":"other","venue":"The Australian Nuclear Science and Technology Organisation Institutional Repository (The Australian Nuclear Science and Technology Organisation)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"International Atomic Energy Agency; European Commission; Centre International de Recherche sur le Cancer","keywords":"Contamination; Tracing; TRACER; Sorption; Radionuclide; Aquatic ecosystem; Radioactive contamination; Radioactive waste; Natural (archaeology)","score_opus":0.02230178331207351,"score_gpt":0.29505189724117165,"score_spread":0.27275011392909815,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7023913424","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016107075,0.061728626,0.8964657,0.0010868746,0.0006592442,0.0005042088,0.0017151764,0.0026805664,0.019052505],"genre_scores_gemma":[0.11260651,0.0892273,0.7671018,0.0015241161,0.0004948652,0.002120929,0.0020501716,0.0005003317,0.02437391],"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99927455,0.0001266226,0.00004788073,0.0001985063,0.00030119062,0.000051331113],"domain_scores_gemma":[0.9993056,0.00023210273,0.0001267661,0.000092064896,0.00019687355,0.000046555455],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013327788,0.0009171945,0.00096431276,0.0022192313,0.0004823283,0.0016031708,0.0015295488,0.0018935592,0.0053241663],"category_scores_gemma":[0.0013931672,0.0006083031,0.00082399475,0.0024910755,0.00089321285,0.0020665152,0.0011801285,0.0025651886,0.003586546],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021055003,0.000167696,0.0034187168,0.0037355819,0.0001616352,0.00052757066,0.00033448674,0.0073679215,0.6894627,0.028642314,0.008305672,0.25766507],"study_design_scores_gemma":[0.00010829274,0.00095381576,0.008150421,0.00093811844,0.00032256235,0.0022569017,0.00066757033,0.060379244,0.4374072,0.035578933,0.45290747,0.00032945533],"about_ca_topic_score_codex":0.0022759323,"about_ca_topic_score_gemma":0.0030757484,"teacher_disagreement_score":0.0053241663,"about_ca_system_score_codex":0.0011030729,"about_ca_system_score_gemma":0.0010589558,"threshold_uncertainty_score":0.01781106},"labels":[],"label_agreement":null},{"id":"W7024175031","doi":"","title":"RÃ©cit de lâÃ©vÃ©nement et Ã©vÃ©nement du rÃ©cit chez Annie Ernaux, HÃ©lÃ¨ne Cixous et Maurice Blanchot","year":2013,"lang":"fr","type":"other","venue":"Library and Archives Canada (Government of Canada)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Nazism; Confession (law); First world war","score_opus":0.004245717851438136,"score_gpt":0.16993286711071676,"score_spread":0.16568714925927863,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7024175031","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.055197027,0.025196122,0.0035896217,0.07595133,0.005931723,0.000041543517,0.00047628517,0.00018364593,0.8334327],"genre_scores_gemma":[0.40638638,0.007745974,0.0018001077,0.008756996,0.001207853,0.00005781538,0.00017941647,0.00038791675,0.5734775],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9991316,0.00025007426,0.000016481967,0.00018538325,0.00030888707,0.000107677624],"domain_scores_gemma":[0.99848205,0.0009244827,0.00011495701,0.00010102044,0.00026998814,0.00010752575],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007311371,0.0005921518,0.00024086254,0.0010445203,0.0071380087,0.0046285824,0.0006272332,0.0021495647,0.015256336],"category_scores_gemma":[0.0035451471,0.0003312322,0.00020442835,0.0006859642,0.008671275,0.003382042,0.0018186884,0.0037923164,0.0022244353],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014466127,0.000016559838,0.0009472703,0.00031312913,0.000013285117,0.000846537,0.17705129,0.0001262886,0.0016652491,0.5596675,0.22537856,0.033829737],"study_design_scores_gemma":[0.0000032249238,0.000007857222,0.0020105515,0.00020006982,0.000005404126,0.00033378726,0.010994317,0.0000722036,0.0007186907,0.0051784213,0.98045784,0.0000176605],"about_ca_topic_score_codex":0.08050247,"about_ca_topic_score_gemma":0.19807422,"teacher_disagreement_score":0.08050247,"about_ca_system_score_codex":0.0077706324,"about_ca_system_score_gemma":0.0026843778,"threshold_uncertainty_score":0.1600678},"labels":[],"label_agreement":null},{"id":"W7024293474","doi":"","title":"The Role of Attitude, Parenting Styles, and School in First-Language Attrition and Code-switching","year":2022,"lang":"en","type":"other","venue":"OSF Preprints (OSF Preprints)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Attrition; Feeling; Affect (linguistics); Style (visual arts); First language; Language proficiency; On Language; Variation (astronomy)","score_opus":0.007460949059839035,"score_gpt":0.2505892789303832,"score_spread":0.24312832987054414,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7024293474","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9992606,0.000060873048,0.000028327879,0.0000386885,0.0000029651922,0.00000720463,0.000015943264,9.136331e-7,0.0005844241],"genre_scores_gemma":[0.9992729,0.00007425312,0.000083281455,0.000027233275,0.00000326487,0.000011778976,0.00004335771,0.0000016077557,0.00048249259],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9989876,0.00038009556,0.00008044493,0.000112834285,0.00023308389,0.0002059068],"domain_scores_gemma":[0.9935034,0.0020237537,0.002425077,0.00033056224,0.00035996462,0.0013573248],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016663086,0.00031024232,0.00031900342,0.0006407342,0.0011836676,0.0015359978,0.0006041296,0.0006014354,0.003468928],"category_scores_gemma":[0.004581469,0.00022238158,0.000661362,0.0007354018,0.0009923503,0.0006009088,0.0010354577,0.0012746194,0.00029315922],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014189776,0.00046081544,0.9914096,0.000018535897,0.000032338372,0.00010231221,0.0044778767,0.000016259513,0.00039087122,0.00008877178,0.000057231693,0.0028034295],"study_design_scores_gemma":[0.0000029770856,0.00015271737,0.99515754,0.000013088939,0.000012548519,0.000054210355,0.004311697,0.00006389966,0.00007521469,0.000038541475,0.00011364674,0.00000387074],"about_ca_topic_score_codex":0.0135634765,"about_ca_topic_score_gemma":0.021022685,"teacher_disagreement_score":0.0135634765,"about_ca_system_score_codex":0.00089020556,"about_ca_system_score_gemma":0.0010281281,"threshold_uncertainty_score":0.026969075},"labels":[],"label_agreement":null},{"id":"W7025142188","doi":"","title":"2 11 Thurs Pdf For Sharing","year":2021,"lang":"en","type":"other","venue":"Bulletin of Miscellaneous Information (Royal Gardens Kew)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Quality (philosophy); Service (business); Best practice; The Internet; Frequently asked questions; Postal service","score_opus":0.007585167775309394,"score_gpt":0.21250786982196718,"score_spread":0.20492270204665777,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7025142188","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00018063196,0.00033761785,0.0013726398,0.0013983599,0.0036093676,0.00018807023,0.029785965,0.0067777704,0.9563496],"genre_scores_gemma":[0.00067811966,0.0002994822,0.0005499254,0.0005490866,0.0003278286,0.000107517975,0.009364154,0.0024406642,0.98568326],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9990933,0.0000971881,0.00005079524,0.00011043712,0.0005321318,0.000116214695],"domain_scores_gemma":[0.9963431,0.00040948385,0.000115430186,0.0007490652,0.0016311655,0.00075178395],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.0011239921,0.00088805874,0.0010799407,0.0025655571,0.0015886954,0.00765933,0.0016043856,0.0015238726,0.94939613],"category_scores_gemma":[0.008408544,0.0005550219,0.0008891658,0.003576861,0.00045375997,0.0037957302,0.004321517,0.0017795906,0.9502776],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000011964648,0.0000100705165,0.000023571598,0.000043146465,9.5275067e-7,0.000012232016,0.000014063852,0.00001523224,0.000059349597,0.00052806694,0.9783678,0.02091342],"study_design_scores_gemma":[0.000006716001,0.0000061054143,0.00012231633,0.000045156296,0.0000011718662,0.000022712444,0.000026142932,0.000024119949,0.000051224004,0.00044791706,0.999241,0.0000055506625],"about_ca_topic_score_codex":0.0035582138,"about_ca_topic_score_gemma":0.0061386507,"teacher_disagreement_score":0.050603867,"about_ca_system_score_codex":0.0010970561,"about_ca_system_score_gemma":0.001302382,"threshold_uncertainty_score":0.07218021},"labels":[],"label_agreement":null},{"id":"W7027230488","doi":"","title":"Canada and the Ukrainian Question, 1939-1945: A Study in Statecraft by <string-name><given-name>Bohdan S.</given-name>s&amp;gt;<surname>Kordan</surname></string-name> (review)","year":2022,"lang":"en","type":"article","venue":"Project Muse (Johns Hopkins University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Nucleofection; Gestational period; TSG101; Liquation; Dysgeusia; Diafiltration; Emperipolesis; Triacetin; Demotion","score_opus":0.011056822742613572,"score_gpt":0.23257875073523387,"score_spread":0.2215219279926203,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7027230488","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0130446935,0.97264063,0.00003515589,0.0053520896,0.00013972401,0.0000067583937,0.00052035385,0.0000036531403,0.008257039],"genre_scores_gemma":[0.18507987,0.8095611,0.00007637013,0.002255388,0.00017783635,0.000014526176,0.000743809,0.000008889825,0.0020821933],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9995197,0.00008280138,0.000041975025,0.00007439304,0.00017710647,0.00010412972],"domain_scores_gemma":[0.9972326,0.0011290316,0.00039570875,0.000056748508,0.0009693368,0.0002165641],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012945746,0.0002063774,0.0006753339,0.0021123718,0.0016157512,0.0019000798,0.0009257509,0.0009060511,0.003832391],"category_scores_gemma":[0.0043592458,0.00020283971,0.00039825065,0.008121447,0.004063226,0.0018087012,0.0009424633,0.00092838705,0.00030843774],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00060169754,0.00008081894,0.09995921,0.020106044,0.0009856863,0.0013713555,0.017810186,0.0006228413,0.00035314035,0.033671983,0.09654487,0.7278921],"study_design_scores_gemma":[0.000047643167,0.00004528358,0.41739216,0.015656937,0.00063702004,0.0010268156,0.018511752,0.00008059641,0.0003903196,0.0035213507,0.5426263,0.00006387507],"about_ca_topic_score_codex":0.91228503,"about_ca_topic_score_gemma":0.943322,"teacher_disagreement_score":0.08771497,"about_ca_system_score_codex":0.016679425,"about_ca_system_score_gemma":0.044003215,"threshold_uncertainty_score":0.17646301},"labels":[],"label_agreement":null},{"id":"W7028681110","doi":"","title":"Fortress Paper Announces Second Quarter 2016 Results and Appointment of Director","year":2016,"lang":"en","type":"other","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Quarter (Canadian coin); Fortress (chess); Work (physics); Government (linguistics)","score_opus":0.0061106276392092025,"score_gpt":0.24646385407831128,"score_spread":0.2403532264391021,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7028681110","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0023011032,0.00076531,0.0018656623,0.01805749,0.02243117,0.00028734613,0.012094732,0.004002159,0.938195],"genre_scores_gemma":[0.0010555966,0.00009401765,0.00019100435,0.00027639835,0.0003352125,0.000017116368,0.0013806454,0.00028453267,0.99636555],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9986468,0.000102108555,0.00004472285,0.00018687484,0.00075842434,0.00026101473],"domain_scores_gemma":[0.99622595,0.00028006587,0.00010424497,0.000440107,0.001564078,0.0013855498],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.002049802,0.0008278563,0.00090909324,0.0021561608,0.0028701103,0.008953914,0.0010998201,0.0021239908,0.6276103],"category_scores_gemma":[0.004363822,0.00035731727,0.0008919065,0.0016564454,0.0005827703,0.0026304978,0.0019918527,0.0023629407,0.54910016],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007079392,0.000050379487,0.00015985119,0.000027494569,0.000002289922,0.000034401168,0.00001524218,0.0000305219,0.00021181772,0.001575956,0.9795517,0.018269515],"study_design_scores_gemma":[0.000022130716,0.000042096326,0.0009375943,0.000022100998,0.000004718678,0.000028058728,0.00007157263,0.00021161175,0.00069801527,0.001133552,0.99681985,0.000008712722],"about_ca_topic_score_codex":0.009033272,"about_ca_topic_score_gemma":0.02584355,"teacher_disagreement_score":0.6276103,"about_ca_system_score_codex":0.0022027316,"about_ca_system_score_gemma":0.0033325027,"threshold_uncertainty_score":0.5311687},"labels":[],"label_agreement":null},{"id":"W7030158633","doi":"","title":"L'uniformisation rÃ©gionale du droit, processus comparÃ©s et projet de l'OHADA en transport routier","year":2000,"lang":"fr","type":"other","venue":"Library and Archives Canada (Government of Canada)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Work (physics); Term (time); MEDLINE","score_opus":0.0036714286861919544,"score_gpt":0.1644929105910733,"score_spread":0.16082148190488135,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7030158633","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.073959365,0.011504117,0.7594707,0.01006434,0.0014751131,0.0009456069,0.0058391285,0.015301559,0.12144013],"genre_scores_gemma":[0.36616322,0.007591672,0.37276575,0.0012037541,0.00055241474,0.0007856648,0.011445024,0.004240416,0.23525207],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9875486,0.0038166943,0.00110362,0.0028218448,0.0039349115,0.0007742834],"domain_scores_gemma":[0.9857006,0.0034760893,0.0009406602,0.0043201772,0.005103873,0.0004586178],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.015468192,0.0011439725,0.0009326502,0.004128189,0.0033384236,0.0156802,0.0024845542,0.0024176764,0.019712899],"category_scores_gemma":[0.017852217,0.0010474434,0.0014681009,0.0052362406,0.0030363135,0.0118709,0.0031263328,0.0031737324,0.005071409],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00081230194,0.00031845836,0.011265109,0.0006704275,0.000094178584,0.00024018127,0.0032492545,0.01297016,0.0102070235,0.21725598,0.0372174,0.70569956],"study_design_scores_gemma":[0.00025731587,0.00033221432,0.038362153,0.0006792086,0.00014015367,0.00066427473,0.0049150074,0.06762465,0.04029277,0.046444625,0.79999024,0.0002975035],"about_ca_topic_score_codex":0.42142522,"about_ca_topic_score_gemma":0.23430672,"teacher_disagreement_score":0.42142522,"about_ca_system_score_codex":0.01880184,"about_ca_system_score_gemma":0.017865524,"threshold_uncertainty_score":0.8379445},"labels":[],"label_agreement":null},{"id":"W7032945298","doi":"","title":"Official++... ↭Help Usa@ +1-800-686-6918...((( QuickBooKs support phone number... ↭Help Usa@ +1-800-686-6918...((( QuickBooKs support number @@","year":2016,"lang":"en","type":"other","venue":"OSF Preprints (OSF Preprints)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Phone; Payroll; Service (business); Telephone number; Customer service; Point of sale; Telephone line; Customer advocacy","score_opus":0.013893604235830363,"score_gpt":0.282747237990209,"score_spread":0.26885363375437865,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7032945298","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00049574644,0.00025767784,0.00070817396,0.0016146211,0.0018894719,0.00017930116,0.0018266416,0.002549802,0.99047863],"genre_scores_gemma":[0.00063534384,0.00015796251,0.00019837289,0.00033836337,0.0001324628,0.000029685316,0.0004615553,0.00031712657,0.9977291],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9993124,0.000060524704,0.000025128922,0.00016220669,0.00031548907,0.00012423839],"domain_scores_gemma":[0.9966484,0.00019950469,0.0000912062,0.00015269578,0.0012677897,0.0016404011],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.00056814094,0.0010802215,0.0009466963,0.0011715987,0.002236847,0.0048883623,0.0010304957,0.0013799479,0.94428813],"category_scores_gemma":[0.0030920736,0.0005685239,0.00048676322,0.0011182155,0.0005046266,0.0028472561,0.0023706418,0.0016133684,0.948222],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000022002587,0.0000343117,0.00010861123,0.00004008354,9.4818625e-7,0.000024777486,0.00002625213,0.000015304351,0.0003166619,0.00043366814,0.97703856,0.0219388],"study_design_scores_gemma":[0.000013092626,0.00003125683,0.0004179032,0.000053408923,0.0000033547487,0.00007244042,0.00011029824,0.000060676623,0.00018730844,0.00014017921,0.9989033,0.0000068451095],"about_ca_topic_score_codex":0.003539339,"about_ca_topic_score_gemma":0.0062284693,"teacher_disagreement_score":0.055711865,"about_ca_system_score_codex":0.0009783966,"about_ca_system_score_gemma":0.0014704927,"threshold_uncertainty_score":0.079466105},"labels":[],"label_agreement":null},{"id":"W7033053620","doi":"","title":"Pattern Energy Enters Agreement to be Acquired by Canada Pension Plan Investment Board - Today's Market News (MARKCOMM) News","year":2019,"lang":"en","type":"other","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Plan (archaeology); Investment (military); Pension plan; Energy (signal processing); Measure (data warehouse)","score_opus":0.0103362946786796,"score_gpt":0.22006977038829428,"score_spread":0.20973347570961468,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7033053620","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0014939769,0.00035852572,0.0006793767,0.012186864,0.006664023,0.00012998508,0.010013155,0.0016183517,0.96685576],"genre_scores_gemma":[0.0016364357,0.00008734751,0.00016334515,0.0006685361,0.00016555835,0.000010119157,0.00168504,0.00014380582,0.9954399],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9991266,0.000018649655,0.000010996591,0.000058267968,0.00060252013,0.00018296568],"domain_scores_gemma":[0.99855286,0.00007031198,0.00003393994,0.0000971019,0.0008554887,0.00039028274],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0006214515,0.00054390135,0.00041334683,0.001063214,0.0032594255,0.0061303796,0.0008162747,0.0022482534,0.4331709],"category_scores_gemma":[0.0018997362,0.00030571272,0.0004649174,0.0011568405,0.0005443022,0.0014150343,0.0011814755,0.0024952723,0.23489137],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000024253623,0.000012153342,0.00012059023,0.00001161102,0.0000010566839,0.000030930754,0.000009757822,0.000017621933,0.00012639567,0.0014006046,0.9881094,0.010135588],"study_design_scores_gemma":[0.0000073976107,0.000008291571,0.001169332,0.00001357504,0.000001718431,0.000014781803,0.00005437924,0.00010501584,0.00026308835,0.00041084242,0.99794656,0.0000051213906],"about_ca_topic_score_codex":0.25393066,"about_ca_topic_score_gemma":0.60537297,"teacher_disagreement_score":0.7460693,"about_ca_system_score_codex":0.0044792877,"about_ca_system_score_gemma":0.008187534,"threshold_uncertainty_score":0.8085129},"labels":[],"label_agreement":null},{"id":"W7033082288","doi":"","title":"Odyssée numérique : un jeu sur la compétence numérique","year":2024,"lang":"fr","type":"article","venue":"R-libre (Université Téluq)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Nucleofection; Gestational period; Demotion; Filter (signal processing); Pretext","score_opus":0.01116503745743323,"score_gpt":0.22409536742221373,"score_spread":0.2129303299647805,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7033082288","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.099050656,0.09023908,0.004308331,0.52308553,0.009352214,0.00008353335,0.0005769911,0.00014693843,0.2731567],"genre_scores_gemma":[0.653328,0.056360107,0.0057179327,0.06682731,0.0021852076,0.00014504005,0.00043421003,0.00024001845,0.21476215],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9959728,0.0010754298,0.000122050296,0.00040300708,0.0013984351,0.0010283495],"domain_scores_gemma":[0.9937358,0.0010929361,0.0002715485,0.00023042638,0.0027428253,0.0019264517],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0052352496,0.0006620655,0.0007045987,0.0025327925,0.014172652,0.009731467,0.001259339,0.003726149,0.010497316],"category_scores_gemma":[0.006263699,0.00029638576,0.0004328202,0.0021602344,0.01519494,0.007307408,0.00642286,0.007933073,0.0011517281],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011710682,0.00013596474,0.019130565,0.00050369033,0.00003815428,0.0007253289,0.15001598,0.00019830203,0.0009328191,0.3553347,0.2601542,0.21271327],"study_design_scores_gemma":[0.000013047766,0.00005231879,0.023726618,0.0015123272,0.000018460798,0.00032276704,0.05045061,0.00023390903,0.000319115,0.011795811,0.91147935,0.000075589865],"about_ca_topic_score_codex":0.7915927,"about_ca_topic_score_gemma":0.8642715,"teacher_disagreement_score":0.7915927,"about_ca_system_score_codex":0.036179844,"about_ca_system_score_gemma":0.052264433,"threshold_uncertainty_score":0.41926926},"labels":[],"label_agreement":null},{"id":"W7034253873","doi":"","title":"StabilitÃ© magnÃ©tohydrodynamique des cuves d'Ã©lectrolyse : aspects physiques et idÃ©es nouvelles","year":2012,"lang":"fr","type":"other","venue":"Library and Archives Canada (Government of Canada)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Field (mathematics); Term (time); Identification (biology); Set (abstract data type); Work (physics)","score_opus":0.004296620806350779,"score_gpt":0.16909897368984225,"score_spread":0.1648023528834915,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7034253873","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.40490326,0.0075061535,0.44914752,0.0039152554,0.001914009,0.00016205484,0.00085223577,0.00234617,0.12925325],"genre_scores_gemma":[0.84621656,0.002444616,0.03757652,0.00032379015,0.000321233,0.0002181708,0.0004403667,0.0005990384,0.111859806],"study_design_codex":"bench_or_experimental","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99974626,0.000039017254,0.000010341501,0.00005974985,0.00010752542,0.000037085625],"domain_scores_gemma":[0.99974185,0.00007421064,0.000037572994,0.000040959763,0.00007269343,0.000032777],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00025410988,0.0008026455,0.0006996195,0.00053103466,0.001083014,0.0023486386,0.0005004477,0.0008330664,0.008018455],"category_scores_gemma":[0.00080183946,0.00040328773,0.00048641372,0.00023358996,0.0010768843,0.0011121522,0.0012257384,0.0010474134,0.0013631858],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005300031,0.00022882121,0.0031153774,0.0005455162,0.0001470688,0.00045577326,0.00068129593,0.23664658,0.31082034,0.23213717,0.018882401,0.19580966],"study_design_scores_gemma":[0.00007531378,0.0002014623,0.0053607104,0.00006718296,0.000031807092,0.00028610774,0.000249314,0.7931504,0.09409087,0.03841793,0.067953825,0.00011501066],"about_ca_topic_score_codex":0.0058763004,"about_ca_topic_score_gemma":0.0040904195,"teacher_disagreement_score":0.008018455,"about_ca_system_score_codex":0.0012310006,"about_ca_system_score_gemma":0.0009309603,"threshold_uncertainty_score":0.026824415},"labels":[],"label_agreement":null},{"id":"W7034392209","doi":"","title":"Suncor Energy to release first quarter 2014 financial results and hold Annual General Meeting of shareholders","year":2014,"lang":"en","type":"other","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Quarter (Canadian coin); Shareholder; Energy (signal processing); Term (time)","score_opus":0.006306448704255838,"score_gpt":0.22285324126992187,"score_spread":0.21654679256566603,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7034392209","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0022973348,0.00034797413,0.0032838453,0.0075365435,0.0044138217,0.00013232192,0.006773474,0.0038388704,0.97137576],"genre_scores_gemma":[0.004724194,0.0001249834,0.00067809457,0.0003714008,0.00025151524,0.000028275368,0.0023558855,0.00071612454,0.9907494],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99876165,0.00008936351,0.00002815981,0.000109459135,0.0008165515,0.00019474197],"domain_scores_gemma":[0.99752146,0.00025237614,0.00010022949,0.00039360768,0.0012500443,0.0004822803],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0017209281,0.00067480956,0.00062561076,0.0013738668,0.0015503553,0.00559199,0.0008173882,0.0013639884,0.5773817],"category_scores_gemma":[0.006153307,0.00032765896,0.0004695497,0.0011215151,0.00039451133,0.0022535974,0.0015422704,0.0019663002,0.37058696],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000067273366,0.000028270884,0.00016713413,0.00001922683,0.0000025859908,0.000019769244,0.00000959275,0.0000707357,0.0002223584,0.004788747,0.9612941,0.033310212],"study_design_scores_gemma":[0.000021075504,0.000028858412,0.0008009128,0.000032040283,0.0000042119987,0.000029151732,0.000054225602,0.0008230909,0.0010837641,0.0051495363,0.9919624,0.000010695755],"about_ca_topic_score_codex":0.005172229,"about_ca_topic_score_gemma":0.01178942,"teacher_disagreement_score":0.5773817,"about_ca_system_score_codex":0.0016337414,"about_ca_system_score_gemma":0.0024079036,"threshold_uncertainty_score":0.6028137},"labels":[],"label_agreement":null},{"id":"W7034394634","doi":"","title":"A Teaching Opportunity? -- The Daily Beaver Morning Show","year":2023,"lang":"en","type":"other","venue":"Bulletin of Miscellaneous Information (Royal Gardens Kew)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Voting; Government (linguistics); Language change; Plan (archaeology); Memoir","score_opus":0.009729404281150728,"score_gpt":0.21361097600209444,"score_spread":0.2038815717209437,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7034394634","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004342883,0.0016308501,0.0015372983,0.030729992,0.021648992,0.00026966882,0.0027497734,0.005408216,0.93168235],"genre_scores_gemma":[0.007911745,0.0006099726,0.00097346155,0.0050924183,0.0015229026,0.00010793236,0.0010041993,0.0011556857,0.9816218],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99939466,0.00010139822,0.000014885389,0.00007824328,0.00018701139,0.00022374136],"domain_scores_gemma":[0.9956726,0.00012388744,0.00009156325,0.00016447798,0.00038620035,0.0035613496],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0008587619,0.001039364,0.00054619974,0.0005470508,0.005635158,0.007796872,0.0011006009,0.002035407,0.6434566],"category_scores_gemma":[0.002766034,0.0004509075,0.00059948274,0.00060950214,0.00083817425,0.005181551,0.0065219365,0.004227045,0.5493119],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000012357572,0.000049766357,0.000096267075,0.000021246386,8.0242836e-7,0.000056747496,0.00010289274,0.0000045953393,0.00008526229,0.00038596208,0.99176425,0.007419913],"study_design_scores_gemma":[0.000006386567,0.000017592456,0.00048174147,0.000038938586,0.0000011568455,0.00007412219,0.0007533771,0.000016625383,0.00007892667,0.00023088124,0.99829406,0.000006055198],"about_ca_topic_score_codex":0.0039985743,"about_ca_topic_score_gemma":0.02119444,"teacher_disagreement_score":0.6434566,"about_ca_system_score_codex":0.0012785443,"about_ca_system_score_gemma":0.0019194247,"threshold_uncertainty_score":0.5085659},"labels":[],"label_agreement":null},{"id":"W7036096161","doi":"","title":"Art and Ideology in China's Postsocialist Stage Productions of <i>A Doll's House</i>","year":2018,"lang":"en","type":"article","venue":"Project Muse (Johns Hopkins University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Nucleofection; Gestational period; TSG101; Dysgeusia; Diafiltration; Liquation; Emperipolesis; Triacetin; Demotion","score_opus":0.010980609909252467,"score_gpt":0.23750969970533217,"score_spread":0.22652908979607972,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7036096161","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9284796,0.00058981654,0.00015069904,0.0005447479,0.000018832861,0.000009947844,0.00004567711,0.0000039310185,0.0701568],"genre_scores_gemma":[0.99611086,0.00011813955,0.000052538217,0.000014277957,0.0000044022904,0.000002170297,0.000012889258,0.0000022971265,0.0036823722],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9994936,0.00013832949,0.00001835812,0.00005528398,0.00010785386,0.0001865573],"domain_scores_gemma":[0.99944884,0.000194089,0.00014714907,0.000042118485,0.00006497355,0.00010288867],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007891418,0.00019356814,0.00015390042,0.0017053287,0.005536874,0.002164563,0.000449484,0.00038887677,0.006577843],"category_scores_gemma":[0.0010464008,0.00015864983,0.00016028731,0.002511258,0.0055121514,0.00095112313,0.0011172338,0.00082443614,0.00019542559],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003049414,0.00011829925,0.13860166,0.00021946833,0.000060542818,0.0026889502,0.47477308,0.0005482129,0.0030148518,0.29902056,0.0038497215,0.076799646],"study_design_scores_gemma":[0.000022946673,0.00009692168,0.73199344,0.00016904762,0.000039116898,0.00024950368,0.18833786,0.0007420117,0.0011686955,0.011339518,0.06579401,0.00004702608],"about_ca_topic_score_codex":0.10582224,"about_ca_topic_score_gemma":0.34190112,"teacher_disagreement_score":0.10582224,"about_ca_system_score_codex":0.0061951205,"about_ca_system_score_gemma":0.0033432676,"threshold_uncertainty_score":0.21041256},"labels":[],"label_agreement":null},{"id":"W7036408285","doi":"","title":"Bon usage des inhibiteurs de la pompe à protons à l'officine","year":2023,"lang":"fr","type":"dissertation","venue":"HAL AMU","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Pharmaceutical technology; Iatrogenic disease","score_opus":0.02212091463634494,"score_gpt":0.3203811642730099,"score_spread":0.298260249636665,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7036408285","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8746356,0.08659441,0.0038274108,0.0024179325,0.0009444918,0.00036493398,0.0010436472,0.00039190103,0.029779669],"genre_scores_gemma":[0.93874645,0.024111902,0.0038037885,0.0011320402,0.00049911457,0.00018903069,0.0007853395,0.000054094446,0.03067821],"study_design_codex":"bench_or_experimental","study_design_gemma":"not_applicable","domain_scores_codex":[0.9995993,0.00008964254,0.00003215559,0.000051556137,0.00016958732,0.00005777263],"domain_scores_gemma":[0.9994929,0.00015548535,0.00012168298,0.000044377182,0.000103249135,0.00008232603],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004944234,0.00044840618,0.00054289034,0.00041343886,0.0003622677,0.00055524596,0.00027538725,0.00060743757,0.008544953],"category_scores_gemma":[0.0006651062,0.00015067757,0.0005342346,0.00024085725,0.00034160493,0.00037359414,0.00032803902,0.0010071712,0.0014375274],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0048348513,0.0011467669,0.0034185466,0.0021835654,0.00021862773,0.0004618818,0.00027251377,0.00016019953,0.7687295,0.0008415023,0.001985067,0.21574692],"study_design_scores_gemma":[0.0012487568,0.03831415,0.06922846,0.0005211715,0.0006925249,0.004026537,0.0004864983,0.00075381773,0.6221899,0.00063038873,0.26182425,0.000083637555],"about_ca_topic_score_codex":0.0007364317,"about_ca_topic_score_gemma":0.0010594967,"teacher_disagreement_score":0.008544953,"about_ca_system_score_codex":0.00026373597,"about_ca_system_score_gemma":0.0003956376,"threshold_uncertainty_score":0.028585732},"labels":[],"label_agreement":null},{"id":"W7036517809","doi":"","title":"Black Guillemots as indicators of change in the near-shore Arctic marine ecosystem","year":2007,"lang":"en","type":"other","venue":"Library and Archives Canada (Government of Canada)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Foraging; Arctic; Habitat; Sea ice; Arctic ice pack; Ecosystem; Energy expenditure; Beaufort sea; Capelin","score_opus":0.005359239146570066,"score_gpt":0.1829882703194887,"score_spread":0.17762903117291862,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7036517809","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9993092,0.000070016955,0.00009370233,0.000015322343,0.0000025254067,0.0000032829166,0.000073863266,0.0000027507492,0.00042939227],"genre_scores_gemma":[0.99902844,0.000044882418,0.0005905143,0.000010583572,0.0000034318366,0.0000065261534,0.0001216197,0.0000017858296,0.0001921251],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9998103,0.00006225403,0.000011434179,0.00004106289,0.000039864248,0.000035119803],"domain_scores_gemma":[0.9991185,0.00017113164,0.000425398,0.00003295836,0.00011380414,0.00013809821],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005298955,0.00019302829,0.00016340143,0.0015506225,0.00059016456,0.0006532983,0.00021247954,0.00026402884,0.0007295374],"category_scores_gemma":[0.001526037,0.0001280403,0.00012345202,0.00075814466,0.0003734238,0.000417069,0.00050677574,0.00027615528,0.000101233745],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005263146,0.000012111355,0.9947242,0.000009664665,0.000015908254,0.000057344285,0.00070728105,0.00006780615,0.0009197838,0.000033846038,0.00006140649,0.0033378268],"study_design_scores_gemma":[3.8397343e-7,0.000011386131,0.99926525,0.0000020875168,0.0000020932364,0.000022582775,0.00043954435,0.00011364254,0.000037632162,0.000014192189,0.00008940234,0.0000016512389],"about_ca_topic_score_codex":0.027755367,"about_ca_topic_score_gemma":0.10692088,"teacher_disagreement_score":0.9722446,"about_ca_system_score_codex":0.00046796593,"about_ca_system_score_gemma":0.00018932257,"threshold_uncertainty_score":0.055187643},"labels":[],"label_agreement":null},{"id":"W7036540934","doi":"","title":"Cani da cadavere: “dispositivo biologico specializzato” nell’individuazione di tracce ematiche latenti sulla scena del crimine. Da mito a prova scientifica.","year":2016,"lang":"en","type":"article","venue":"Institutional Research Information System University of Ferrara (University of Ferrara)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Cadaveric spasm; Reliability (semiconductor); Confounding; Blocking (statistics); Protocol (science)","score_opus":0.09662494367718787,"score_gpt":0.28840924783801086,"score_spread":0.19178430416082298,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7036540934","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.48473468,0.16904163,0.13300268,0.033466917,0.007041314,0.0015529505,0.001707661,0.0018581174,0.16759402],"genre_scores_gemma":[0.75097877,0.054729823,0.10936683,0.013021943,0.0031179118,0.00076151657,0.0013628996,0.00029926826,0.066361055],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9985782,0.00055828574,0.00009446883,0.00031526512,0.00035362676,0.00010007678],"domain_scores_gemma":[0.99775547,0.00056595996,0.00047467827,0.0006246625,0.0003145764,0.00026457896],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035183427,0.00062486203,0.00033780467,0.0008716484,0.00078225374,0.0016409594,0.00084302056,0.0014475273,0.0085735],"category_scores_gemma":[0.003311963,0.00031985642,0.0004174051,0.00040707493,0.0031051063,0.0017341311,0.0012486683,0.0011235393,0.0026138618],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001564906,0.0010286039,0.06329982,0.0022136595,0.0001960915,0.004071776,0.002011979,0.00032293628,0.11818732,0.0102035,0.042983565,0.75391585],"study_design_scores_gemma":[0.00019286359,0.009697032,0.13588655,0.0032052177,0.0005172419,0.061493482,0.0040868125,0.0018069241,0.14590064,0.011707574,0.6252618,0.00024388521],"about_ca_topic_score_codex":0.0016450853,"about_ca_topic_score_gemma":0.0048278724,"teacher_disagreement_score":0.0085735,"about_ca_system_score_codex":0.0007749456,"about_ca_system_score_gemma":0.0014160844,"threshold_uncertainty_score":0.028681219},"labels":[],"label_agreement":null},{"id":"W7036804693","doi":"","title":"Coldplay Add Second Shows In Los Angeles, San Diego, And Vancouver On 'Music Of The Spheres' World Tour","year":2023,"lang":"en","type":"other","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Government (linguistics); Field (mathematics); Work (physics); Agency (philosophy); Exposition (narrative)","score_opus":0.012445295967601504,"score_gpt":0.2415953135684909,"score_spread":0.2291500176008894,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7036804693","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.002797871,0.0011846797,0.0002530683,0.004298929,0.0033264745,0.000045402212,0.0025680182,0.0005700321,0.9849555],"genre_scores_gemma":[0.004117137,0.00031984597,0.00012691402,0.0002423732,0.00015770216,0.000014453998,0.0007381663,0.00031376042,0.99396956],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99965537,0.000028658988,0.0000046895875,0.0000361014,0.00013521813,0.0001399188],"domain_scores_gemma":[0.9994461,0.00003341036,0.000009183312,0.000027929353,0.00015577728,0.00032772162],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.00039863735,0.0010521269,0.0005948875,0.0014804912,0.011311152,0.00755904,0.0008862287,0.0014865637,0.50628567],"category_scores_gemma":[0.00066639896,0.0004921624,0.0004743616,0.0020505844,0.0010218249,0.001533679,0.0029349462,0.0027859332,0.12456693],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000035770605,0.000015717924,0.00021806419,0.000021829233,0.0000025165784,0.000051594234,0.00024282942,0.000026841582,0.000094749004,0.0014684653,0.988299,0.0095225945],"study_design_scores_gemma":[0.0000053645626,0.0000051251145,0.0011656508,0.000024425688,0.0000018675345,0.000011763109,0.00091825484,0.00002402942,0.000049198265,0.00016752802,0.9976223,0.000004486519],"about_ca_topic_score_codex":0.3124224,"about_ca_topic_score_gemma":0.8716982,"teacher_disagreement_score":0.6875776,"about_ca_system_score_codex":0.0044223694,"about_ca_system_score_gemma":0.004449826,"threshold_uncertainty_score":0.7042236},"labels":[],"label_agreement":null},{"id":"W7037106607","doi":"","title":"Development of Bioactive Cannabis Extracts Through Optimization of Green Supercritical Fluid Process","year":2023,"lang":"fr","type":"other","venue":"PolyPublie (École Polytechnique de Montréal)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Cannabinol; Tetrahydrocannabinol; Pharmaceutical technology; Cannabis","score_opus":0.018037350095287924,"score_gpt":0.2714444185559685,"score_spread":0.2534070684606806,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7037106607","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9464683,0.008998155,0.029419634,0.00045902643,0.0002708974,0.00051631394,0.0011199352,0.0004071493,0.012340682],"genre_scores_gemma":[0.9550653,0.0059679933,0.028788837,0.00017287793,0.000052440715,0.0002606491,0.00082661334,0.000102287,0.008762909],"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9997354,0.000029645833,0.000017475351,0.000042497213,0.00014007195,0.00003484738],"domain_scores_gemma":[0.9998926,0.000021483056,0.000022363316,0.000008051991,0.000041597094,0.000013840641],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00031710102,0.00054352457,0.00039833479,0.00066226965,0.00033558838,0.00047241652,0.00030440136,0.00043022766,0.0019573234],"category_scores_gemma":[0.000323695,0.00013898066,0.0007154386,0.0005435583,0.00031894227,0.00042990848,0.00030816763,0.00048829307,0.00055194576],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023332851,0.00012296769,0.0003416466,0.00041602107,0.000029091989,0.000125084,0.000061899584,0.0011538422,0.9849338,0.000396245,0.00027513513,0.011911046],"study_design_scores_gemma":[0.000023406259,0.00042668305,0.0016314015,0.000028648481,0.000041802643,0.00011087098,0.00004531424,0.0026424683,0.9880095,0.000110096335,0.006906704,0.00002295236],"about_ca_topic_score_codex":0.0020237842,"about_ca_topic_score_gemma":0.0041193883,"teacher_disagreement_score":0.0020237842,"about_ca_system_score_codex":0.00037705264,"about_ca_system_score_gemma":0.0006815893,"threshold_uncertainty_score":0.006547928},"labels":[],"label_agreement":null},{"id":"W7037375746","doi":"","title":"EDP with losses of 76 million in the first quarter of 2022 – Energia","year":2022,"lang":"en","type":"other","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Quarter (Canadian coin); Energy (signal processing); Electric energy; Energy consumption; Measure (data warehouse)","score_opus":0.005089662253442901,"score_gpt":0.2254029026382822,"score_spread":0.2203132403848393,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7037375746","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024831459,0.0053623957,0.015390408,0.03732273,0.008304657,0.00009930731,0.04026317,0.0077457833,0.86068004],"genre_scores_gemma":[0.10997641,0.0049201595,0.00883848,0.003932644,0.0004931646,0.000102969054,0.02655533,0.0012700263,0.8439109],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9997514,0.000029059116,0.000009713916,0.00004454327,0.00010929502,0.000055895518],"domain_scores_gemma":[0.9997712,0.000033189204,0.00002238356,0.000039162125,0.00009648745,0.00003754859],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004211327,0.0007069514,0.0002662775,0.0006335531,0.0007286203,0.0020000155,0.00058759784,0.0012201634,0.09166157],"category_scores_gemma":[0.001501667,0.0001823706,0.00042517853,0.0013543138,0.00038634284,0.0018577478,0.0015523193,0.0015547259,0.043556936],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024177117,0.000042772936,0.0015413978,0.00017802046,0.000028448061,0.00032329813,0.00005639263,0.0019190452,0.0011863204,0.04315094,0.797213,0.15411866],"study_design_scores_gemma":[0.0000095352125,0.000014761456,0.0017147965,0.00005594912,0.000005171751,0.00011927379,0.0001202369,0.0008326653,0.0006949954,0.005390472,0.99103516,0.0000069977664],"about_ca_topic_score_codex":0.01206438,"about_ca_topic_score_gemma":0.010915627,"teacher_disagreement_score":0.09166157,"about_ca_system_score_codex":0.0014887488,"about_ca_system_score_gemma":0.0015253462,"threshold_uncertainty_score":0.30663848},"labels":[],"label_agreement":null},{"id":"W7037502215","doi":"","title":"Empire and nation the American Revolution in the Atlantic world","year":2005,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Empire; Atlantic World; Politics; Mainland; World history; British Empire","score_opus":0.013711102530547548,"score_gpt":0.279183362502735,"score_spread":0.2654722599721875,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7037502215","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0388265,0.04435536,0.005383564,0.056658387,0.0023992085,0.000011356075,0.00032531668,0.0001598603,0.8518805],"genre_scores_gemma":[0.6395885,0.037979424,0.00704007,0.009253989,0.0013803118,0.00004028789,0.0007218311,0.00024915443,0.30374655],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9996995,0.000119221426,0.000009828454,0.000042224132,0.00008502927,0.00004439702],"domain_scores_gemma":[0.9995359,0.0002592576,0.000047723282,0.000043497654,0.0000715274,0.00004210087],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00050773215,0.0001586146,0.00014777532,0.0005936105,0.0023865122,0.004399031,0.00026322628,0.00051650143,0.007438318],"category_scores_gemma":[0.0019970566,0.00016246602,0.00016563636,0.0015010494,0.0044374713,0.004788511,0.0009690956,0.001569082,0.0011436974],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000027419017,0.000012843805,0.0024086144,0.000060479004,0.00000957501,0.00011383035,0.008231952,0.00024444642,0.00021549354,0.76587206,0.12734438,0.095458835],"study_design_scores_gemma":[0.0000056016534,0.000005999893,0.006648584,0.00020205803,0.000010451184,0.00012724719,0.0072048893,0.00054601004,0.00014078991,0.19758679,0.7875097,0.000011956646],"about_ca_topic_score_codex":0.043150473,"about_ca_topic_score_gemma":0.09596452,"teacher_disagreement_score":0.043150473,"about_ca_system_score_codex":0.0023498086,"about_ca_system_score_gemma":0.002322716,"threshold_uncertainty_score":0.08579862},"labels":[],"label_agreement":null},{"id":"W7037917959","doi":"","title":"[FULL]WATCH.!! The Croods: A New Age (2020) Full_Movie || Online `Free [HD]","year":2020,"lang":"en","type":"other","venue":"OSF Preprints (OSF Preprints)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Battle; Join (topology); Commission; The Internet; Information Age; Free access","score_opus":0.013630198396749863,"score_gpt":0.26366931123411724,"score_spread":0.2500391128373674,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7037917959","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0009069789,0.0004726688,0.0012787526,0.003283548,0.0051503126,0.0003996767,0.0071020476,0.008734281,0.97267175],"genre_scores_gemma":[0.002810553,0.0004483577,0.0010660718,0.0020592634,0.001185348,0.00020546035,0.004281749,0.003993083,0.98395],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9995944,0.00004037416,0.000015160278,0.00005679537,0.00019550211,0.000097794364],"domain_scores_gemma":[0.997288,0.00016689421,0.00006663203,0.00022950998,0.00088981434,0.0013592397],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.0006424326,0.0007037078,0.00052141014,0.00088395586,0.002396888,0.0049509215,0.0011713208,0.0018764874,0.9153236],"category_scores_gemma":[0.0039595254,0.0004429756,0.00058943324,0.00061161246,0.0004264194,0.004447957,0.0042715957,0.002378099,0.8378524],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000022089176,0.0000173575,0.000036299378,0.00003359414,9.557083e-7,0.000012539094,0.000023607166,0.00000469274,0.000080993654,0.00028600366,0.9887504,0.010731441],"study_design_scores_gemma":[0.000026586345,0.00001943936,0.0004391911,0.00005695228,0.000002523265,0.000039633454,0.00008558581,0.000033437762,0.00012857917,0.00019964497,0.9989598,0.000008619223],"about_ca_topic_score_codex":0.003975699,"about_ca_topic_score_gemma":0.012326134,"teacher_disagreement_score":0.084676385,"about_ca_system_score_codex":0.0005601819,"about_ca_system_score_gemma":0.000640639,"threshold_uncertainty_score":0.12078053},"labels":[],"label_agreement":null},{"id":"W7038219141","doi":"","title":"“Goodbye to All That”: Propertius’ <italic>magnum iter</italic> between <italic>Elegies</italic> 3.16 and 3.21","year":2017,"lang":"en","type":"article","venue":"Project Muse (Johns Hopkins University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Volume (thermodynamics); Association (psychology); Set (abstract data type); Perspective (graphical); Work (physics)","score_opus":0.03420331903693148,"score_gpt":0.26041983711377054,"score_spread":0.22621651807683907,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7038219141","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0045943717,0.03151835,0.015101836,0.1255454,0.030716572,0.000053119464,0.00037639964,0.0006900703,0.7914039],"genre_scores_gemma":[0.14459212,0.010988401,0.0080214115,0.029420651,0.009793178,0.00009422979,0.00039632572,0.002732716,0.79396087],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99872905,0.00047902786,0.00005804748,0.0002352805,0.00035942876,0.00013903966],"domain_scores_gemma":[0.99917465,0.0002622361,0.00006679388,0.00016376704,0.0002439063,0.0000885684],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017105946,0.0006998815,0.0004066419,0.0009577158,0.0042141187,0.005266435,0.00067790644,0.0016980892,0.028603474],"category_scores_gemma":[0.0041634417,0.00037252987,0.00037777275,0.0009178922,0.010119764,0.009189126,0.0026444886,0.0060767387,0.010952896],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000033693443,0.000010096182,0.00018237237,0.00007519799,0.000007035049,0.00006176345,0.0041366094,0.000039065697,0.00020646144,0.42683083,0.54617625,0.022240616],"study_design_scores_gemma":[0.0000035094233,0.000006080568,0.0002607947,0.00008796699,0.0000032001603,0.000114088376,0.00083012495,0.000061252875,0.00019964176,0.026369654,0.97205454,0.000009028761],"about_ca_topic_score_codex":0.011589166,"about_ca_topic_score_gemma":0.028503198,"teacher_disagreement_score":0.028603474,"about_ca_system_score_codex":0.0042170775,"about_ca_system_score_gemma":0.002083757,"threshold_uncertainty_score":0.095688105},"labels":[],"label_agreement":null},{"id":"W7039027687","doi":"","title":"LignÃ©e de MaÃ¯s 5307 RÃ©sistant Aux Insectes","year":2014,"lang":"fr","type":"other","venue":"Contact-less Assessment of In-vivo Body Signals Using Microwave Doppler Radar (InTech)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Nucleofection; Gestational period; Diafiltration; TSG101; Liquation; Dysgeusia; Fusible alloy; Emperipolesis; Hyporeflexia","score_opus":0.02880862244165785,"score_gpt":0.3099307375318129,"score_spread":0.28112211509015506,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7039027687","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.75961524,0.025605487,0.030245876,0.0010308763,0.00071965327,0.00068562216,0.007707021,0.0016400582,0.17275009],"genre_scores_gemma":[0.6085544,0.021952907,0.032899477,0.00091745314,0.000078804915,0.00027627472,0.00979019,0.00047949931,0.32505098],"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9998517,0.0000093047165,0.0000048631355,0.000031255255,0.000079231104,0.000023536611],"domain_scores_gemma":[0.99989927,0.000011966738,0.00001899703,0.000010164237,0.000038092287,0.000021553795],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00023155105,0.00051724265,0.00032695313,0.0004252722,0.0003798762,0.00055393606,0.0002524533,0.00031987106,0.016983112],"category_scores_gemma":[0.00011884587,0.00015117465,0.00041716898,0.00036604688,0.00026867463,0.00021383005,0.00019258213,0.00063707284,0.003431428],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00032002645,0.00005291131,0.00046441198,0.00019265557,0.000019418037,0.00010802444,0.00003511193,0.00013675442,0.96707267,0.00048724547,0.0020047287,0.029106157],"study_design_scores_gemma":[0.000059894166,0.0015112828,0.017886415,0.000119967866,0.00012093946,0.00045456114,0.00016506645,0.0008805252,0.82958543,0.0002376932,0.14895275,0.000025409227],"about_ca_topic_score_codex":0.023944108,"about_ca_topic_score_gemma":0.07816069,"teacher_disagreement_score":0.023944108,"about_ca_system_score_codex":0.00095316884,"about_ca_system_score_gemma":0.001135993,"threshold_uncertainty_score":0.056814194},"labels":[],"label_agreement":null},{"id":"W7039218278","doi":"","title":"Les organismes communautaires au service des immigrants : 30 ans de changement","year":2024,"lang":"fr","type":"article","venue":"Érudit (Université de Montréal)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Immigration; Service (business); Community service; Social assistance","score_opus":0.016884500811325788,"score_gpt":0.21846131424648665,"score_spread":0.20157681343516087,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7039218278","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.97256833,0.00740439,0.00038967555,0.0027006522,0.00025854816,0.00006142047,0.009557775,0.000045926845,0.0070132976],"genre_scores_gemma":[0.96078235,0.007441362,0.0007122547,0.0005125972,0.0001546727,0.00014391927,0.0070318608,0.00002711696,0.023193954],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9994691,0.00006355556,0.000027370068,0.00007616463,0.00012901946,0.00023474364],"domain_scores_gemma":[0.99866414,0.0001216519,0.00025450636,0.000029630675,0.0005731703,0.00035686223],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00062816555,0.00051240437,0.00036486442,0.0013146951,0.0021633839,0.0011504365,0.00064643833,0.0012998886,0.004086701],"category_scores_gemma":[0.0017386996,0.00022014808,0.0005401171,0.0016819304,0.00083948317,0.00069153926,0.0013733264,0.0014429085,0.0009955686],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009073936,0.00021421834,0.89068675,0.0003368778,0.00017489027,0.00315345,0.021779448,0.0006032952,0.0023126877,0.0009659671,0.015162516,0.06370253],"study_design_scores_gemma":[0.0000044407,0.00006931607,0.9819478,0.000055188164,0.0000212436,0.00036909667,0.0046935156,0.000064356565,0.00023459454,0.0000769281,0.012445424,0.000018083281],"about_ca_topic_score_codex":0.7535335,"about_ca_topic_score_gemma":0.7992076,"teacher_disagreement_score":0.24646652,"about_ca_system_score_codex":0.0069220033,"about_ca_system_score_gemma":0.004768965,"threshold_uncertainty_score":0.49583596},"labels":[],"label_agreement":null},{"id":"W7040757063","doi":"","title":"Cleveland, Ohio","year":2009,"lang":"en","type":"other","venue":"OhioLink ETD Center (Ohio Library and Information Network)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Cabinet (room); Spring (device); Coffin; Hydrology (agriculture)","score_opus":0.006845633398497602,"score_gpt":0.20865630796693208,"score_spread":0.20181067456843446,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7040757063","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.000490873,0.002404785,0.00071628764,0.0023470863,0.0026606831,0.00010670208,0.005970849,0.0009238096,0.9843789],"genre_scores_gemma":[0.0027024564,0.002159838,0.00073933875,0.0006225904,0.00048230923,0.00010172595,0.002550585,0.0004690747,0.9901721],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9988858,0.00011174033,0.000063616004,0.00046271857,0.00033556725,0.00014053911],"domain_scores_gemma":[0.9986713,0.00020394895,0.00008940205,0.00020481163,0.00040778684,0.00042281818],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0008260065,0.0018700806,0.0010602276,0.0019022957,0.002686834,0.006411177,0.001344546,0.0028483137,0.87917566],"category_scores_gemma":[0.00232282,0.0006130385,0.0007392425,0.0024325403,0.00069423433,0.0033559117,0.002545295,0.0019973314,0.66368216],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007929581,0.00007569462,0.00043799824,0.0001419044,0.000009689779,0.0002675698,0.000076744946,0.000068962814,0.0003185969,0.006585563,0.8786092,0.113328755],"study_design_scores_gemma":[0.000014669058,0.000011391244,0.00037341708,0.00005700747,0.000003036957,0.00006539723,0.000040728308,0.000030375204,0.00004713037,0.00037411894,0.99897647,0.000006247908],"about_ca_topic_score_codex":0.007447435,"about_ca_topic_score_gemma":0.016766448,"teacher_disagreement_score":0.87917566,"about_ca_system_score_codex":0.0016845496,"about_ca_system_score_gemma":0.0021710775,"threshold_uncertainty_score":0.17234117},"labels":[],"label_agreement":null},{"id":"W7042360769","doi":"","title":"PORTAGE Phrase-Based System for Chinese-to-English Translation","year":2006,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Phrase; Machine translation; Task (project management); Translation (biology); Smoothing; Suspect","score_opus":0.008682828750867271,"score_gpt":0.2479192912981204,"score_spread":0.23923646254725314,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7042360769","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.032701608,0.00066755427,0.7842628,0.0003840117,0.00046386174,0.0006433459,0.010221064,0.1582071,0.012448681],"genre_scores_gemma":[0.15478323,0.00035159153,0.807527,0.00024869427,0.00018076456,0.00070686924,0.022079641,0.0034195364,0.0107026715],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9990277,0.00025176746,0.0001426996,0.00025292407,0.00028024247,0.000044646902],"domain_scores_gemma":[0.9973442,0.00094896293,0.00014848214,0.0006883262,0.0007844688,0.000085543936],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015558188,0.00084034225,0.0008224177,0.0014984409,0.0007821397,0.0011874784,0.0015684136,0.00070702855,0.023001693],"category_scores_gemma":[0.007125543,0.00053395866,0.0005731827,0.0016493504,0.00035341905,0.0025661502,0.0015570141,0.0012877132,0.012677872],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010777229,0.00038809417,0.0033324729,0.0010841461,0.00022560415,0.0010919695,0.0006869605,0.010272275,0.11041871,0.013768003,0.14266038,0.7149936],"study_design_scores_gemma":[0.00040624585,0.00084260065,0.0065143188,0.00011545462,0.00042085818,0.0028970104,0.0003745054,0.48331866,0.22383535,0.019501455,0.26146072,0.0003129417],"about_ca_topic_score_codex":0.003394989,"about_ca_topic_score_gemma":0.0047756005,"teacher_disagreement_score":0.023001693,"about_ca_system_score_codex":0.00044673143,"about_ca_system_score_gemma":0.0012604437,"threshold_uncertainty_score":0.076948345},"labels":[],"label_agreement":null},{"id":"W7043940457","doi":"","title":"WN-LEXICAL: An ACT-R module built from the WordNet lexical database","year":2006,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Lexical database; WordNet; Focus (optics); Domain (mathematical analysis); Cognition; Architecture; Natural language; Cognitive architecture","score_opus":0.02100929181997913,"score_gpt":0.2842755969141184,"score_spread":0.2632663050941393,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7043940457","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012903298,0.00040971537,0.58126473,0.00036966338,0.0003060998,0.0007925387,0.019734666,0.36415848,0.020060727],"genre_scores_gemma":[0.16168244,0.00081410544,0.67646027,0.0011670323,0.00022317447,0.0020978693,0.070741475,0.035949115,0.050864518],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99947554,0.00007684745,0.00006752683,0.0002092591,0.00012711999,0.000043723096],"domain_scores_gemma":[0.9992747,0.00024643305,0.00005128478,0.0002403714,0.00012448584,0.000062748295],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00097321265,0.00172402,0.0011341203,0.0022583446,0.0004301241,0.0022250111,0.0026760828,0.0008666459,0.03876682],"category_scores_gemma":[0.0032593699,0.0010319906,0.001170893,0.0014341057,0.0005002653,0.0045515117,0.0019440629,0.0009402243,0.026122373],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018906435,0.0005663223,0.007934441,0.0023729976,0.0006344402,0.0008034675,0.0007165993,0.009906929,0.025424289,0.037650432,0.37088385,0.5412156],"study_design_scores_gemma":[0.00048627733,0.0004183039,0.0063270135,0.00027363852,0.00044817326,0.0012436584,0.00046379186,0.37991294,0.07983604,0.08207912,0.44819963,0.00031147493],"about_ca_topic_score_codex":0.003268768,"about_ca_topic_score_gemma":0.0045858095,"teacher_disagreement_score":0.03876682,"about_ca_system_score_codex":0.00063672225,"about_ca_system_score_gemma":0.0010640383,"threshold_uncertainty_score":0.12968796},"labels":[],"label_agreement":null},{"id":"W7065903924","doi":"","title":"23 especies vegetales medicinales de uso frecuente en la población de Tabay.","year":2002,"lang":"es","type":"other","venue":"Actualidad Contable FACES","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Context (archaeology); Order (exchange)","score_opus":0.012030912025114159,"score_gpt":0.2718161088057705,"score_spread":0.25978519678065637,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7065903924","genre_codex":"review","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1945012,0.6771444,0.003625111,0.0054531684,0.004090342,0.00016935622,0.002978939,0.00018660267,0.11185082],"genre_scores_gemma":[0.45866707,0.30821127,0.007891836,0.0023102157,0.00085666694,0.00007297916,0.0026609304,0.00012231294,0.21920668],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99971217,0.000070075526,0.000024277204,0.000055704786,0.000099827426,0.000037883536],"domain_scores_gemma":[0.99976355,0.00003933434,0.000058195346,0.000013990398,0.00009591725,0.000029011982],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000525548,0.0005097006,0.00049503293,0.0018656409,0.0010182821,0.001636405,0.00033380822,0.0005736385,0.014731637],"category_scores_gemma":[0.0004197845,0.00021289183,0.000568694,0.0029608896,0.00034626992,0.0007695683,0.00051935413,0.0007851723,0.0022906326],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009778278,0.00025672847,0.015140326,0.008443035,0.00026146712,0.0019190199,0.004753297,0.00035287306,0.29239306,0.0052789487,0.018314088,0.65190923],"study_design_scores_gemma":[0.000020542733,0.0006695467,0.096771464,0.0015994228,0.0002812899,0.0018000294,0.0029066426,0.00019079624,0.021104451,0.000971441,0.873634,0.00005050556],"about_ca_topic_score_codex":0.015884753,"about_ca_topic_score_gemma":0.056203276,"teacher_disagreement_score":0.015884753,"about_ca_system_score_codex":0.0014113138,"about_ca_system_score_gemma":0.0014115095,"threshold_uncertainty_score":0.049282253},"labels":[],"label_agreement":null},{"id":"W7066217479","doi":"","title":"2007. Improving translation quality by discarding most of the phrasetable","year":2007,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"Advanced Research Projects Agency; Defense Advanced Research Projects Agency","keywords":"BLEU; Phrase; Machine translation; Smoothing; Translation (biology); Quality (philosophy)","score_opus":0.014618568705312935,"score_gpt":0.28638695339763715,"score_spread":0.2717683846923242,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7066217479","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.062911466,0.008136783,0.81063676,0.0082568815,0.002928116,0.00043816247,0.022302136,0.04682624,0.03756341],"genre_scores_gemma":[0.10307393,0.002545364,0.8005431,0.00088782556,0.00086772814,0.00029312624,0.047105163,0.004252292,0.040431503],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988342,0.00042785204,0.00009767275,0.00017933821,0.00040673083,0.000054284326],"domain_scores_gemma":[0.99604946,0.0012637443,0.00032171604,0.0011693968,0.0010840814,0.000111640555],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00278095,0.0015383074,0.00083520333,0.0017970693,0.00066264183,0.0016347669,0.001189783,0.001212257,0.020025155],"category_scores_gemma":[0.008436591,0.00062038976,0.0005776002,0.0026621772,0.0004316418,0.0025937597,0.0014634257,0.001438136,0.029915823],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005152177,0.00012623689,0.0016335223,0.0005074716,0.00009696167,0.00020891415,0.00022675126,0.0033530318,0.047025416,0.0063825394,0.18618701,0.75373685],"study_design_scores_gemma":[0.000597861,0.00055279053,0.018896306,0.00025425453,0.00061537663,0.003178244,0.00029738012,0.18121046,0.21699823,0.04213957,0.53505796,0.00020152243],"about_ca_topic_score_codex":0.0019705598,"about_ca_topic_score_gemma":0.0055827308,"teacher_disagreement_score":0.020025155,"about_ca_system_score_codex":0.00047055876,"about_ca_system_score_gemma":0.0010190605,"threshold_uncertainty_score":0.06699085},"labels":[],"label_agreement":null},{"id":"W7066685239","doi":"","title":"Human antigen R (HuR) regulates Cigarette Smoke- induced inflammation: Implications in COPD","year":2018,"lang":"en","type":"dissertation","venue":"eScholarship@McGill (McGill)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"COPD; Inflammation; Lung; Cigarette smoke; Messenger RNA; Antigen; Immunohistochemistry; Cytoplasm; Pathogenesis","score_opus":0.022700261547950212,"score_gpt":0.28769020717796895,"score_spread":0.26498994563001876,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7066685239","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.97083044,0.025974028,0.0009757995,0.0004359848,0.000046963432,0.000021495665,0.00013337807,0.0000348122,0.0015471161],"genre_scores_gemma":[0.9921312,0.0060003498,0.00057302165,0.00013347258,0.00003506152,0.000012217573,0.00009742794,0.0000032988312,0.0010139496],"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9999238,0.000018856757,0.000005367374,0.000017344264,0.000016345643,0.000018230881],"domain_scores_gemma":[0.99994063,0.000009097544,0.000022941475,0.0000029272035,0.000011434166,0.00001302143],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000119786215,0.00013932273,0.0001678386,0.00023355099,0.00023568819,0.00029377252,0.000100219564,0.00025914513,0.0006909543],"category_scores_gemma":[0.0001092043,0.000093585215,0.00018950876,0.00017169151,0.0002363205,0.00014064256,0.00015749797,0.00022339378,0.00014515095],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00067794876,0.00010314248,0.04509625,0.00029748568,0.000047345537,0.0008825565,0.00021146417,0.000099697114,0.9356431,0.00038877118,0.0003495492,0.016202832],"study_design_scores_gemma":[0.000043556072,0.0009811586,0.83163327,0.000073830255,0.0001558878,0.0034360301,0.0007561747,0.0016399145,0.15318622,0.0007023014,0.007370866,0.000020917614],"about_ca_topic_score_codex":0.00051512354,"about_ca_topic_score_gemma":0.0007719447,"teacher_disagreement_score":0.0006909543,"about_ca_system_score_codex":0.00016297089,"about_ca_system_score_gemma":0.0001338848,"threshold_uncertainty_score":0.0023115277},"labels":[],"label_agreement":null},{"id":"W7067195615","doi":"","title":"Managing multiple land uses : applications in subarctic Urko Kekkonen National Park, Finland","year":2004,"lang":"en","type":"dissertation","venue":"eScholarship@McGill (McGill)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Recreation; Visitor pattern; National park; Subarctic climate; Tourism; Land use; Land management; Protected area","score_opus":0.01489271324648239,"score_gpt":0.2614507683889322,"score_spread":0.24655805514244983,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7067195615","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7594766,0.0099121975,0.057957076,0.0066649406,0.00018804072,0.00061582867,0.023779469,0.010949992,0.13045591],"genre_scores_gemma":[0.7538324,0.0067532887,0.15858766,0.00016993267,0.00007539328,0.00036049343,0.010092235,0.0005493105,0.069579184],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9998838,0.000026443135,0.000008216613,0.000035125267,0.000029854751,0.000016569482],"domain_scores_gemma":[0.9995733,0.0002135412,0.000040013878,0.000031670523,0.000071013645,0.000070454145],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00052977254,0.00026621835,0.00023321233,0.0010499366,0.00079558097,0.0013423625,0.00062300346,0.00044322122,0.015151661],"category_scores_gemma":[0.0009469849,0.00013479803,0.0002225916,0.0022811757,0.00019551729,0.0013907176,0.00080687786,0.00016746421,0.0018111272],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037782016,0.0003093767,0.105470344,0.0013148176,0.00011184393,0.0027855828,0.0074434355,0.020222556,0.0071568317,0.0047538867,0.048990857,0.8010626],"study_design_scores_gemma":[0.00022468319,0.00028061602,0.28643334,0.0007933922,0.0003074701,0.002312335,0.03242725,0.13127638,0.008667721,0.014543645,0.52260226,0.00013091114],"about_ca_topic_score_codex":0.055355806,"about_ca_topic_score_gemma":0.16425604,"teacher_disagreement_score":0.055355806,"about_ca_system_score_codex":0.0008286965,"about_ca_system_score_gemma":0.0011324352,"threshold_uncertainty_score":0.11006725},"labels":[],"label_agreement":null},{"id":"W7067437880","doi":"","title":"Long and winding road adolescents and youth in Canada today","year":2001,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Government (linguistics); Work (physics); Agency (philosophy); Population","score_opus":0.009782146385781666,"score_gpt":0.22815612752337977,"score_spread":0.2183739811375981,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7067437880","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.974625,0.0019685933,0.00008253174,0.0076308893,0.00011682633,0.000036548012,0.0029458825,0.00001899327,0.012574618],"genre_scores_gemma":[0.97776586,0.00204384,0.00013683618,0.0012616989,0.000028667575,0.000026488884,0.0011570592,0.00001579068,0.017563792],"study_design_codex":"observational","study_design_gemma":"not_applicable","domain_scores_codex":[0.998516,0.00007742092,0.000040420902,0.00010397706,0.00026736554,0.0009947544],"domain_scores_gemma":[0.9975944,0.00008950704,0.0002713866,0.000023068476,0.0006778524,0.0013437321],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00049145456,0.0002287517,0.0004822918,0.0016669331,0.009391264,0.0025969837,0.0014229957,0.0012255587,0.008107787],"category_scores_gemma":[0.0012967619,0.00031796287,0.0003927909,0.0043715066,0.0013338075,0.001269067,0.0019134725,0.0024516669,0.00078663154],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009257534,0.00015193252,0.8974431,0.00008128004,0.00002135604,0.0011188608,0.045664735,0.00007962103,0.000215804,0.0024082756,0.02030203,0.032420434],"study_design_scores_gemma":[0.0000067689575,0.000029815956,0.68181497,0.0001821274,0.000024990735,0.00027346754,0.2960802,0.00009818114,0.00011233809,0.00030332446,0.021045234,0.000028646828],"about_ca_topic_score_codex":0.99432254,"about_ca_topic_score_gemma":0.99884856,"teacher_disagreement_score":0.029784516,"about_ca_system_score_codex":0.029784516,"about_ca_system_score_gemma":0.044539455,"threshold_uncertainty_score":0.21610278},"labels":[],"label_agreement":null},{"id":"W7067512125","doi":"","title":"Mamook Komtax Chinuk Pipa / Learning to Write Chinook Jargon : Indigenous Peoples and Literacy Strategies in the South Central Interior of British Columbia in the Late Nineteenth Century","year":2017,"lang":"en","type":"other","venue":"University of Hertfordshire Research Archive (University of Hertfordshire)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Indigenous; Literacy; Jargon; Permission; Economic Justice; Adult literacy","score_opus":0.010092479212767756,"score_gpt":0.24716508999670997,"score_spread":0.2370726107839422,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7067512125","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9433618,0.0019057319,0.00013775755,0.001979909,0.00002720641,0.00002097982,0.00010724303,0.000010826897,0.05244846],"genre_scores_gemma":[0.9715729,0.001516104,0.00015555239,0.00015619531,0.0000041697826,0.00001258097,0.000053406107,0.000016002272,0.02651306],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9996691,0.00006788966,0.000012934474,0.000042640535,0.00007356987,0.00013385892],"domain_scores_gemma":[0.9996136,0.000110279136,0.000035944682,0.000015632244,0.0001136129,0.000111022455],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004405918,0.00020850741,0.0002844664,0.0010171297,0.0072245034,0.0034231935,0.0004472649,0.00054275425,0.0066165966],"category_scores_gemma":[0.0011025508,0.00019824866,0.00007973549,0.0018982105,0.004216779,0.0011112193,0.0016748091,0.0010752467,0.00046824903],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000037729526,0.000028817429,0.01545753,0.000103051665,0.0000052107994,0.00081957463,0.9358744,0.000031903433,0.00085272675,0.004596237,0.0029771463,0.039215736],"study_design_scores_gemma":[0.0000033102317,0.000011546553,0.0648378,0.00018748242,0.000012878162,0.00021040674,0.89374113,0.0000562642,0.00029768673,0.0007304958,0.039894737,0.000016306061],"about_ca_topic_score_codex":0.9142549,"about_ca_topic_score_gemma":0.9749594,"teacher_disagreement_score":0.085745096,"about_ca_system_score_codex":0.011003111,"about_ca_system_score_gemma":0.0117621245,"threshold_uncertainty_score":0.17250007},"labels":[],"label_agreement":null},{"id":"W7067601573","doi":"","title":"L’histoire de la pharmacie au Québec. Collin (Johanne), Nouvelle ordonnance. Quatre siècles d’histoire de la pharmacie au Québec, Montréal, Les Presses de l’Université de Montréal, 2020","year":2020,"lang":"fr","type":"other","venue":"Persée (Ministère de lEnseignement supérieur et de la Recherche)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Context (archaeology); Barrel (horology); Assembly line","score_opus":0.029708166210423416,"score_gpt":0.2937784684352638,"score_spread":0.26407030222484035,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7067601573","genre_codex":"review","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0026212758,0.47231367,0.0032347573,0.23084076,0.023848455,0.00020767574,0.02002058,0.0005249486,0.24638793],"genre_scores_gemma":[0.02757905,0.10118739,0.0035583028,0.017369756,0.0009295836,0.00009978827,0.0047461307,0.00035046085,0.8441796],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9978219,0.0002077944,0.00010701186,0.00025156626,0.0012692378,0.00034248823],"domain_scores_gemma":[0.9944378,0.00042486814,0.00024432538,0.00015748077,0.003893674,0.00084187096],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024357082,0.00083210797,0.00042841752,0.0025082224,0.003915028,0.0047413167,0.0012258606,0.0023402036,0.064953364],"category_scores_gemma":[0.0047286274,0.0006286788,0.0005032202,0.006446612,0.0023745797,0.002423618,0.0009684062,0.0030976287,0.012450614],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000017921811,0.0000073680144,0.0005733429,0.0001566095,0.0000052460236,0.00003901897,0.00032298957,0.00007802607,0.00017384511,0.004528633,0.9582493,0.03584779],"study_design_scores_gemma":[0.000002982056,0.0000029691555,0.0036243543,0.00020256516,0.0000035805222,0.000017290933,0.00026980665,0.000021542526,0.000081507234,0.00025169353,0.99551165,0.000010040414],"about_ca_topic_score_codex":0.98340535,"about_ca_topic_score_gemma":0.9925184,"teacher_disagreement_score":0.066195115,"about_ca_system_score_codex":0.066195115,"about_ca_system_score_gemma":0.09432385,"threshold_uncertainty_score":0.4802814},"labels":[],"label_agreement":null},{"id":"W7070290275","doi":"","title":"Belt","year":2008,"lang":"en","type":"other","venue":"Wolverhampton Intellectual Repository and E-Theses (University of Wolverhampton)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"The arts; Art gallery; Study abroad","score_opus":0.00899498234243636,"score_gpt":0.19421693524202352,"score_spread":0.18522195289958715,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7070290275","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0009169978,0.0004645935,0.0018504397,0.0012774934,0.0017426461,0.00011857765,0.004009889,0.0013608551,0.9882585],"genre_scores_gemma":[0.0043283054,0.00028108884,0.0007868152,0.00042201785,0.00014406207,0.00005327116,0.0028463656,0.0004469627,0.99069107],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9994567,0.000056825465,0.000030511772,0.00013899032,0.00022529351,0.00009169045],"domain_scores_gemma":[0.9991891,0.00008231003,0.00004328723,0.00013331752,0.00039663675,0.00015535466],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00068782887,0.00091746025,0.00049158453,0.0020037498,0.002300261,0.0037323195,0.0013104117,0.0017148785,0.7827153],"category_scores_gemma":[0.0021142522,0.0003023504,0.00045348195,0.0016721339,0.00040947672,0.0031281211,0.0025592747,0.0011469956,0.5761231],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000120270626,0.000061683706,0.0006329623,0.00020289622,0.0000044258404,0.00015971516,0.00036667072,0.00008674193,0.00075555884,0.020130431,0.849452,0.1280266],"study_design_scores_gemma":[0.0000044594967,0.000007224719,0.0003343282,0.00004315902,0.0000013449616,0.000056181267,0.00011790455,0.000036659232,0.00013442193,0.00078287406,0.99847823,0.0000031302627],"about_ca_topic_score_codex":0.0038567334,"about_ca_topic_score_gemma":0.007706367,"teacher_disagreement_score":0.7827153,"about_ca_system_score_codex":0.0009494609,"about_ca_system_score_gemma":0.0010347984,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W7072301867","doi":"","title":"Unsupervised Learning of Morphology for English and Inukitut","year":2003,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Unsupervised learning; Morphology (biology); Pattern recognition (psychology); Feature (linguistics); Statistical learning","score_opus":0.011041158528513848,"score_gpt":0.24879766127639755,"score_spread":0.2377565027478837,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7072301867","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.30267146,0.0017981753,0.6181966,0.0017991299,0.0009793302,0.0002790782,0.014952503,0.020945355,0.03837841],"genre_scores_gemma":[0.7152543,0.0006033855,0.2261893,0.00021508773,0.00024214463,0.00018431111,0.03206279,0.0026978238,0.022550829],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99923027,0.00021850891,0.00006388306,0.00030643083,0.000090813446,0.00009020604],"domain_scores_gemma":[0.99748623,0.0010335881,0.00017015822,0.00038392917,0.000782436,0.00014359033],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007816111,0.0009192803,0.0005986408,0.0019123753,0.0011776283,0.0022578042,0.0011837499,0.0008736828,0.010920153],"category_scores_gemma":[0.0043366337,0.0005025115,0.0014211959,0.0014843859,0.0005641505,0.0028591063,0.0017176928,0.0021318253,0.0075169215],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005000867,0.00020020503,0.0112176025,0.0005138126,0.00016714296,0.0007915989,0.0010874671,0.006837732,0.037802823,0.014886341,0.039704617,0.88629067],"study_design_scores_gemma":[0.00013746847,0.0005340087,0.043399688,0.00034939015,0.00046079865,0.0025320326,0.003589914,0.703783,0.07726331,0.06370486,0.104060054,0.00018551902],"about_ca_topic_score_codex":0.004563325,"about_ca_topic_score_gemma":0.011209767,"teacher_disagreement_score":0.010920153,"about_ca_system_score_codex":0.0008366115,"about_ca_system_score_gemma":0.0013100591,"threshold_uncertainty_score":0.036531627},"labels":[],"label_agreement":null},{"id":"W7082251445","doi":"10.48448/hhwp-0r98","title":"TRANSLATIONCORRECT: A Unified Framework for Machine Translation Post-Editing with Predictive Error Assistance","year":2025,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Annotation; Machine translation; Translation (biology); Error detection and correction; Quality (philosophy); Interface (matter)","score_opus":0.01731375629733018,"score_gpt":0.30561847800683156,"score_spread":0.2883047217095014,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7082251445","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0004695473,0.00020865376,0.9408549,0.000147747,0.000096163094,0.00017952597,0.00064589886,0.0561614,0.0012360658],"genre_scores_gemma":[0.02326827,0.00055290427,0.9543774,0.00028223678,0.00017940505,0.0006539286,0.00436727,0.011534483,0.004784112],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99420255,0.0018601649,0.0007677023,0.0011567703,0.0017379019,0.0002748397],"domain_scores_gemma":[0.9904544,0.0036993667,0.0007512337,0.002581343,0.0021225866,0.0003910868],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007930784,0.0027539688,0.0016793901,0.0038589619,0.0014783035,0.0056661586,0.0043537957,0.0026016557,0.0152629055],"category_scores_gemma":[0.023327075,0.0014482326,0.0030068906,0.0023618427,0.0019225065,0.004606497,0.006447085,0.003783926,0.014737418],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008790199,0.0002529053,0.0024772636,0.0025675683,0.00051507994,0.0011291532,0.0025859901,0.03679303,0.017794417,0.097637326,0.10948895,0.7278792],"study_design_scores_gemma":[0.00016604889,0.00025116693,0.0011386358,0.0006160466,0.00022203739,0.0011190292,0.00045061714,0.4337335,0.037445437,0.15382767,0.3706699,0.00035989514],"about_ca_topic_score_codex":0.0050087012,"about_ca_topic_score_gemma":0.0073194336,"teacher_disagreement_score":0.0152629055,"about_ca_system_score_codex":0.0011977953,"about_ca_system_score_gemma":0.00409061,"threshold_uncertainty_score":0.051059484},"labels":[],"label_agreement":null},{"id":"W7083846372","doi":"","title":"Birthday Roundup: Conan O'Brien, Melissa Joan Hart, and Maria Bello","year":2019,"lang":"en","type":"other","venue":"Internet Archive (Internet Archive)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Honor; Performance art; White (mutation)","score_opus":0.009113083610378516,"score_gpt":0.2397990733391486,"score_spread":0.23068598972877008,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7083846372","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.002394678,0.0027243278,0.00056608365,0.022521757,0.013155064,0.000096073745,0.001266279,0.0009241873,0.9563516],"genre_scores_gemma":[0.0017528955,0.0004088443,0.00007852752,0.0008190703,0.00018788847,0.000013646844,0.00016199131,0.00019316746,0.9963839],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99974424,0.00003179573,0.0000059361323,0.00005050927,0.000094139024,0.00007337129],"domain_scores_gemma":[0.9991773,0.000046501602,0.00003046057,0.00003906361,0.00019091801,0.0005157346],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.00041506664,0.00069042173,0.00039246675,0.00066638284,0.0057797064,0.003731054,0.0006409181,0.0011801578,0.6419622],"category_scores_gemma":[0.0013129145,0.0004280177,0.00029955752,0.0005896987,0.00052488566,0.0029407847,0.002564176,0.002853595,0.32773158],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000010411833,0.000010825602,0.00008264564,0.000014495405,4.9132484e-7,0.000052344625,0.0001802389,0.000002618412,0.0000711631,0.0006834632,0.9892993,0.009592019],"study_design_scores_gemma":[0.0000018900378,0.000005747616,0.0004402129,0.00003091732,7.365456e-7,0.00007358246,0.00069519336,0.000011330952,0.000097575044,0.00012091443,0.9985185,0.0000034424759],"about_ca_topic_score_codex":0.01610895,"about_ca_topic_score_gemma":0.08435301,"teacher_disagreement_score":0.6419622,"about_ca_system_score_codex":0.0014345755,"about_ca_system_score_gemma":0.0010461052,"threshold_uncertainty_score":0.5106975},"labels":[],"label_agreement":null},{"id":"W7084086837","doi":"10.5281/zenodo.16367716","title":"Citation Parsing and Analysis with Language Models","year":2025,"lang":"en","type":"other","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Metadata; Citation; Parsing; Open research; Citation analysis; Computational linguistics","score_opus":0.01861051643420738,"score_gpt":0.25122767174844157,"score_spread":0.2326171553142342,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7084086837","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008308783,0.0011824814,0.9494216,0.0018884462,0.00042241823,0.00017823225,0.0054678824,0.021376848,0.011753383],"genre_scores_gemma":[0.27369505,0.0016376522,0.6714119,0.0006908895,0.00080322265,0.0007488057,0.023960207,0.009616815,0.017435469],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99289274,0.0035457734,0.00062682584,0.0013589156,0.0011425492,0.00043325662],"domain_scores_gemma":[0.9826276,0.011807036,0.00068451505,0.0023854743,0.0022713707,0.00022407225],"candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.00622957,0.0016828881,0.0015144409,0.008367862,0.0022311544,0.008078913,0.002757474,0.002160883,0.018743053],"category_scores_gemma":[0.03002559,0.0014545053,0.0042125965,0.009238061,0.0016719771,0.010714687,0.004371884,0.004154841,0.0119986925],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036770612,0.00033056157,0.0032365003,0.0008029146,0.0002792717,0.00043241415,0.0014032979,0.06579311,0.0034173762,0.41088033,0.06457636,0.44848016],"study_design_scores_gemma":[0.00004044237,0.000025947746,0.00065613055,0.00011674042,0.0001051014,0.00013575073,0.00026443021,0.4400921,0.004646169,0.51247555,0.041356795,0.0000847485],"about_ca_topic_score_codex":0.009534411,"about_ca_topic_score_gemma":0.008369741,"teacher_disagreement_score":0.9937704,"about_ca_system_score_codex":0.0023747492,"about_ca_system_score_gemma":0.0040325564,"threshold_uncertainty_score":0.06270182},"labels":[],"label_agreement":null},{"id":"W7086614411","doi":"","title":"Toward a luminescence chronology for coastal dune and beach deposits on Calvert Island, British Columbia central coast, Canada","year":2015,"lang":"en","type":"article","venue":"University of Washington Tacoma Digital Commons (University of Washington Tacoma)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Chronology; Radiocarbon dating; Thermoluminescence dating; Optically stimulated luminescence; Plateau (mathematics); Glacial period; Quaternary; Pleistocene","score_opus":0.01173226436661877,"score_gpt":0.18484572412413522,"score_spread":0.17311345975751646,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7086614411","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9299523,0.0041418015,0.0060946094,0.00074331847,0.000097196586,0.00054813386,0.016961241,0.00041684636,0.041044425],"genre_scores_gemma":[0.93821335,0.0026133792,0.025720995,0.00032963083,0.000021071228,0.00025328374,0.008808311,0.00013747244,0.02390252],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99960023,0.00002154061,0.0000320658,0.00008399586,0.00018205347,0.00008017479],"domain_scores_gemma":[0.9968509,0.000062916406,0.00020019311,0.00006631824,0.0026403274,0.00017929953],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005916094,0.0003839453,0.00019498279,0.006428556,0.0029160825,0.001852965,0.0010117023,0.0002593263,0.0030258722],"category_scores_gemma":[0.0012949909,0.00023423632,0.00014171007,0.0052431277,0.0005254844,0.0004310881,0.00060434325,0.00053959433,0.00060796237],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011261583,0.00007954966,0.79554677,0.00026546157,0.00004886039,0.00044122638,0.004223152,0.0014383503,0.018608602,0.0009830209,0.013700896,0.16455151],"study_design_scores_gemma":[0.000008160699,0.000025374444,0.96096826,0.00015729025,0.000023816543,0.000104449464,0.0036303236,0.001092568,0.0019290794,0.00006466196,0.031975646,0.000020336796],"about_ca_topic_score_codex":0.9880918,"about_ca_topic_score_gemma":0.99827707,"teacher_disagreement_score":0.02200258,"about_ca_system_score_codex":0.02200258,"about_ca_system_score_gemma":0.020763533,"threshold_uncertainty_score":0.15964061},"labels":[],"label_agreement":null},{"id":"W7089367948","doi":"10.5281/zenodo.17333830","title":"Replication Package for SANER26","year":2025,"lang":"en","type":"other","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Replication (statistics); Security token; R package; Key (lock); Raw data","score_opus":0.023685938417670048,"score_gpt":0.27655025201151834,"score_spread":0.2528643135938483,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7089367948","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":"reproducibility","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":"reproducibility","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0009342441,0.00012877803,0.052499514,0.000761311,0.002886142,0.0006315344,0.7729628,0.15034598,0.01884973],"genre_scores_gemma":[0.0056890827,0.00015512371,0.06586643,0.0006843472,0.00067295064,0.0044325655,0.6854214,0.2080707,0.029007409],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9900058,0.0027960245,0.0012817216,0.0026245676,0.0025774932,0.0007144244],"domain_scores_gemma":[0.9528634,0.011667685,0.0016356476,0.01956539,0.01279565,0.0014721608],"candidate_categories":["metaresearch","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.013782235,0.003218962,0.003611467,0.0051266854,0.0028985264,0.0058574113,0.0062467153,0.0021456033,0.58153635],"category_scores_gemma":[0.06690232,0.0029522437,0.0047882358,0.005932956,0.001020599,0.0040945862,0.005850715,0.0055296794,0.51545143],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023957493,0.000032012947,0.00024705988,0.0004494674,0.00010767361,0.000038271857,0.000087164975,0.0003570722,0.00048860797,0.0014943634,0.98796695,0.00849172],"study_design_scores_gemma":[0.0006330396,0.00007283452,0.0015476341,0.0002314628,0.00012596625,0.00012956202,0.000072663155,0.002111622,0.0029244295,0.011088048,0.9809075,0.00015534372],"about_ca_topic_score_codex":0.0068488875,"about_ca_topic_score_gemma":0.009098212,"teacher_disagreement_score":0.98621774,"about_ca_system_score_codex":0.0014800768,"about_ca_system_score_gemma":0.0054944707,"threshold_uncertainty_score":0.5968876},"labels":[],"label_agreement":null},{"id":"W7089531290","doi":"10.5281/zenodo.17333778","title":"Replication Package for SANER26","year":2025,"lang":"en","type":"other","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Replication (statistics); Security token; R package; Key (lock); Raw data","score_opus":0.023685938417670048,"score_gpt":0.27655025201151834,"score_spread":0.2528643135938483,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7089531290","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":"reproducibility","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":"reproducibility","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0009342441,0.00012877803,0.052499514,0.000761311,0.002886142,0.0006315344,0.7729628,0.15034598,0.01884973],"genre_scores_gemma":[0.0056890827,0.00015512371,0.06586643,0.0006843472,0.00067295064,0.0044325655,0.6854214,0.2080707,0.029007409],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9900058,0.0027960245,0.0012817216,0.0026245676,0.0025774932,0.0007144244],"domain_scores_gemma":[0.9528634,0.011667685,0.0016356476,0.01956539,0.01279565,0.0014721608],"candidate_categories":["metaresearch","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.013782235,0.003218962,0.003611467,0.0051266854,0.0028985264,0.0058574113,0.0062467153,0.0021456033,0.58153635],"category_scores_gemma":[0.06690232,0.0029522437,0.0047882358,0.005932956,0.001020599,0.0040945862,0.005850715,0.0055296794,0.51545143],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023957493,0.000032012947,0.00024705988,0.0004494674,0.00010767361,0.000038271857,0.000087164975,0.0003570722,0.00048860797,0.0014943634,0.98796695,0.00849172],"study_design_scores_gemma":[0.0006330396,0.00007283452,0.0015476341,0.0002314628,0.00012596625,0.00012956202,0.000072663155,0.002111622,0.0029244295,0.011088048,0.9809075,0.00015534372],"about_ca_topic_score_codex":0.0068488875,"about_ca_topic_score_gemma":0.009098212,"teacher_disagreement_score":0.98621774,"about_ca_system_score_codex":0.0014800768,"about_ca_system_score_gemma":0.0054944707,"threshold_uncertainty_score":0.5968876},"labels":[],"label_agreement":null},{"id":"W7091186536","doi":"","title":"Happiness is Sharing a Vocabulary: A Study of Transliteration Methods","year":2025,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Division of Human Resource Development; Institute for Information and Communications Technology Promotion; Korea Institute for Advancement of Technology; Ministry of Science and ICT, South Korea; Institute for Computing, Information and Cognitive Systems; Ministry of Trade, Industry and Energy","keywords":"Transliteration; Inference; Security token; Natural language; Scripting language; Factor (programming language); Romanization","score_opus":0.056753991681672736,"score_gpt":0.26166107145234596,"score_spread":0.20490707977067324,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7091186536","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.22986679,0.0038362965,0.73385626,0.0030770395,0.00019886412,0.00031014552,0.0003251438,0.0012056549,0.02732382],"genre_scores_gemma":[0.87633634,0.0019231054,0.11083095,0.00050216983,0.00022087236,0.00033312588,0.0007999898,0.0012786362,0.0077747633],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9906827,0.0063302177,0.00037987644,0.0015055344,0.0008379529,0.00026367433],"domain_scores_gemma":[0.9602539,0.031846836,0.0015583016,0.004191515,0.0017534245,0.00039599798],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.015291945,0.0008844964,0.0009854875,0.0014215615,0.001293753,0.0039734924,0.0023834375,0.0013450922,0.0047403513],"category_scores_gemma":[0.051288772,0.000709356,0.0010741773,0.0016097181,0.002864168,0.015021844,0.004281515,0.0020597891,0.0020138763],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010372512,0.00042628465,0.030873738,0.00097384065,0.0005237553,0.00061353855,0.015333607,0.084352076,0.011844598,0.20622042,0.0063053113,0.6414955],"study_design_scores_gemma":[0.000168892,0.00059021637,0.0074371,0.000344096,0.00025563233,0.0010905667,0.006206314,0.76112753,0.0149898045,0.17782368,0.029832304,0.00013385926],"about_ca_topic_score_codex":0.0033764231,"about_ca_topic_score_gemma":0.0017754887,"teacher_disagreement_score":0.015291945,"about_ca_system_score_codex":0.0015452647,"about_ca_system_score_gemma":0.0011538761,"threshold_uncertainty_score":0.080872476},"labels":[],"label_agreement":null},{"id":"W7095700816","doi":"","title":"University of Alberta Web-assisted Anaphora Resolution","year":2016,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Permission; Resolution (logic); Anaphora (linguistics); Association (psychology)","score_opus":0.00979825388227285,"score_gpt":0.21924956628412356,"score_spread":0.20945131240185072,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7095700816","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013961724,0.01313055,0.10497324,0.008363447,0.003320871,0.00041427382,0.017313743,0.026793761,0.81172836],"genre_scores_gemma":[0.082370155,0.008819919,0.18572494,0.0007195993,0.00035438803,0.0001642608,0.0188113,0.0025254337,0.70051],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.998885,0.00017175227,0.000047501875,0.00020668318,0.00058470975,0.0001043732],"domain_scores_gemma":[0.99798965,0.00040145515,0.000048198675,0.00025141737,0.0010849456,0.00022423625],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016380567,0.0007942701,0.0006202245,0.0022800073,0.0019917635,0.00535295,0.0013261789,0.0011217309,0.1233429],"category_scores_gemma":[0.0031471432,0.0004969543,0.00042498973,0.0024376612,0.00076564495,0.0015128864,0.0019101306,0.00089770544,0.044458907],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00042652895,0.00010557432,0.00060551666,0.00049334986,0.000018640505,0.00040813017,0.00061220286,0.002215282,0.006919285,0.020710262,0.52167284,0.44581237],"study_design_scores_gemma":[0.000061180726,0.00002448325,0.0018200447,0.00026672104,0.000023809387,0.00032168374,0.00046743394,0.008304906,0.0065328004,0.0074040554,0.97472453,0.000048236434],"about_ca_topic_score_codex":0.33428845,"about_ca_topic_score_gemma":0.427373,"teacher_disagreement_score":0.33428845,"about_ca_system_score_codex":0.004695882,"about_ca_system_score_gemma":0.00948906,"threshold_uncertainty_score":0.66468537},"labels":[],"label_agreement":null},{"id":"W7096073816","doi":"","title":"H.: Improved word alignment using a symmetric lexicon model","year":2013,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Lexicon; Word (group theory); Smoothing; Natural language; Task (project management); Translation (biology); Machine translation","score_opus":0.02303242515188392,"score_gpt":0.2711153241812731,"score_spread":0.24808289902938918,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7096073816","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029665396,0.0004490998,0.93792343,0.0004599896,0.000268855,0.00021545454,0.0013108545,0.023717763,0.0059891995],"genre_scores_gemma":[0.2815311,0.00040018585,0.68087524,0.0008196377,0.00031827664,0.00040597163,0.008429966,0.0035684067,0.023651218],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9987913,0.00038152744,0.000093536655,0.00036042155,0.0002572407,0.0001158882],"domain_scores_gemma":[0.9985379,0.00040828375,0.00010101038,0.0005974887,0.00030057362,0.000054677897],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013328893,0.0012597529,0.0011985697,0.0017332145,0.0008315658,0.0015089178,0.0016231794,0.0014163704,0.0088616],"category_scores_gemma":[0.003018826,0.0007613175,0.0013081278,0.0018832413,0.0006591194,0.0036340193,0.0019828065,0.0011829935,0.0099297175],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006473326,0.00033160494,0.0021866295,0.00029551313,0.00027609587,0.00037754234,0.00031442693,0.079527,0.038489934,0.021845423,0.050717153,0.8049914],"study_design_scores_gemma":[0.00019833929,0.00013792003,0.0010514479,0.000016206184,0.000087189226,0.00020541444,0.000085985776,0.93768513,0.01861192,0.029762512,0.012092143,0.000065857326],"about_ca_topic_score_codex":0.013170661,"about_ca_topic_score_gemma":0.019817412,"teacher_disagreement_score":0.013170661,"about_ca_system_score_codex":0.0007426745,"about_ca_system_score_gemma":0.0026507478,"threshold_uncertainty_score":0.029645026},"labels":[],"label_agreement":null},{"id":"W7096309851","doi":"","title":"Centre for Computational Linguistics","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computational linguistics; Word (group theory); Transfer (computing); Corpus linguistics; Test (biology)","score_opus":0.019798328504111392,"score_gpt":0.28076107746267964,"score_spread":0.26096274895856825,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7096309851","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004237107,0.02706111,0.13073874,0.051853128,0.014722508,0.0009764562,0.051432017,0.034254957,0.68472403],"genre_scores_gemma":[0.07416094,0.017045349,0.18739434,0.008951301,0.0047480757,0.0024936951,0.056542583,0.015691,0.6329727],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9929924,0.0014730403,0.00069340976,0.0028393217,0.0015503649,0.00045152195],"domain_scores_gemma":[0.98141265,0.0060886457,0.0010150046,0.0050608185,0.004674459,0.0017483969],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.004902574,0.0018256865,0.003426854,0.0037824684,0.002468557,0.008988074,0.0033963814,0.0029901369,0.4330736],"category_scores_gemma":[0.025260184,0.0011590315,0.0015094739,0.00462538,0.002187841,0.009439968,0.0071058087,0.005307662,0.34921756],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004060851,0.000068478555,0.0013699055,0.0019287395,0.0000896178,0.00038414766,0.00061664206,0.00038348493,0.0019417934,0.1626043,0.4644366,0.36577016],"study_design_scores_gemma":[0.00013789136,0.00004055518,0.0010880042,0.00048221202,0.000037407135,0.00030851152,0.00017374588,0.0013519065,0.00081175595,0.05043794,0.94507813,0.000052028856],"about_ca_topic_score_codex":0.0045435466,"about_ca_topic_score_gemma":0.0034370094,"teacher_disagreement_score":0.56692636,"about_ca_system_score_codex":0.002296325,"about_ca_system_score_gemma":0.0044636205,"threshold_uncertainty_score":0.8086517},"labels":[],"label_agreement":null},{"id":"W7096530871","doi":"","title":"Lightweight Morphology: A Methodology for Improving Text Search","year":2004,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Table (database); Grammar; Query language; Natural language; Rule-based machine translation; Recall","score_opus":0.05615558254751257,"score_gpt":0.34956515865157795,"score_spread":0.2934095761040654,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7096530871","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0062566055,0.00035414606,0.9835306,0.00019618103,0.000040274488,0.00023058355,0.000290346,0.0070445775,0.0020567616],"genre_scores_gemma":[0.047203686,0.00035069723,0.9469943,0.00018831251,0.000061916384,0.00023239547,0.00094426074,0.0012126184,0.0028118272],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99430746,0.0016316025,0.0007327501,0.0008424396,0.0022509694,0.000234785],"domain_scores_gemma":[0.9896303,0.004951859,0.0007824957,0.0030301197,0.0014424926,0.000162745],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035572215,0.0012744616,0.0013820475,0.005093807,0.0013075598,0.0029193177,0.0028498296,0.0015299974,0.0066353716],"category_scores_gemma":[0.017545179,0.0010703927,0.0024179572,0.005078878,0.0017434177,0.0076912567,0.0038095703,0.001524422,0.0043386254],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023461701,0.00018603698,0.0025995495,0.0008669136,0.00019342093,0.00049687584,0.0013498869,0.015463341,0.042482994,0.048836425,0.01573858,0.8715514],"study_design_scores_gemma":[0.00026244565,0.00080961315,0.0036401837,0.0003443068,0.00043416835,0.0038414851,0.0011034706,0.47058722,0.107619055,0.22316062,0.18784021,0.00035727036],"about_ca_topic_score_codex":0.0029002978,"about_ca_topic_score_gemma":0.0047612265,"teacher_disagreement_score":0.0066353716,"about_ca_system_score_codex":0.0008944793,"about_ca_system_score_gemma":0.001832532,"threshold_uncertainty_score":0.022197545},"labels":[],"label_agreement":null},{"id":"W7096629773","doi":"","title":"TransType2 — an innovative computerassisted translation system","year":2004,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"German; Translation (biology); Machine translation; Work (physics)","score_opus":0.02147469112172053,"score_gpt":0.2751014486514363,"score_spread":0.2536267575297158,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7096629773","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024973197,0.0014954433,0.738065,0.00083306583,0.0018088473,0.0009641933,0.009324099,0.17355774,0.048978318],"genre_scores_gemma":[0.113469504,0.0010661248,0.7538622,0.0011579243,0.0006828318,0.0009971128,0.03187642,0.014409888,0.082477935],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99765646,0.00096235395,0.0001944766,0.00048320286,0.00060571963,0.00009780398],"domain_scores_gemma":[0.99740726,0.0006923152,0.00013694835,0.0007249433,0.0008664503,0.00017206125],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020793946,0.0012323774,0.0008152665,0.0015667686,0.0010138439,0.0020722107,0.0017736717,0.0012788025,0.029242465],"category_scores_gemma":[0.0042828037,0.00060744514,0.0007540563,0.0015088344,0.0006788504,0.00271062,0.002535456,0.0015029866,0.018382104],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009529687,0.00026021688,0.0012926343,0.0009820629,0.000169072,0.00092571037,0.0013445206,0.0016291833,0.123721525,0.0112863,0.20201659,0.6554193],"study_design_scores_gemma":[0.00035260848,0.0007381422,0.002662544,0.00013590611,0.0001616604,0.003607699,0.00045879625,0.032027557,0.08172767,0.008018547,0.8697902,0.0003186409],"about_ca_topic_score_codex":0.0019808966,"about_ca_topic_score_gemma":0.0030334797,"teacher_disagreement_score":0.029242465,"about_ca_system_score_codex":0.00050042267,"about_ca_system_score_gemma":0.0016114638,"threshold_uncertainty_score":0.097825766},"labels":[],"label_agreement":null},{"id":"W7096651338","doi":"","title":"Association for Computational Linguistics. Extensions to HMM-based Statistical Word Alignment Models","year":2013,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Bigram; Word (group theory); Viterbi algorithm; Hidden Markov model; Statistical model; Association (psychology); Error detection and correction","score_opus":0.020282637737832194,"score_gpt":0.2955953691796271,"score_spread":0.2753127314417949,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7096651338","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0014839297,0.02962903,0.73608005,0.018988088,0.00848555,0.00065749837,0.043410197,0.041861393,0.11940421],"genre_scores_gemma":[0.042766936,0.031867046,0.6862435,0.0060009365,0.0042904345,0.0044693965,0.064043336,0.012774771,0.14754368],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9952018,0.0014100041,0.00078768196,0.0013354776,0.0011066034,0.00015836774],"domain_scores_gemma":[0.9771518,0.013032286,0.0014525891,0.0049750395,0.002808968,0.00057933974],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0054701334,0.0025293531,0.002398315,0.0056675724,0.0013935746,0.005187121,0.0034218095,0.004042627,0.20376039],"category_scores_gemma":[0.026276747,0.0026206283,0.0018497364,0.011413217,0.0021161595,0.011681025,0.0048175147,0.006397101,0.17100942],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010030433,0.000081916456,0.0016924492,0.002476977,0.0002217247,0.00036219406,0.0007168219,0.0026220155,0.001260818,0.09779133,0.44259524,0.4500783],"study_design_scores_gemma":[0.000078829966,0.000036265155,0.0009639829,0.0013633067,0.0000970132,0.00048352202,0.0001556435,0.015059334,0.0005842468,0.14214855,0.83896637,0.0000629411],"about_ca_topic_score_codex":0.006663295,"about_ca_topic_score_gemma":0.0059825387,"teacher_disagreement_score":0.20376039,"about_ca_system_score_codex":0.0014567933,"about_ca_system_score_gemma":0.004615783,"threshold_uncertainty_score":0.6816464},"labels":[],"label_agreement":null},{"id":"W7096728578","doi":"","title":"NAACL HLT 2010 Computational Linguistics in a World of Social Media","year":2010,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Marketing buzz; Social media; Multitude; Quarter (Canadian coin); EPIC; Range (aeronautics)","score_opus":0.012795225145494662,"score_gpt":0.28558543907964934,"score_spread":0.2727902139341547,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7096728578","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011023353,0.04122034,0.49629098,0.15542264,0.02593449,0.0011743464,0.043407023,0.045007005,0.18051979],"genre_scores_gemma":[0.082061015,0.023384232,0.51935416,0.026986117,0.00987798,0.0023373766,0.12506685,0.012817383,0.19811489],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99136037,0.004512386,0.00081624987,0.0014157774,0.001566643,0.00032859863],"domain_scores_gemma":[0.98711944,0.0070729135,0.00030608018,0.0025685478,0.0019267321,0.0010062502],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009544958,0.0018225679,0.0024794575,0.0031905596,0.0046971706,0.014310028,0.0048121084,0.004582868,0.07008141],"category_scores_gemma":[0.020449914,0.002170589,0.0019073262,0.0031026967,0.004060874,0.02094936,0.009922932,0.008313278,0.0331496],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000788331,0.00009392891,0.00041045208,0.000636022,0.0001057709,0.00018229896,0.0006450411,0.0010519534,0.0006977339,0.046984773,0.8596155,0.08949773],"study_design_scores_gemma":[0.00009147613,0.000024638672,0.0005134609,0.00044663996,0.00007437007,0.00041301007,0.00063026714,0.009063072,0.0008642977,0.14057434,0.8472426,0.00006184965],"about_ca_topic_score_codex":0.0143958125,"about_ca_topic_score_gemma":0.036513798,"teacher_disagreement_score":0.07008141,"about_ca_system_score_codex":0.0051178792,"about_ca_system_score_gemma":0.008113147,"threshold_uncertainty_score":0.23444569},"labels":[],"label_agreement":null},{"id":"W7096779754","doi":"","title":"METIS-II: Machine Translation for Low Resource Languages","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Machine translation; Translation (biology); Machine translation software usability; Example-based machine translation; Universal Networking Language; Computer-assisted translation; Transfer-based machine translation; Resource (disambiguation)","score_opus":0.018420309085174235,"score_gpt":0.2790106823640569,"score_spread":0.2605903732788827,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7096779754","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0118384855,0.0007092842,0.9186169,0.0004365583,0.00037944745,0.00051789853,0.0024519959,0.05233428,0.012715245],"genre_scores_gemma":[0.081329286,0.00046380854,0.88612515,0.00029652737,0.00018243694,0.00061620644,0.009101297,0.006182879,0.015702553],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989241,0.00029978307,0.000075386335,0.00025750243,0.0003475638,0.00009563557],"domain_scores_gemma":[0.9990771,0.00031400326,0.00006586463,0.00029131485,0.0002006888,0.000051057203],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011390564,0.0011599163,0.0012252516,0.0012596828,0.00083961827,0.0022599236,0.0017386068,0.0008963469,0.015322512],"category_scores_gemma":[0.0023878203,0.0006331927,0.0007233304,0.0009602522,0.0006535583,0.0027497867,0.0016400637,0.0016936823,0.010144073],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014352844,0.00026125365,0.0011707315,0.001973525,0.00029044313,0.0009767924,0.0008944964,0.011014142,0.1525934,0.09421739,0.0800177,0.6551549],"study_design_scores_gemma":[0.000625287,0.00088500883,0.002060471,0.00034854977,0.00024276001,0.002397049,0.0005546375,0.27530357,0.26120102,0.07764042,0.37853584,0.00020541287],"about_ca_topic_score_codex":0.00068122114,"about_ca_topic_score_gemma":0.0010337076,"teacher_disagreement_score":0.015322512,"about_ca_system_score_codex":0.0005950916,"about_ca_system_score_gemma":0.00094086275,"threshold_uncertainty_score":0.051258862},"labels":[],"label_agreement":null},{"id":"W7096877795","doi":"","title":"Evaluation of a Machine Translation System for Low Resource Languages: METIS-II","year":2015,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Set (abstract data type); Parsing; Test set; Machine translation; Test (biology); Training set; Resource (disambiguation)","score_opus":0.05002207825669083,"score_gpt":0.33487982980323655,"score_spread":0.2848577515465457,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7096877795","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.81331843,0.0035168957,0.068403006,0.0014125457,0.0009921218,0.002113013,0.029554669,0.03215077,0.048538424],"genre_scores_gemma":[0.6490867,0.00095709425,0.16649593,0.0005340916,0.0002649458,0.0012301897,0.15041532,0.0041305623,0.026885195],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99521637,0.0019182883,0.00051074434,0.0007106038,0.0014226654,0.0002213349],"domain_scores_gemma":[0.99612695,0.0013839594,0.00013920678,0.0005619128,0.0015373786,0.0002506434],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036186462,0.0014776222,0.0013889484,0.0015767571,0.0011087402,0.0020009007,0.0015162368,0.0010435688,0.007611526],"category_scores_gemma":[0.007226343,0.00036416677,0.0005900965,0.0015531789,0.00052515796,0.0022029097,0.0014373297,0.0009885503,0.0051268386],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00528789,0.0019785392,0.012357462,0.005783352,0.0009984268,0.0030401202,0.0041705095,0.02971903,0.28107747,0.0071450532,0.115244955,0.53319716],"study_design_scores_gemma":[0.003114641,0.0063314443,0.055758685,0.0004970032,0.00080290425,0.0048540975,0.0050333063,0.15373778,0.48835087,0.0027396674,0.27836525,0.00041436523],"about_ca_topic_score_codex":0.0042655664,"about_ca_topic_score_gemma":0.004813474,"teacher_disagreement_score":0.007611526,"about_ca_system_score_codex":0.0011289776,"about_ca_system_score_gemma":0.0015705199,"threshold_uncertainty_score":0.025463045},"labels":[],"label_agreement":null},{"id":"W7096948591","doi":"","title":"Abstract Estimating Upper and Lower Bounds on the Performance of Word-Sense Disambiguation Programs","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Upper and lower bounds; Context (archaeology); Measure (data warehouse); Baseline (sea); Word-sense disambiguation","score_opus":0.02063433429826272,"score_gpt":0.2600485199419613,"score_spread":0.2394141856436986,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7096948591","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5646531,0.013702703,0.39081606,0.001787304,0.0002909678,0.00024800736,0.0028386984,0.007277367,0.018385822],"genre_scores_gemma":[0.89430076,0.0008357389,0.097560346,0.0002573112,0.00026639318,0.00031193148,0.0035087091,0.0008902123,0.0020686286],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9542898,0.016045813,0.0028579766,0.010238571,0.013665697,0.0029021485],"domain_scores_gemma":[0.70599633,0.25507396,0.008217007,0.013445407,0.015102748,0.0021645771],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.026897024,0.0024457949,0.0026779238,0.006774405,0.0015793293,0.0055787796,0.0025030773,0.004030267,0.0027354718],"category_scores_gemma":[0.16579822,0.0014810129,0.0012574531,0.0037651008,0.0029198148,0.0073057553,0.0041987114,0.003045719,0.0018446601],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.007814916,0.00089869765,0.061236784,0.0020434903,0.0012950993,0.0006917863,0.0017291103,0.43914512,0.045613155,0.022161478,0.01578397,0.40158638],"study_design_scores_gemma":[0.00008381767,0.00081709074,0.025430331,0.00022606937,0.000181834,0.00033702582,0.0004666727,0.8978152,0.051997144,0.019625837,0.0028298679,0.00018908754],"about_ca_topic_score_codex":0.0079753725,"about_ca_topic_score_gemma":0.0050687823,"teacher_disagreement_score":0.026897024,"about_ca_system_score_codex":0.0034785268,"about_ca_system_score_gemma":0.0015597533,"threshold_uncertainty_score":0.14224678},"labels":[],"label_agreement":null},{"id":"W7097229270","doi":"","title":"GOFAIsum: a symbolic summarizer for DUC","year":2007,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Relevance (law); Task (project management); XSLT; NIST; Automatic summarization; XML; Pruning; Rule-based machine translation","score_opus":0.013085236327472351,"score_gpt":0.2971260345383019,"score_spread":0.28404079821082956,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7097229270","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03847934,0.0033136958,0.5644138,0.00063437346,0.00047099,0.00058176747,0.0142720835,0.36028674,0.017547242],"genre_scores_gemma":[0.21291253,0.0013968911,0.6926486,0.00047419997,0.00041358487,0.0007921108,0.04836608,0.00913994,0.033856034],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999602,0.00008561062,0.000043745316,0.000096035285,0.00014549795,0.000027145872],"domain_scores_gemma":[0.9991456,0.0002783061,0.00008718941,0.00018143315,0.0002435531,0.00006379801],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000581722,0.00090959226,0.00068432005,0.001567746,0.0005698169,0.00113963,0.0009991451,0.0006370825,0.008429414],"category_scores_gemma":[0.002379407,0.0002504937,0.00037513537,0.0008949984,0.00019761774,0.0012881954,0.0007663479,0.000684685,0.004005647],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000732889,0.000118286116,0.001534439,0.00094983366,0.00014873463,0.0005300665,0.00091132906,0.0051334985,0.058994032,0.004647355,0.14842953,0.77787006],"study_design_scores_gemma":[0.00036819818,0.0010046016,0.009175703,0.00022941292,0.00035509738,0.0015259107,0.0006722159,0.16987354,0.1209283,0.008829271,0.6867375,0.00030030086],"about_ca_topic_score_codex":0.0044279406,"about_ca_topic_score_gemma":0.008539514,"teacher_disagreement_score":0.008429414,"about_ca_system_score_codex":0.0006121385,"about_ca_system_score_gemma":0.0005928015,"threshold_uncertainty_score":0.028199196},"labels":[],"label_agreement":null},{"id":"W7097308886","doi":"","title":"c○2011 The Association for Computational Linguistics","year":2011,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Session (web analytics); Honor; Pleasure; Process (computing); Dozen; Conjunction (astronomy); Association (psychology)","score_opus":0.029053754543969957,"score_gpt":0.2776906280050725,"score_spread":0.24863687346110253,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7097308886","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0031942313,0.029857606,0.08102838,0.08809863,0.12698627,0.0008662198,0.033435345,0.014547838,0.6219855],"genre_scores_gemma":[0.021730268,0.023803653,0.048692413,0.020599572,0.017305214,0.0014755395,0.066094965,0.014026229,0.7862722],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9957573,0.0011583555,0.000543785,0.0010749361,0.0011640119,0.00030165017],"domain_scores_gemma":[0.9866823,0.0026514106,0.00066334713,0.0030748793,0.0058771516,0.0010508721],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.0064116837,0.0011308979,0.0014950173,0.003964271,0.002661434,0.011783905,0.0023319123,0.0027558135,0.33870807],"category_scores_gemma":[0.01737118,0.000994259,0.0012496999,0.0042927936,0.0021772264,0.0057879323,0.0046054213,0.0051040803,0.36652806],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000043559194,0.000021691449,0.00038190896,0.0003076355,0.000021179874,0.000055575238,0.00010150064,0.00009188614,0.0005074866,0.00961021,0.8980625,0.09079492],"study_design_scores_gemma":[0.0000073688657,0.0000048352204,0.0002912944,0.000231804,0.0000067219366,0.000057556314,0.000067651636,0.0002176775,0.00017614936,0.0037583653,0.99517196,0.000008684327],"about_ca_topic_score_codex":0.0067630983,"about_ca_topic_score_gemma":0.004373819,"teacher_disagreement_score":0.66129196,"about_ca_system_score_codex":0.002184436,"about_ca_system_score_gemma":0.004820716,"threshold_uncertainty_score":0.9432527},"labels":[],"label_agreement":null},{"id":"W7097335046","doi":"","title":"Identification of Idioms by Machine Translation:","year":2015,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Identification (biology); Machine translation; Translation (biology); Simple (philosophy); Power (physics); Statistical analysis","score_opus":0.02187306866986977,"score_gpt":0.28688427447484627,"score_spread":0.2650112058049765,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7097335046","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.78685504,0.004831422,0.14956321,0.0008420148,0.00046675102,0.0005757766,0.0025947683,0.0031536466,0.051117416],"genre_scores_gemma":[0.81089795,0.00066754466,0.17927738,0.00020029511,0.0001024975,0.00022928824,0.0045736623,0.00032082337,0.0037306722],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9948605,0.0025442638,0.0004492853,0.0006849106,0.001284016,0.00017700865],"domain_scores_gemma":[0.9863179,0.009089323,0.0008535945,0.0018972368,0.0017645663,0.00007737488],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037031039,0.0009184697,0.00059645326,0.0020555896,0.0007013709,0.001628545,0.0012569352,0.0011236266,0.006466157],"category_scores_gemma":[0.016707962,0.0001930272,0.0005019875,0.0032646304,0.0005660078,0.0024036774,0.001186185,0.0005635488,0.002331951],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0050457995,0.0008791406,0.024887586,0.0029954277,0.0006616566,0.00086975406,0.0017659976,0.030486606,0.048338532,0.019998677,0.01235604,0.8517148],"study_design_scores_gemma":[0.00096459064,0.0034214458,0.05742334,0.000387533,0.00089912047,0.0047040577,0.00322344,0.49772304,0.33574012,0.026267322,0.069035694,0.00021031057],"about_ca_topic_score_codex":0.0016038205,"about_ca_topic_score_gemma":0.0023093587,"teacher_disagreement_score":0.006466157,"about_ca_system_score_codex":0.0007810925,"about_ca_system_score_gemma":0.00046242366,"threshold_uncertainty_score":0.02163142},"labels":[],"label_agreement":null},{"id":"W7097340999","doi":"","title":"Real-word spelling correction with trigrams: A reconsideration of the Mays, Damerau, and Mercer model. http://ftp.cs.toronto.edu/ pub/gh/WilcoxOHearn-etal-2006.pdf","year":2006,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Trigram; Spelling; Variation (astronomy); Natural language","score_opus":0.008762943875778636,"score_gpt":0.23325686502902337,"score_spread":0.22449392115324474,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7097340999","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01670642,0.0013945551,0.97139794,0.0010230835,0.0003290132,0.00008931559,0.00036003056,0.0060283775,0.0026712],"genre_scores_gemma":[0.34863046,0.0016584493,0.6354792,0.0005856405,0.00029409837,0.00014390262,0.0009450374,0.0017583907,0.010504793],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.998235,0.00073481107,0.00010127476,0.00030018916,0.0005433619,0.00008533454],"domain_scores_gemma":[0.9919396,0.0040408745,0.00039143427,0.0021378326,0.0013139739,0.0001763399],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037110539,0.0010455601,0.0012162692,0.0013798624,0.0006874465,0.001851061,0.0027888983,0.0011148348,0.003399805],"category_scores_gemma":[0.016216476,0.0005791305,0.00070175435,0.0017247219,0.00077882415,0.004138458,0.0012216031,0.0019410556,0.0037234717],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00063539343,0.00021128962,0.0069903415,0.00038140407,0.00027839924,0.0003865471,0.00066187483,0.058166355,0.012835852,0.019866416,0.01955789,0.8800282],"study_design_scores_gemma":[0.00006122917,0.00020163516,0.0024342833,0.000079080535,0.00013173436,0.00056161726,0.00019286771,0.9313542,0.024058912,0.027946739,0.012872804,0.00010486715],"about_ca_topic_score_codex":0.006552347,"about_ca_topic_score_gemma":0.015994055,"teacher_disagreement_score":0.006552347,"about_ca_system_score_codex":0.00068185694,"about_ca_system_score_gemma":0.0018021069,"threshold_uncertainty_score":0.0196262},"labels":[],"label_agreement":null},{"id":"W7097934990","doi":"","title":"N-gram-based Techniques","year":2006,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Work (physics); Identification (biology); Focus (optics); Perspective (graphical); Agency (philosophy)","score_opus":0.005815851802406953,"score_gpt":0.2439080820867369,"score_spread":0.23809223028432994,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7097934990","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008963638,0.007711609,0.93137306,0.0013818389,0.001404949,0.00028849323,0.0046528005,0.02125207,0.022971528],"genre_scores_gemma":[0.13100268,0.0051903455,0.7829927,0.0010251834,0.0012672945,0.0002533888,0.017636674,0.0027243786,0.057907254],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9973687,0.00076787913,0.00021313727,0.0005581027,0.0008911849,0.00020087078],"domain_scores_gemma":[0.9969543,0.00088573154,0.0001951422,0.0008483522,0.0010196011,0.00009690616],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012579311,0.0020819174,0.0015887755,0.002869652,0.0019195005,0.0023608913,0.0019029931,0.0019232194,0.023969095],"category_scores_gemma":[0.0056213955,0.00063903484,0.0011469371,0.003880823,0.00075640867,0.0039991136,0.0020967107,0.0021798024,0.03542176],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003144659,0.00014657219,0.0009534601,0.0006066361,0.00015095198,0.0002896602,0.00015648037,0.005623873,0.032748826,0.017566767,0.07886861,0.8625736],"study_design_scores_gemma":[0.00009669481,0.00032574177,0.0034187797,0.00048733523,0.00037198965,0.0020246531,0.00037609425,0.38023543,0.1287247,0.10994257,0.3737258,0.0002702295],"about_ca_topic_score_codex":0.0054004393,"about_ca_topic_score_gemma":0.01565508,"teacher_disagreement_score":0.023969095,"about_ca_system_score_codex":0.00065897434,"about_ca_system_score_gemma":0.0017660981,"threshold_uncertainty_score":0.08018458},"labels":[],"label_agreement":null},{"id":"W7097976155","doi":"","title":"NLP Technologies Inc.","year":2013,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Machine translation; Government (linguistics); Translation (biology); Quality (philosophy); Order (exchange)","score_opus":0.007837151832003588,"score_gpt":0.23890470955432677,"score_spread":0.23106755772232318,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7097976155","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0021608805,0.012050379,0.23852645,0.0095735565,0.0030581427,0.00126669,0.14569716,0.13386714,0.45379964],"genre_scores_gemma":[0.018759234,0.008699202,0.33920866,0.0040042438,0.000729154,0.0023953582,0.21372434,0.018028915,0.39445093],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99660444,0.0006288073,0.00042031257,0.0010304118,0.0011641195,0.00015188316],"domain_scores_gemma":[0.9948137,0.0014230239,0.00049611577,0.0013744584,0.0015101348,0.00038262198],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.0032736978,0.0019750136,0.0018736887,0.0050786394,0.0017421196,0.0065184715,0.0020317256,0.0020916576,0.34052446],"category_scores_gemma":[0.010474749,0.0011115278,0.0010075796,0.005913056,0.001061845,0.005930749,0.0042736162,0.0030988834,0.40278804],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020399678,0.00005432617,0.0005540318,0.0016471631,0.00005616137,0.00034648128,0.00033221845,0.00047537725,0.005033843,0.021938436,0.5373086,0.4320493],"study_design_scores_gemma":[0.00005844003,0.000020381949,0.00043362292,0.00033267637,0.000025253292,0.00032889485,0.000117883916,0.0019333422,0.001981412,0.021730255,0.9730143,0.00002366703],"about_ca_topic_score_codex":0.0048194625,"about_ca_topic_score_gemma":0.005399684,"teacher_disagreement_score":0.65947556,"about_ca_system_score_codex":0.0013163083,"about_ca_system_score_gemma":0.0031272185,"threshold_uncertainty_score":0.9406618},"labels":[],"label_agreement":null},{"id":"W7098101795","doi":"","title":"Responsible Professor:","year":2012,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Sentence; Semantics (computer science); Embedding; Key (lock); Machine translation; Architecture; Artificial neural network","score_opus":0.020601389776694914,"score_gpt":0.3084356682947958,"score_spread":0.2878342785181009,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7098101795","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.003852878,0.009709761,0.0052696913,0.19854024,0.16526215,0.0010692426,0.014415909,0.004060844,0.59781927],"genre_scores_gemma":[0.009865522,0.0019731205,0.0011387571,0.016024996,0.007308076,0.00027243892,0.0019466424,0.00058705296,0.9608834],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99752265,0.0003165872,0.00015970613,0.0008237409,0.0007503533,0.00042698064],"domain_scores_gemma":[0.9918424,0.0006221122,0.00031826182,0.00052097783,0.0033111435,0.0033851247],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.002256054,0.0008083347,0.001023681,0.0013361725,0.002579931,0.0049482863,0.0018476782,0.004073013,0.722992],"category_scores_gemma":[0.011256156,0.00038873145,0.0005895959,0.0010542208,0.00076536194,0.0028039988,0.0032898164,0.0032794338,0.47661418],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000074783144,0.000020660766,0.0003453897,0.00011667986,0.0000037431494,0.00013953076,0.00008036868,0.000025159683,0.00014977847,0.0035298439,0.95962787,0.035886142],"study_design_scores_gemma":[0.000011770134,0.00001673885,0.00031235433,0.00007290425,0.0000028112013,0.0001692947,0.000083428014,0.000038522034,0.00010579572,0.0006302205,0.9985513,0.000004869472],"about_ca_topic_score_codex":0.0014214591,"about_ca_topic_score_gemma":0.0021313415,"teacher_disagreement_score":0.277008,"about_ca_system_score_codex":0.0024055417,"about_ca_system_score_gemma":0.004994691,"threshold_uncertainty_score":0.39511824},"labels":[],"label_agreement":null},{"id":"W7098977503","doi":"","title":"PUB TYPE Reference Materials Directories/Catalogs (132) Reports Research/Technical (143)","year":2016,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Local history; Type (biology); Component (thermodynamics); Local government; Data collection","score_opus":0.06885147137577356,"score_gpt":0.3678471748161214,"score_spread":0.2989957034403478,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7098977503","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0006603495,0.0011642036,0.001968569,0.0010002448,0.00076659286,0.00094384624,0.05505996,0.004587725,0.9338485],"genre_scores_gemma":[0.0015755404,0.002482955,0.0027324597,0.00035835986,0.00022838142,0.00024178658,0.02326123,0.0016751903,0.96744406],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9987159,0.000057430338,0.00008039046,0.000094291936,0.0008889172,0.00016318994],"domain_scores_gemma":[0.99219453,0.00056868437,0.00033091602,0.0009025029,0.0050631864,0.00094020367],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.001129981,0.0012779721,0.0011452907,0.00962243,0.0030051,0.005392902,0.0015005199,0.0010807073,0.73526806],"category_scores_gemma":[0.0056696692,0.0012156606,0.0005041927,0.01844348,0.001046797,0.003184923,0.0011584935,0.0010460184,0.67150027],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000010795454,0.000021643056,0.00032555006,0.00019954011,0.0000012148407,0.000028288825,0.00010693832,0.00005236638,0.00025582075,0.0013240511,0.9356746,0.061999217],"study_design_scores_gemma":[0.0000029032108,0.000004438942,0.0008130265,0.000057865713,0.0000011496113,0.00001739206,0.00006387394,0.000017368142,0.000054990196,0.000103305254,0.998858,0.0000057010625],"about_ca_topic_score_codex":0.32203612,"about_ca_topic_score_gemma":0.4599651,"teacher_disagreement_score":0.32203612,"about_ca_system_score_codex":0.007883204,"about_ca_system_score_gemma":0.014412771,"threshold_uncertainty_score":0.6403233},"labels":[],"label_agreement":null},{"id":"W7100079131","doi":"","title":"Example-based Translation without Parallel Corpora: First experiments on a prototype","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Noun phrase; Translation (biology); Machine translation; Sentence; Noun; Metis; Conjunction (astronomy); Line (geometry)","score_opus":0.07750123563541172,"score_gpt":0.29881958974768513,"score_spread":0.2213183541122734,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7100079131","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8153194,0.001737263,0.12307866,0.001108425,0.0005309397,0.0037473815,0.006050234,0.030585052,0.017842589],"genre_scores_gemma":[0.57227945,0.0009356626,0.39333904,0.0005873243,0.0001467559,0.0018042731,0.016585944,0.004289033,0.010032544],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9954881,0.002349897,0.0005597818,0.00072330143,0.0007206413,0.00015820826],"domain_scores_gemma":[0.98141736,0.012790514,0.00037240662,0.0028864262,0.002058483,0.0004747541],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0040921643,0.00150109,0.0016571861,0.0007710474,0.0012126416,0.0017927496,0.0040250807,0.0032460068,0.015154865],"category_scores_gemma":[0.021229992,0.0010437677,0.00068525196,0.0027422742,0.001171095,0.005974055,0.0027425964,0.0018959272,0.006377207],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.013921185,0.012357774,0.0050391406,0.015051981,0.0013595232,0.006703305,0.015886497,0.03924719,0.18036512,0.0067634094,0.07509375,0.62821114],"study_design_scores_gemma":[0.020767113,0.019158501,0.02378614,0.00084901456,0.002048597,0.00813203,0.011436999,0.3943346,0.35802132,0.012942656,0.14763507,0.00088787614],"about_ca_topic_score_codex":0.004246865,"about_ca_topic_score_gemma":0.004043061,"teacher_disagreement_score":0.015154865,"about_ca_system_score_codex":0.00066158397,"about_ca_system_score_gemma":0.0009250397,"threshold_uncertainty_score":0.0506981},"labels":[],"label_agreement":null},{"id":"W7100268972","doi":"","title":"(1)","year":2015,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Automatic summarization; Sentence; Set (abstract data type); Compression (physics); Point (geometry); Position (finance)","score_opus":0.0286147362486767,"score_gpt":0.2891211610510687,"score_spread":0.260506424802392,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7100268972","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007591545,0.0030295902,0.018604688,0.005870326,0.002177089,0.00033865968,0.019188343,0.003492795,0.939707],"genre_scores_gemma":[0.024499964,0.0011110911,0.006587129,0.00073363667,0.000237138,0.00007022217,0.0062440136,0.0008086575,0.9597082],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99920267,0.00008744287,0.000020846499,0.0002459345,0.00031366624,0.00012954535],"domain_scores_gemma":[0.99892575,0.00010163298,0.000055897155,0.0000889772,0.0006682222,0.00015963592],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0006615922,0.00108282,0.0006699178,0.001887263,0.0024956248,0.0026754173,0.0013846251,0.001533409,0.63077825],"category_scores_gemma":[0.0017884286,0.00027252047,0.00041433884,0.0022493533,0.00076105347,0.0012830292,0.0010060818,0.00097233726,0.39578554],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014300902,0.000037347458,0.002596795,0.00031637747,0.00002106816,0.00046153186,0.00021615683,0.0007385893,0.004155983,0.02488125,0.83052176,0.13591011],"study_design_scores_gemma":[0.000019494393,0.000020970192,0.0023136304,0.00008953252,0.000008688965,0.0002980521,0.00015599611,0.0012693993,0.0011815667,0.0021771651,0.9924385,0.000027020305],"about_ca_topic_score_codex":0.24265265,"about_ca_topic_score_gemma":0.37467054,"teacher_disagreement_score":0.36922175,"about_ca_system_score_codex":0.00519136,"about_ca_system_score_gemma":0.0040611983,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W7100328590","doi":"","title":"The Use of Approximate String Matching Techniques in the Alignment of Sentences in Parallel Corpora Abstract","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Lexicon; Task (project management); Parallel corpora; Terminology; Matching (statistics); String (physics); Similarity (geometry)","score_opus":0.055867522205732385,"score_gpt":0.27857689311603434,"score_spread":0.22270937091030196,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7100328590","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04344424,0.00076438027,0.9496068,0.00029564038,0.00010539694,0.00017681396,0.00048011297,0.0018410448,0.0032856248],"genre_scores_gemma":[0.14708966,0.00045057014,0.8490657,0.00008863355,0.00009548923,0.00015759871,0.0014822024,0.00030223693,0.0012679213],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9943699,0.0029682193,0.0005550165,0.0009341106,0.0010107114,0.0001621467],"domain_scores_gemma":[0.98852843,0.0071626203,0.0010073866,0.001724171,0.0014646783,0.00011263],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00404757,0.00055511104,0.0009082763,0.0044249212,0.0013143766,0.0026397717,0.0015880428,0.0012776989,0.0035525516],"category_scores_gemma":[0.02763077,0.000557732,0.00054222345,0.00965272,0.0013107792,0.0054650614,0.0021412962,0.0010813031,0.002906796],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006672682,0.00017688857,0.0048126723,0.00066817045,0.00014860423,0.00080864254,0.0017351698,0.03414351,0.036822774,0.04920599,0.006245851,0.8645645],"study_design_scores_gemma":[0.0001252566,0.00044117597,0.0070841117,0.00016303388,0.00020344884,0.0016648339,0.001737778,0.7700178,0.062037006,0.1185514,0.03786076,0.000113449925],"about_ca_topic_score_codex":0.0018758631,"about_ca_topic_score_gemma":0.0023267728,"teacher_disagreement_score":0.0044249212,"about_ca_system_score_codex":0.0006972223,"about_ca_system_score_gemma":0.0011927282,"threshold_uncertainty_score":0.021405876},"labels":[],"label_agreement":null},{"id":"W7100839479","doi":"","title":"Introduction to","year":2009,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Summit; Machine translation; Government (linguistics); State (computer science); Web site","score_opus":0.006196393684257976,"score_gpt":0.26103583213801973,"score_spread":0.2548394384537618,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7100839479","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0007649549,0.008401065,0.013909132,0.018368974,0.02232508,0.00040541997,0.010665387,0.00418749,0.9209725],"genre_scores_gemma":[0.006239736,0.006845765,0.007849153,0.008067396,0.0034435787,0.00041457842,0.010371268,0.0016079934,0.95516056],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99810284,0.00028871087,0.00013742338,0.00043162817,0.00084786065,0.00019159833],"domain_scores_gemma":[0.9967783,0.00047950709,0.000118145224,0.00056532054,0.0016054554,0.00045328407],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.001742423,0.0010758528,0.00078715006,0.0020887137,0.0019013092,0.0072275447,0.0018713991,0.0024733418,0.56493443],"category_scores_gemma":[0.007921594,0.00035580274,0.0007248199,0.002729827,0.000968392,0.0054628327,0.0035904727,0.0026413037,0.51504004],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000028704733,0.000026824273,0.00016944413,0.0001529249,0.0000034569762,0.000040712726,0.00014190486,0.00005593912,0.00023997946,0.017514436,0.86001575,0.12160994],"study_design_scores_gemma":[0.0000022181546,0.0000065015447,0.00012705047,0.00006359388,0.000001138429,0.000039127008,0.000050559505,0.00002118431,0.000049415856,0.0021252048,0.99751043,0.0000036053134],"about_ca_topic_score_codex":0.0038016394,"about_ca_topic_score_gemma":0.0045945747,"teacher_disagreement_score":0.43506557,"about_ca_system_score_codex":0.0018431067,"about_ca_system_score_gemma":0.0031091769,"threshold_uncertainty_score":0.6205682},"labels":[],"label_agreement":null},{"id":"W7101075978","doi":"","title":"METIS: STATISTICAL MACHINE TRANSLATION USING MONOLINGUAL CORPORA","year":2014,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Machine translation; Statistical analysis; Translation (biology); Feature (linguistics); Interpretation (philosophy)","score_opus":0.02802617749279907,"score_gpt":0.30206492261902373,"score_spread":0.27403874512622467,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7101075978","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011281494,0.0022177473,0.8074025,0.002176211,0.0026921844,0.00077573146,0.027348248,0.09817279,0.047933068],"genre_scores_gemma":[0.11643178,0.0023973377,0.7032591,0.00077155733,0.0011090543,0.0019460631,0.10172482,0.022906937,0.04945342],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9946571,0.0024850185,0.00050633226,0.0011669333,0.0009816738,0.00020297813],"domain_scores_gemma":[0.9948144,0.0021655734,0.00027817147,0.0014934919,0.001068933,0.00017937644],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035125797,0.0021449786,0.0013545196,0.0053384886,0.0016957105,0.0043687075,0.0020193646,0.00140109,0.052047424],"category_scores_gemma":[0.016363675,0.0010926313,0.0016258662,0.0076586087,0.0012785845,0.004623675,0.0052249166,0.0028760678,0.03347779],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010478161,0.000359621,0.0010382485,0.0035064386,0.0006351369,0.0013903666,0.0014507432,0.014586967,0.020522233,0.082540885,0.24875455,0.6241671],"study_design_scores_gemma":[0.00043574232,0.00036020658,0.0018527308,0.00061528943,0.00018242373,0.00093385216,0.0011552862,0.14301084,0.042769898,0.15718529,0.65129673,0.00020171794],"about_ca_topic_score_codex":0.0031760687,"about_ca_topic_score_gemma":0.0041069444,"teacher_disagreement_score":0.052047424,"about_ca_system_score_codex":0.0015287219,"about_ca_system_score_gemma":0.0032654419,"threshold_uncertainty_score":0.17411602},"labels":[],"label_agreement":null},{"id":"W7101209141","doi":"","title":"A Modular Architecture for Separating Hypothesis Formation from Hypothesis Evaluation in Data-driven Machine Translation","year":2007,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Machine translation; Probabilistic logic; Modular design; Translation (biology); Selection (genetic algorithm); Graph; Example-based machine translation","score_opus":0.07825084653497913,"score_gpt":0.32604412087177426,"score_spread":0.24779327433679513,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7101209141","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0011351018,0.000104133105,0.9937125,0.0001745983,0.000025786749,0.00012959735,0.000057268266,0.004174584,0.00048638374],"genre_scores_gemma":[0.045708667,0.00016774537,0.95115805,0.0002640483,0.000102595666,0.0005008998,0.0004138669,0.00041366662,0.0012704408],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9951891,0.0018263549,0.0005499596,0.0010504521,0.001158494,0.00022557925],"domain_scores_gemma":[0.9851099,0.008435232,0.0006015985,0.0038072728,0.0016482647,0.00039769834],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010463501,0.001498665,0.0019085421,0.0027027938,0.0012590806,0.004386481,0.0046826093,0.0026850752,0.007406834],"category_scores_gemma":[0.020956833,0.0021578728,0.0018065966,0.0024255002,0.0036328442,0.006902237,0.005673961,0.003694134,0.005473081],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009572534,0.00050576025,0.0023098118,0.0008888627,0.00050920737,0.0005081549,0.0012177071,0.042967472,0.042303402,0.086028956,0.010610223,0.8111931],"study_design_scores_gemma":[0.00021252883,0.00040233732,0.0013166078,0.00016998497,0.0003456365,0.0005657877,0.00016188799,0.69755584,0.042317268,0.23702775,0.019714635,0.00020967273],"about_ca_topic_score_codex":0.0026149398,"about_ca_topic_score_gemma":0.003456646,"teacher_disagreement_score":0.010463501,"about_ca_system_score_codex":0.0011138316,"about_ca_system_score_gemma":0.0025343325,"threshold_uncertainty_score":0.055336952},"labels":[],"label_agreement":null},{"id":"W7101406955","doi":"10.1016/j.jeca.2025.e00440","title":"Does female participation improve firm value? Board gender diversity reform and asymmetric market responses","year":2025,"lang":"en","type":"article","venue":"The Journal of Economic Asymmetries","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"University of Scranton","keywords":"Gender diversity; Diversity (politics); Gender equality; Gender discrimination; Gender inequality; Inequality","score_opus":0.016945384311557253,"score_gpt":0.27980131920589707,"score_spread":0.26285593489433984,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7101406955","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9923384,0.00011548865,0.0006034753,0.00040035264,0.000019141826,0.00003738959,0.00007576163,0.0000076176866,0.006402253],"genre_scores_gemma":[0.9985927,0.000026855127,0.00013270546,0.000088849854,0.00001704725,0.00001688569,0.000018308245,0.0000018693919,0.0011047204],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9983165,0.0008637389,0.00005635535,0.00018919402,0.00025512915,0.00031900153],"domain_scores_gemma":[0.98792404,0.005391976,0.0046871924,0.0008385548,0.00028062097,0.00087763584],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003117601,0.00018284317,0.00037499756,0.0002487098,0.00033637698,0.0011629238,0.000264801,0.0008529998,0.012649235],"category_scores_gemma":[0.010409188,0.0000992832,0.00021364052,0.0002296111,0.0008002162,0.00066671567,0.00070179213,0.00054502673,0.00068015297],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.021162711,0.016681572,0.56197536,0.00082728826,0.0004875432,0.0013790999,0.0040391646,0.011151052,0.1361326,0.034738753,0.0052029463,0.20622192],"study_design_scores_gemma":[0.0008477516,0.01415289,0.9145431,0.00007391827,0.00022427349,0.00020881336,0.003763463,0.0062813973,0.025666494,0.021070689,0.013100217,0.000067056695],"about_ca_topic_score_codex":0.00066181883,"about_ca_topic_score_gemma":0.00069592154,"teacher_disagreement_score":0.012649235,"about_ca_system_score_codex":0.00059056125,"about_ca_system_score_gemma":0.0004925719,"threshold_uncertainty_score":0.0423159},"labels":[],"label_agreement":null},{"id":"W7103883256","doi":"10.18653/v1/2023.gwc-1.33","title":"Correcting Sense Annotations Using Wordnets and Translations","year":2023,"lang":"","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Alberta Machine Intelligence Institute; Natural Sciences and Engineering Research Council of Canada; McKnight Foundation","keywords":"Sequence (biology); Term (time); Sense (electronics); Feature (linguistics)","score_opus":0.049514853097095766,"score_gpt":0.33301930875351615,"score_spread":0.2835044556564204,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7103883256","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07405387,0.0014207199,0.90358806,0.00094374834,0.0016077034,0.00021403094,0.0019938573,0.011042552,0.0051354864],"genre_scores_gemma":[0.2959382,0.0008571905,0.6837722,0.0005213132,0.00034973703,0.00028416747,0.007721527,0.0030149694,0.0075406404],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9914362,0.002156834,0.0010434979,0.0032202061,0.0018491548,0.00029405465],"domain_scores_gemma":[0.9771065,0.00798639,0.0022938559,0.0068052853,0.0054185805,0.00038938707],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006260254,0.0023319665,0.0016134812,0.0055748555,0.0022668939,0.0030628024,0.0019583195,0.001847463,0.004236171],"category_scores_gemma":[0.030274808,0.0011613866,0.0011624935,0.0042318306,0.0021709679,0.0057721217,0.0051445724,0.0025887229,0.003877744],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00097455544,0.00021689145,0.010938413,0.0011784517,0.00041870843,0.001781636,0.0028998454,0.040666908,0.07073503,0.041358575,0.028623613,0.80020744],"study_design_scores_gemma":[0.00017787208,0.0002802503,0.006714189,0.0004485225,0.0003446325,0.0019889302,0.003374783,0.501729,0.14670421,0.2197044,0.11826666,0.00026660337],"about_ca_topic_score_codex":0.0031584874,"about_ca_topic_score_gemma":0.009588124,"teacher_disagreement_score":0.006260254,"about_ca_system_score_codex":0.0009813354,"about_ca_system_score_gemma":0.0030269586,"threshold_uncertainty_score":0.033107817},"labels":[],"label_agreement":null},{"id":"W7104378963","doi":"10.18653/v1/2025.hcinlp-1.15","title":"Time Is Effort: Estimating Human Post-Editing Time for Grammar Error Correction Tool Evaluation","year":2025,"lang":"","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute; University of Waterloo","funders":"Institute for Information and Communications Technology Promotion; Ministry of Science and ICT, South Korea; Government of Canada; Canadian Institute for Advanced Research","keywords":"Error detection and correction; Grammar; Error analysis; Human error","score_opus":0.01819553215468054,"score_gpt":0.3381604627708629,"score_spread":0.31996493061618236,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7104378963","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7706803,0.0040923897,0.15913968,0.00055265945,0.0006778038,0.001341344,0.026598884,0.022291403,0.014625376],"genre_scores_gemma":[0.7958253,0.00058059546,0.1533044,0.00027666043,0.00017235457,0.0014372809,0.040300693,0.0020469378,0.0060558314],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9854462,0.0051130047,0.0016720475,0.0027559046,0.004458794,0.0005541768],"domain_scores_gemma":[0.9140602,0.05179284,0.008500405,0.007554193,0.016000343,0.0020919759],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0075582997,0.0018788314,0.0008315413,0.0058130664,0.00074347615,0.0023675864,0.001483632,0.001360825,0.0023284834],"category_scores_gemma":[0.059929572,0.000365486,0.00077421335,0.0032091541,0.0007094557,0.002549491,0.001644913,0.0010658051,0.002430647],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0030126802,0.0010891878,0.13458477,0.0038157338,0.0007175409,0.0006987074,0.0056093074,0.017859224,0.047416627,0.0019783205,0.066060185,0.71715766],"study_design_scores_gemma":[0.0004882908,0.004331551,0.50083977,0.00068197487,0.0005075736,0.0020180251,0.0054275286,0.2510137,0.10981401,0.007698093,0.11642937,0.0007500808],"about_ca_topic_score_codex":0.0044863545,"about_ca_topic_score_gemma":0.01059103,"teacher_disagreement_score":0.0075582997,"about_ca_system_score_codex":0.0010364595,"about_ca_system_score_gemma":0.0011985798,"threshold_uncertainty_score":0.039972603},"labels":[],"label_agreement":null},{"id":"W7104389014","doi":"10.18653/v1/2025.newsum-main.10","title":"Multi2: Multi-Agent Test-Time Scalable Framework for Multi-Document Processing","year":2025,"lang":"","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Scalability; Scale (ratio); Key (lock); Context (archaeology); Process (computing); Feature (linguistics)","score_opus":0.02569015319742484,"score_gpt":0.3419693545959577,"score_spread":0.3162792013985328,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7104389014","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0024333508,0.0002451406,0.97985625,0.00018592538,0.00007846709,0.0001425905,0.0003455965,0.014972894,0.0017397614],"genre_scores_gemma":[0.18020332,0.00025140372,0.8103097,0.00021954815,0.000120807745,0.000755418,0.0019072592,0.002215242,0.004017208],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9984908,0.00046445444,0.00010316962,0.00031275538,0.0004672529,0.00016149553],"domain_scores_gemma":[0.9975231,0.0010058798,0.00019947159,0.0004950831,0.0005187216,0.00025776395],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002811634,0.0019126039,0.0014528676,0.0013281817,0.00085877214,0.0025765356,0.0042278296,0.0015516654,0.010950975],"category_scores_gemma":[0.007625626,0.000947322,0.0017003187,0.0011165214,0.0009670988,0.0028458526,0.0030095878,0.0023173254,0.0031913377],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00089315255,0.00027401376,0.0022616023,0.00048051783,0.00029858647,0.00073477,0.00033080086,0.62779117,0.010423939,0.044148196,0.04629899,0.26606438],"study_design_scores_gemma":[0.000044174154,0.000030349394,0.0000876492,0.000008251799,0.000014850406,0.000034412053,0.000017787648,0.9828415,0.0010920845,0.011857743,0.0039578206,0.000013448556],"about_ca_topic_score_codex":0.012417783,"about_ca_topic_score_gemma":0.015969504,"teacher_disagreement_score":0.012417783,"about_ca_system_score_codex":0.0015793631,"about_ca_system_score_gemma":0.002901248,"threshold_uncertainty_score":0.036634684},"labels":[],"label_agreement":null},{"id":"W7106836265","doi":"10.5281/zenodo.17719730","title":"Efficient LLM Self-Hosting using Adapters and VLLM Deployment","year":2025,"lang":"en","type":"article","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Software deployment; Modular design; Orchestration; Abstraction; Dependency (UML); Personalization; Architecture; Inference","score_opus":0.022764461538552508,"score_gpt":0.26150220000319274,"score_spread":0.23873773846464025,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7106836265","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04849971,0.0005025724,0.90545344,0.00049657514,0.00011116743,0.00021381152,0.00021901369,0.035822008,0.008681697],"genre_scores_gemma":[0.50048923,0.0005211167,0.4872372,0.00040927617,0.0000623346,0.00029122765,0.001226678,0.0035234706,0.0062394696],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99812156,0.00064163585,0.00015475639,0.0003584552,0.0005183099,0.00020523096],"domain_scores_gemma":[0.99683726,0.0007751301,0.00018359134,0.0017772865,0.00032838562,0.00009836445],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027052064,0.00053703727,0.00046338234,0.00092330464,0.00047265342,0.00208738,0.0020478785,0.0008158509,0.0028816387],"category_scores_gemma":[0.008330837,0.0004938942,0.00084720686,0.0007166403,0.0007117199,0.003762331,0.0034383098,0.0013787801,0.0015159351],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00043372627,0.00033666205,0.01077351,0.00076538953,0.00020174473,0.0010563656,0.0023317733,0.058843642,0.07774653,0.09398905,0.017050752,0.73647076],"study_design_scores_gemma":[0.00008880967,0.000260405,0.005811982,0.00022454376,0.00016638734,0.0010004966,0.00086302386,0.6654749,0.11551273,0.081537835,0.12889852,0.00016038347],"about_ca_topic_score_codex":0.0017548967,"about_ca_topic_score_gemma":0.0017690143,"teacher_disagreement_score":0.0028816387,"about_ca_system_score_codex":0.0008883051,"about_ca_system_score_gemma":0.00092867995,"threshold_uncertainty_score":0.0143066645},"labels":[],"label_agreement":null},{"id":"W7106839461","doi":"10.5281/zenodo.17719731","title":"Efficient LLM Self-Hosting using Adapters and VLLM Deployment","year":2025,"lang":"en","type":"article","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Software deployment; Modular design; Orchestration; Abstraction; Dependency (UML); Personalization; Architecture; Inference","score_opus":0.022764461538552508,"score_gpt":0.26150220000319274,"score_spread":0.23873773846464025,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7106839461","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04849971,0.0005025724,0.90545344,0.00049657514,0.00011116743,0.00021381152,0.00021901369,0.035822008,0.008681697],"genre_scores_gemma":[0.50048923,0.0005211167,0.4872372,0.00040927617,0.0000623346,0.00029122765,0.001226678,0.0035234706,0.0062394696],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99812156,0.00064163585,0.00015475639,0.0003584552,0.0005183099,0.00020523096],"domain_scores_gemma":[0.99683726,0.0007751301,0.00018359134,0.0017772865,0.00032838562,0.00009836445],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027052064,0.00053703727,0.00046338234,0.00092330464,0.00047265342,0.00208738,0.0020478785,0.0008158509,0.0028816387],"category_scores_gemma":[0.008330837,0.0004938942,0.00084720686,0.0007166403,0.0007117199,0.003762331,0.0034383098,0.0013787801,0.0015159351],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00043372627,0.00033666205,0.01077351,0.00076538953,0.00020174473,0.0010563656,0.0023317733,0.058843642,0.07774653,0.09398905,0.017050752,0.73647076],"study_design_scores_gemma":[0.00008880967,0.000260405,0.005811982,0.00022454376,0.00016638734,0.0010004966,0.00086302386,0.6654749,0.11551273,0.081537835,0.12889852,0.00016038347],"about_ca_topic_score_codex":0.0017548967,"about_ca_topic_score_gemma":0.0017690143,"teacher_disagreement_score":0.0028816387,"about_ca_system_score_codex":0.0008883051,"about_ca_system_score_gemma":0.00092867995,"threshold_uncertainty_score":0.0143066645},"labels":[],"label_agreement":null},{"id":"W7107853802","doi":"10.5281/zenodo.17641070","title":"Teaching translation students about data in the age of generative AI","year":2025,"lang":"en","type":"book-chapter","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Translation (biology); Point (geometry); Generative grammar; Key (lock); Machine translation; Visualization","score_opus":0.05865548396735978,"score_gpt":0.32187783363248756,"score_spread":0.2632223496651278,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7107853802","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019007634,0.008594462,0.29981738,0.041987427,0.0024255493,0.00013610219,0.00019772898,0.001864017,0.6259696],"genre_scores_gemma":[0.21391656,0.018159209,0.22879066,0.013842236,0.0010796438,0.0002657549,0.00058222894,0.0020473995,0.5213163],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99938464,0.00028104766,0.000020037805,0.000097812735,0.00016525565,0.00005118356],"domain_scores_gemma":[0.99848455,0.0011277377,0.000031035164,0.00012273979,0.00014087978,0.00009303862],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014946822,0.00044582254,0.00031377794,0.00057490566,0.0014967873,0.0041619632,0.00068266975,0.0015057507,0.017916447],"category_scores_gemma":[0.004188878,0.00034122137,0.00035107092,0.0007500491,0.0031515635,0.009130727,0.002191298,0.0047006537,0.006808444],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000022472086,0.00008489658,0.00030025942,0.00020877017,0.0000045737834,0.00020373476,0.015859498,0.0006822441,0.0020813236,0.6944538,0.076685585,0.20941281],"study_design_scores_gemma":[0.000007755021,0.00002452398,0.00016275559,0.00017982307,0.000003980766,0.0004028132,0.0039268374,0.0016178637,0.0019570824,0.2338724,0.757833,0.0000111180725],"about_ca_topic_score_codex":0.0006849448,"about_ca_topic_score_gemma":0.001732976,"teacher_disagreement_score":0.017916447,"about_ca_system_score_codex":0.0014043902,"about_ca_system_score_gemma":0.0011627492,"threshold_uncertainty_score":0.059936464},"labels":[],"label_agreement":null},{"id":"W7107871876","doi":"10.5281/zenodo.17641069","title":"Teaching translation students about data in the age of generative AI","year":2025,"lang":"en","type":"book-chapter","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Translation (biology); Point (geometry); Generative grammar; Key (lock); Machine translation; Visualization","score_opus":0.05865548396735978,"score_gpt":0.32187783363248756,"score_spread":0.2632223496651278,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7107871876","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.019007634,0.008594462,0.29981738,0.041987427,0.0024255493,0.00013610219,0.00019772898,0.001864017,0.6259696],"genre_scores_gemma":[0.21391656,0.018159209,0.22879066,0.013842236,0.0010796438,0.0002657549,0.00058222894,0.0020473995,0.5213163],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99938464,0.00028104766,0.000020037805,0.000097812735,0.00016525565,0.00005118356],"domain_scores_gemma":[0.99848455,0.0011277377,0.000031035164,0.00012273979,0.00014087978,0.00009303862],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014946822,0.00044582254,0.00031377794,0.00057490566,0.0014967873,0.0041619632,0.00068266975,0.0015057507,0.017916447],"category_scores_gemma":[0.004188878,0.00034122137,0.00035107092,0.0007500491,0.0031515635,0.009130727,0.002191298,0.0047006537,0.006808444],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000022472086,0.00008489658,0.00030025942,0.00020877017,0.0000045737834,0.00020373476,0.015859498,0.0006822441,0.0020813236,0.6944538,0.076685585,0.20941281],"study_design_scores_gemma":[0.000007755021,0.00002452398,0.00016275559,0.00017982307,0.000003980766,0.0004028132,0.0039268374,0.0016178637,0.0019570824,0.2338724,0.757833,0.0000111180725],"about_ca_topic_score_codex":0.0006849448,"about_ca_topic_score_gemma":0.001732976,"teacher_disagreement_score":0.017916447,"about_ca_system_score_codex":0.0014043902,"about_ca_system_score_gemma":0.0011627492,"threshold_uncertainty_score":0.059936464},"labels":[],"label_agreement":null},{"id":"W7109084896","doi":"10.1007/s10664-025-10768-1","title":"Output format biases in the evaluation of large language models for code translation","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada); Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada; Vector Institute","keywords":"Executable; Source code; Disk formatting; Code (set theory); Code review; Translation (biology); Empirical research; Software","score_opus":0.07234957541314808,"score_gpt":0.3707188625070804,"score_spread":0.29836928709393234,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7109084896","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8468252,0.001633839,0.13430431,0.0019327302,0.00033322655,0.0003252197,0.0018938138,0.0042976458,0.008453984],"genre_scores_gemma":[0.9652106,0.00022131298,0.029844569,0.00033207235,0.00008796525,0.00016288679,0.0023049137,0.0010036937,0.00083190866],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9609644,0.030734723,0.0022993083,0.0019341254,0.0036167782,0.00045068658],"domain_scores_gemma":[0.58200186,0.38894907,0.005332486,0.0118328715,0.010684276,0.0011995554],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.03710896,0.0014422827,0.0011425877,0.0022448876,0.0010562702,0.0047627543,0.0018145934,0.0025088727,0.003623045],"category_scores_gemma":[0.31157106,0.00071633206,0.0009893671,0.0024673468,0.002073056,0.0071293428,0.003557967,0.002905993,0.0012278149],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.011445871,0.002084312,0.08483738,0.0033373667,0.0014912729,0.00081946206,0.0073672715,0.31085327,0.017593995,0.03318355,0.018846728,0.50813943],"study_design_scores_gemma":[0.00092866394,0.0012454706,0.012000351,0.00038440563,0.0005370032,0.00033630387,0.0015364939,0.89983594,0.028812708,0.050049208,0.004175148,0.00015839397],"about_ca_topic_score_codex":0.0036910358,"about_ca_topic_score_gemma":0.0038109133,"teacher_disagreement_score":0.96289104,"about_ca_system_score_codex":0.0021573314,"about_ca_system_score_gemma":0.0020369978,"threshold_uncertainty_score":0.1962533},"labels":[],"label_agreement":null},{"id":"W71098462","doi":"10.1007/978-3-642-54516-0_3","title":"Issues in Analogical Inference Over Sequences of Symbols: A Case Study on Proper Name Transliteration","year":2014,"lang":"en","type":"book-chapter","venue":"Studies in computational intelligence","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Analogy; Computer science; Natural language processing; Variety (cybernetics); Artificial intelligence; Inference; Transliteration; Task (project management); Linguistics; Engineering; Philosophy","score_opus":0.12369861157189486,"score_gpt":0.4236749413063557,"score_spread":0.2999763297344608,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W71098462","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13962118,0.0050727725,0.60024506,0.013044414,0.0004204285,0.00020070944,0.00029093516,0.0011158616,0.23998864],"genre_scores_gemma":[0.7971969,0.002273025,0.17346065,0.0012439789,0.0003511339,0.00010010509,0.00030714646,0.0006044206,0.024462584],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9928986,0.0045030946,0.000377265,0.0008904936,0.0010860291,0.00024466295],"domain_scores_gemma":[0.950122,0.04419867,0.0006673043,0.0036418366,0.001177961,0.00019219218],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007586103,0.0004986207,0.00070750376,0.0011845907,0.0027145478,0.004739751,0.0030385312,0.0034698993,0.012108653],"category_scores_gemma":[0.044355713,0.0005691273,0.0008970927,0.0029014922,0.006197426,0.018892583,0.0035878634,0.004673633,0.0012547792],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011216364,0.00011934134,0.0009623073,0.00022804982,0.000020822163,0.0018592393,0.009142154,0.0036679334,0.0012720895,0.8489293,0.0047411392,0.12894554],"study_design_scores_gemma":[0.000030203768,0.000030166519,0.00043219852,0.00011356077,0.000027999,0.0014948002,0.0021878926,0.017598659,0.0028511614,0.93791586,0.03728853,0.000028932003],"about_ca_topic_score_codex":0.0042487085,"about_ca_topic_score_gemma":0.0041563227,"teacher_disagreement_score":0.012108653,"about_ca_system_score_codex":0.0022972047,"about_ca_system_score_gemma":0.0016129329,"threshold_uncertainty_score":0.040507495},"labels":[],"label_agreement":null},{"id":"W7110977668","doi":"10.5281/zenodo.17850892","title":"The Collapse Index (CI) CrackTest: Morphology-Aligned Perturbation Testing Reveals Systematic Collapse Inheritance in Frontier Language Models","year":2025,"lang":"en","type":"preprint","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Glycemic Index Laboratories","funders":"","keywords":"Inference; Bounded function; Robustness (evolution); Perturbation (astronomy); Observable; Progressive collapse","score_opus":0.03268787693774515,"score_gpt":0.2676161464680126,"score_spread":0.2349282695302674,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7110977668","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20917413,0.00018129229,0.7738673,0.00037737633,0.00009127632,0.00014517497,0.0015050436,0.010616251,0.004042226],"genre_scores_gemma":[0.8631113,0.00007390738,0.131402,0.00014727113,0.000035590336,0.00014722592,0.0024917792,0.0016051646,0.0009858308],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99645656,0.0015062373,0.00016828268,0.0006105799,0.0010493512,0.00020897071],"domain_scores_gemma":[0.98796403,0.007438964,0.0008017292,0.0024891961,0.0010205786,0.00028549263],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003808691,0.0009876767,0.000548342,0.0012437186,0.0005594477,0.0019397129,0.0014189902,0.0009384088,0.003050357],"category_scores_gemma":[0.023568153,0.00038495337,0.0008964262,0.0004788481,0.0016231701,0.0024400859,0.002634785,0.0019496713,0.0008136053],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015533775,0.00033893678,0.029229462,0.00048989657,0.00036220413,0.0009122522,0.0012545262,0.5261137,0.07907905,0.07259831,0.014432708,0.27363557],"study_design_scores_gemma":[0.000026784555,0.00018904069,0.0026192104,0.000029960798,0.000027508053,0.00015076138,0.00012889648,0.9378646,0.020024074,0.037170462,0.0017237214,0.00004499639],"about_ca_topic_score_codex":0.0030081475,"about_ca_topic_score_gemma":0.00341268,"teacher_disagreement_score":0.003808691,"about_ca_system_score_codex":0.00096379704,"about_ca_system_score_gemma":0.0015929759,"threshold_uncertainty_score":0.020142555},"labels":[],"label_agreement":null},{"id":"W7111103067","doi":"10.5281/zenodo.17850893","title":"The Collapse Index (CI) CrackTest: Morphology-Aligned Perturbation Testing Reveals Systematic Collapse Inheritance in Frontier Language Models","year":2025,"lang":"en","type":"preprint","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Glycemic Index Laboratories","funders":"","keywords":"Inference; Bounded function; Robustness (evolution); Perturbation (astronomy); Observable; Progressive collapse","score_opus":0.03268787693774515,"score_gpt":0.2676161464680126,"score_spread":0.2349282695302674,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7111103067","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20917413,0.00018129229,0.7738673,0.00037737633,0.00009127632,0.00014517497,0.0015050436,0.010616251,0.004042226],"genre_scores_gemma":[0.8631113,0.00007390738,0.131402,0.00014727113,0.000035590336,0.00014722592,0.0024917792,0.0016051646,0.0009858308],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99645656,0.0015062373,0.00016828268,0.0006105799,0.0010493512,0.00020897071],"domain_scores_gemma":[0.98796403,0.007438964,0.0008017292,0.0024891961,0.0010205786,0.00028549263],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003808691,0.0009876767,0.000548342,0.0012437186,0.0005594477,0.0019397129,0.0014189902,0.0009384088,0.003050357],"category_scores_gemma":[0.023568153,0.00038495337,0.0008964262,0.0004788481,0.0016231701,0.0024400859,0.002634785,0.0019496713,0.0008136053],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015533775,0.00033893678,0.029229462,0.00048989657,0.00036220413,0.0009122522,0.0012545262,0.5261137,0.07907905,0.07259831,0.014432708,0.27363557],"study_design_scores_gemma":[0.000026784555,0.00018904069,0.0026192104,0.000029960798,0.000027508053,0.00015076138,0.00012889648,0.9378646,0.020024074,0.037170462,0.0017237214,0.00004499639],"about_ca_topic_score_codex":0.0030081475,"about_ca_topic_score_gemma":0.00341268,"teacher_disagreement_score":0.003808691,"about_ca_system_score_codex":0.00096379704,"about_ca_system_score_gemma":0.0015929759,"threshold_uncertainty_score":0.020142555},"labels":[],"label_agreement":null},{"id":"W7111753261","doi":"","title":"MTUncertainty:Assessing the Need for Post-editing of Machine Translation Outputs by Fine-tuning OpenAI LLMs","year":2023,"lang":"en","type":"article","venue":"Research Explorer (The University of Manchester)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Open Text (Canada)","funders":"Dell Technologies","keywords":"Machine translation; Translation (biology); Quality (philosophy); Process (computing); Task (project management); Binary number; Work (physics)","score_opus":0.09153001966312556,"score_gpt":0.34514275179968473,"score_spread":0.25361273213655916,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7111753261","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6205522,0.0059222193,0.29086342,0.0013569889,0.0007600252,0.00078588974,0.0038294124,0.06433778,0.011592112],"genre_scores_gemma":[0.83966553,0.000586548,0.14079672,0.0005282029,0.00009935563,0.00046361465,0.011148233,0.0028266383,0.0038850992],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9940692,0.0021185307,0.0006866033,0.0015464214,0.0012704834,0.0003088407],"domain_scores_gemma":[0.9795076,0.011904757,0.0015068575,0.003372543,0.0030584617,0.00064975914],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008395037,0.0023866359,0.0014984158,0.0017917881,0.0010227672,0.0029439267,0.0025612558,0.0025511514,0.003115876],"category_scores_gemma":[0.038633708,0.00077314617,0.0012396045,0.0016096375,0.0010935132,0.0044204867,0.0029239198,0.00342576,0.0029247208],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004475152,0.0013940894,0.03811031,0.0022905087,0.0011718393,0.0007477835,0.002034649,0.24446234,0.06493261,0.0031162684,0.022079898,0.61518455],"study_design_scores_gemma":[0.00023343382,0.0009938945,0.01028769,0.00011708565,0.00037455687,0.00034948066,0.00042728952,0.92722434,0.04609171,0.00373347,0.010024159,0.000142839],"about_ca_topic_score_codex":0.011594181,"about_ca_topic_score_gemma":0.011826889,"teacher_disagreement_score":0.011594181,"about_ca_system_score_codex":0.0014932653,"about_ca_system_score_gemma":0.0019085556,"threshold_uncertainty_score":0.04439771},"labels":[],"label_agreement":null},{"id":"W7113160520","doi":"","title":"Train &amp; Constrain: Phonologically Informed Tongue-Twister Generation from Topics and Paraphrases","year":2025,"lang":"en","type":"article","venue":"Research Explorer (The University of Manchester)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Open Text (Canada)","funders":"","keywords":"Set (abstract data type); Feature (linguistics); Action (physics); Key (lock)","score_opus":0.10663848811608352,"score_gpt":0.3305961335946331,"score_spread":0.22395764547854957,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7113160520","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13551156,0.00027921895,0.7557698,0.00055214064,0.0005489654,0.00042701006,0.0044739177,0.07568018,0.026757225],"genre_scores_gemma":[0.63598096,0.0001590115,0.33282664,0.00015078342,0.00013098492,0.00021013114,0.0074145272,0.0084105935,0.014716365],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993205,0.00018288394,0.000046959478,0.0002100906,0.00015266484,0.00008686403],"domain_scores_gemma":[0.99812967,0.0010756436,0.000066358305,0.00041535022,0.00023197956,0.000080949474],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008104522,0.0012318194,0.0007704081,0.00090647547,0.0006850048,0.0018949836,0.0017548809,0.0012729845,0.031125296],"category_scores_gemma":[0.004548231,0.00088772166,0.0006405752,0.0008093952,0.0007417353,0.0023070364,0.003806611,0.0013516807,0.010729644],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013257904,0.00021052947,0.0020266832,0.00054667005,0.00007349801,0.0014458023,0.0017162281,0.014235732,0.10724613,0.029227786,0.045114737,0.7968304],"study_design_scores_gemma":[0.00034630444,0.00035173414,0.0028040167,0.00011198835,0.00014003726,0.0010243538,0.0015017081,0.69595474,0.16455385,0.07947498,0.05356027,0.00017606653],"about_ca_topic_score_codex":0.0013313885,"about_ca_topic_score_gemma":0.0024799812,"teacher_disagreement_score":0.031125296,"about_ca_system_score_codex":0.0003612455,"about_ca_system_score_gemma":0.00069734897,"threshold_uncertainty_score":0.104124546},"labels":[],"label_agreement":null},{"id":"W7113665792","doi":"","title":"Monolingual Phrase Alignment on Parse Forests","year":2017,"lang":"en","type":"article","venue":"Research Explorer (The University of Manchester)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Open Text (Canada)","funders":"","keywords":"Paraphrase; Phrase; Parsing; Phrase structure rules; Determiner phrase; Endocentric and exocentric","score_opus":0.10358168271371866,"score_gpt":0.3453075364313036,"score_spread":0.24172585371758493,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7113665792","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014102115,0.00059117697,0.9632284,0.000111546055,0.00013582334,0.00022958398,0.0024448729,0.016524807,0.0026316564],"genre_scores_gemma":[0.10146573,0.00031725175,0.87960637,0.000162333,0.00012435752,0.00035556796,0.013776681,0.0016550017,0.002536707],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99693537,0.0007767779,0.00026403094,0.0008574761,0.00088514475,0.00028117286],"domain_scores_gemma":[0.9966363,0.0009543252,0.00031617016,0.000823826,0.0011440899,0.00012524052],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018306699,0.0017319879,0.0012963228,0.0046414053,0.0013392072,0.001696958,0.0019736127,0.0012817712,0.0055648587],"category_scores_gemma":[0.0074657337,0.0008286274,0.0014696906,0.005482677,0.0005528126,0.0032006374,0.00255284,0.0017752842,0.006909994],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036088843,0.00032029187,0.004508867,0.00058974605,0.0002396625,0.0006254216,0.0005162428,0.0144625325,0.0577397,0.012097342,0.044633456,0.86390585],"study_design_scores_gemma":[0.00016299941,0.00045013477,0.007383153,0.00021428525,0.00027894782,0.0022427568,0.0007469524,0.70301014,0.08936085,0.11037569,0.08556698,0.00020709782],"about_ca_topic_score_codex":0.0035471015,"about_ca_topic_score_gemma":0.007968548,"teacher_disagreement_score":0.0055648587,"about_ca_system_score_codex":0.00045073387,"about_ca_system_score_gemma":0.002197897,"threshold_uncertainty_score":0.018616319},"labels":[],"label_agreement":null},{"id":"W7115274031","doi":"","title":"Bitext Lexical Dataset - Language Variants - French","year":2023,"lang":"fr","type":"article","venue":"The COCOON platform (University of Paris)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Complement (music); Vocabulary; Variation (astronomy); French; Lexical item; Component (thermodynamics)","score_opus":0.037266910510041856,"score_gpt":0.27085167011762323,"score_spread":0.23358475960758138,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7115274031","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009104397,0.00052966585,0.0014098134,0.0002063112,0.000159929,0.00014469532,0.97422385,0.007888023,0.0063334364],"genre_scores_gemma":[0.0043659727,0.000072299605,0.001519664,0.00005996443,0.000013481881,0.0001262074,0.9924971,0.00027287408,0.001072455],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99781644,0.00039343713,0.00024622533,0.00064851076,0.00063053524,0.00026483674],"domain_scores_gemma":[0.9980596,0.00054961967,0.00011420196,0.0005558861,0.0005938078,0.00012688107],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011915722,0.003435383,0.0016515111,0.006498291,0.0015252227,0.002631296,0.0022699968,0.002922373,0.049088057],"category_scores_gemma":[0.005409219,0.0007983909,0.0019100879,0.0053411857,0.00069431774,0.0019124572,0.002304599,0.0021027254,0.052706882],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005041095,0.00024409266,0.003872574,0.0012269057,0.00016176401,0.0006892548,0.00015308967,0.0015296832,0.0031169727,0.002370992,0.9544489,0.031681813],"study_design_scores_gemma":[0.00094919186,0.00021518121,0.021947537,0.00047334837,0.00014802339,0.0020671505,0.00051525293,0.0072966157,0.0042817816,0.0026897423,0.9592759,0.00014019305],"about_ca_topic_score_codex":0.025042912,"about_ca_topic_score_gemma":0.033177476,"teacher_disagreement_score":0.049088057,"about_ca_system_score_codex":0.0012557212,"about_ca_system_score_gemma":0.0015165017,"threshold_uncertainty_score":0.16421592},"labels":[],"label_agreement":null},{"id":"W7116675963","doi":"10.18280/mmep.121134","title":"Optimizing Machine Translation with Sequential Tokenizer Embeddings Using Convolutional Neural Network for Low-Resource Language","year":2025,"lang":"","type":"article","venue":"Mathematical Modelling and Engineering Problems","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Machine translation; Convolutional neural network; Translation (biology); Artificial neural network; Natural language","score_opus":0.02106121635771414,"score_gpt":0.25036113061202914,"score_spread":0.229299914254315,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7116675963","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06541434,0.0013946708,0.913307,0.00083639316,0.0005932222,0.00013308447,0.0008787195,0.011533274,0.0059092892],"genre_scores_gemma":[0.51179993,0.00073405233,0.46869949,0.00040862555,0.00027032365,0.00019971534,0.0039216178,0.0015332047,0.0124330195],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990711,0.00021109486,0.00007555406,0.00033586705,0.00016006238,0.0001463051],"domain_scores_gemma":[0.9984471,0.0008035584,0.00010137215,0.00027781838,0.00030609345,0.00006398792],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009872495,0.001436408,0.0015386891,0.00084773614,0.0007420357,0.0016897838,0.001343078,0.0015434627,0.007790128],"category_scores_gemma":[0.00396867,0.00064230093,0.0009928959,0.0017102528,0.0006654036,0.002764321,0.0015890766,0.0020202508,0.004877259],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008360084,0.00032143402,0.0010474125,0.00050279783,0.00019133528,0.0005478814,0.00016332859,0.29406968,0.019023325,0.017816095,0.024006143,0.6414746],"study_design_scores_gemma":[0.000043139484,0.00006991186,0.00015437015,0.000015423677,0.000034355082,0.000065359454,0.00004567532,0.9759182,0.005810796,0.015910478,0.0019165264,0.000015774614],"about_ca_topic_score_codex":0.008022976,"about_ca_topic_score_gemma":0.015867751,"teacher_disagreement_score":0.008022976,"about_ca_system_score_codex":0.0011115072,"about_ca_system_score_gemma":0.0025965145,"threshold_uncertainty_score":0.026060522},"labels":[],"label_agreement":null},{"id":"W7117534066","doi":"10.21203/rs.3.rs-7632448/v1","title":"A Perspective on the Use of Sanskrit Language and Literature in Developing AI and GenAI Systems","year":2025,"lang":"","type":"preprint","venue":"Research Square","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Sanskrit; Generative grammar; Grammar; Computational linguistics; Ambiguity; Perspective (graphical); Natural language","score_opus":0.07149295829794415,"score_gpt":0.4122171987028535,"score_spread":0.34072424040490934,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7117534066","genre_codex":"methods","genre_gemma":"commentary","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"commentary","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.031520516,0.022808494,0.43336397,0.08073636,0.0006815837,0.00018108064,0.0005827743,0.0007157522,0.42940947],"genre_scores_gemma":[0.58766514,0.017769942,0.35916892,0.0035150687,0.00074364693,0.000362507,0.0004988628,0.0006709826,0.029604856],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9949161,0.0036149342,0.00028598253,0.0004224815,0.0005716665,0.00018883949],"domain_scores_gemma":[0.9775889,0.017990472,0.0008148947,0.0013864357,0.0016038204,0.0006154997],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008546928,0.00090605067,0.0006801945,0.011398592,0.0040771603,0.014345093,0.0024067836,0.0033086638,0.006612348],"category_scores_gemma":[0.008904762,0.00093068444,0.0008988262,0.0064328425,0.024444412,0.023493,0.00374324,0.0043824613,0.00151183],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000006828703,0.000007204903,0.00016814012,0.00006716097,0.000003969984,0.00010471527,0.0042429124,0.00016135386,0.00025462644,0.9872372,0.00080738013,0.0069384864],"study_design_scores_gemma":[0.00001425407,0.00002747465,0.0005487175,0.00037792893,0.000017400876,0.0004533359,0.012618522,0.0029652752,0.0018229323,0.8089603,0.172162,0.000031720025],"about_ca_topic_score_codex":0.00986781,"about_ca_topic_score_gemma":0.013265903,"teacher_disagreement_score":0.014345093,"about_ca_system_score_codex":0.005598695,"about_ca_system_score_gemma":0.004341799,"threshold_uncertainty_score":0.045201063},"labels":[],"label_agreement":null},{"id":"W7117712979","doi":"10.18280/isi.301117","title":"Transformer-Based Semantic Search Engine with Morphophonemic Stemming for Low-Resource Sundanese Language","year":2025,"lang":"","type":"article","venue":"Ingénierie des systèmes d information","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Natural language; Search engine indexing; Semantics (computer science); Morphophonology","score_opus":0.008117626699749214,"score_gpt":0.2493247019844946,"score_spread":0.2412070752847454,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7117712979","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07497739,0.0021159078,0.8215229,0.00059622846,0.0003798512,0.00049256283,0.010230951,0.07577727,0.013906896],"genre_scores_gemma":[0.3303469,0.00086287654,0.62798697,0.00036876783,0.000103038896,0.00018792332,0.02776879,0.0030016948,0.009372959],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99937975,0.00009361605,0.000120058394,0.00015382233,0.00017143958,0.00008126254],"domain_scores_gemma":[0.99931717,0.00016768287,0.00003323246,0.000116665644,0.00032341084,0.000041851134],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005191393,0.0008027335,0.0012900339,0.0038701482,0.001022677,0.0015481197,0.0010304536,0.00074258074,0.011474491],"category_scores_gemma":[0.0016192133,0.0003298667,0.0009906556,0.0029264342,0.0003214154,0.0031990008,0.0019460752,0.00069728243,0.006393759],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009377132,0.00041742902,0.0019407636,0.0014740746,0.0002453368,0.000989627,0.0005512292,0.00346197,0.10596851,0.02970298,0.050419092,0.8038913],"study_design_scores_gemma":[0.00052588986,0.00055894593,0.0033237012,0.00020298775,0.0007956658,0.003462519,0.001804456,0.528784,0.23456444,0.09484889,0.13081574,0.0003127918],"about_ca_topic_score_codex":0.00367673,"about_ca_topic_score_gemma":0.008289235,"teacher_disagreement_score":0.011474491,"about_ca_system_score_codex":0.0005143662,"about_ca_system_score_gemma":0.0021794443,"threshold_uncertainty_score":0.038385987},"labels":[],"label_agreement":null},{"id":"W7125517337","doi":"10.55492/v6i02.6736","title":"Mafoko: Structuring and Building Open Multilingual Terminologies for South African NLP","year":2025,"lang":"","type":"article","venue":"Journal of the Digital Humanities Association of Southern Africa","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"University of Pretoria; International Development Research Centre; Nvidia","keywords":"Terminology; Interoperability; Bridging (networking); Consistency (knowledge bases); Structuring; Scalability; Machine translation","score_opus":0.026973400963562027,"score_gpt":0.2787776361997294,"score_spread":0.25180423523616735,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7125517337","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08334444,0.003067711,0.4747057,0.0048040575,0.0009952382,0.0035339538,0.31630695,0.07600853,0.037233394],"genre_scores_gemma":[0.0865445,0.0008288653,0.4794304,0.00057440164,0.000090309935,0.0028031957,0.42033058,0.005701863,0.0036959848],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9965844,0.0013125044,0.0005175198,0.00071641686,0.00064636796,0.00022274208],"domain_scores_gemma":[0.99353987,0.0025706247,0.000447346,0.0021771744,0.0009631714,0.00030190588],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005061758,0.0011565528,0.00072967436,0.00669365,0.0024705022,0.0034443242,0.002122248,0.0015183714,0.010486954],"category_scores_gemma":[0.024501076,0.0007536996,0.0016790656,0.0043795765,0.0015451971,0.0075611295,0.009805645,0.0028706808,0.008413125],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007541061,0.00041789716,0.01175041,0.006438436,0.00029607993,0.0020297854,0.011384202,0.016757343,0.03880847,0.10227085,0.43895775,0.37013465],"study_design_scores_gemma":[0.0002945795,0.00012687057,0.0072592376,0.0010897911,0.00008788932,0.0009806011,0.004633814,0.060302936,0.02036165,0.080035776,0.82464135,0.00018556569],"about_ca_topic_score_codex":0.0086517325,"about_ca_topic_score_gemma":0.014007149,"teacher_disagreement_score":0.010486954,"about_ca_system_score_codex":0.001976848,"about_ca_system_score_gemma":0.005309802,"threshold_uncertainty_score":0.0350824},"labels":[],"label_agreement":null},{"id":"W7125616198","doi":"10.1109/cascon66301.2025.00114","title":"Task-Aware Reduction for Scalable LLM-Database Systems","year":2025,"lang":"","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Trent University; University of Toronto","funders":"","keywords":"Scalability; Reduction (mathematics); Task (project management); Position paper; Upstream (networking); Downstream (manufacturing); Preprocessor; Noise (video)","score_opus":0.014887510264267228,"score_gpt":0.2953410808862676,"score_spread":0.2804535706220004,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7125616198","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.036005143,0.00077069615,0.9331382,0.0013418094,0.00016866084,0.00031914187,0.00078226416,0.023807304,0.003666853],"genre_scores_gemma":[0.40378085,0.0003278358,0.5838565,0.00092385884,0.00018087692,0.0006375343,0.002168548,0.0017764245,0.0063475035],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99733365,0.0005898103,0.00026100362,0.00067044224,0.00080576155,0.0003394773],"domain_scores_gemma":[0.99523395,0.0017701237,0.00023632235,0.0016470811,0.000852625,0.00025981432],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002631564,0.0012880948,0.0013769652,0.0010331008,0.0011855294,0.0028862678,0.004934285,0.0010412337,0.006415221],"category_scores_gemma":[0.010339657,0.0007049308,0.00097512227,0.001430429,0.001134786,0.00463205,0.0041049398,0.002209177,0.0029583825],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017216763,0.00077320094,0.004718131,0.0011484061,0.00018250397,0.00049609016,0.0012807803,0.19528782,0.08485988,0.051342245,0.056112573,0.6020766],"study_design_scores_gemma":[0.000089953304,0.00014075635,0.00047254178,0.000028817096,0.00005602645,0.00011030574,0.00019163062,0.908392,0.031850565,0.046363045,0.012266899,0.000037470232],"about_ca_topic_score_codex":0.0053699007,"about_ca_topic_score_gemma":0.009257563,"teacher_disagreement_score":0.006415221,"about_ca_system_score_codex":0.0023327176,"about_ca_system_score_gemma":0.0031742004,"threshold_uncertainty_score":0.02146107},"labels":[],"label_agreement":null},{"id":"W7125651331","doi":"10.1109/cascon66301.2025.00128","title":"A Fully Automated Agent for End-to-End Code Translation and Validation","year":2025,"lang":"","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Code (set theory); Translation (biology); Automation; Source code; Key (lock)","score_opus":0.027997341703327894,"score_gpt":0.32470502124732664,"score_spread":0.29670767954399874,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7125651331","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0070201755,0.000089404086,0.8993738,0.00021912475,0.00018207566,0.0003700897,0.0006345079,0.08838722,0.0037236253],"genre_scores_gemma":[0.11166574,0.00011329588,0.8649408,0.00035800738,0.00007118364,0.00047544934,0.0030582875,0.008932022,0.01038516],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9955031,0.0013131299,0.00042673064,0.0008544134,0.0016116172,0.0002910156],"domain_scores_gemma":[0.99097323,0.0025512748,0.00048250981,0.0033001248,0.0023748085,0.00031807876],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037068673,0.0012922438,0.001305049,0.001277977,0.0011282708,0.003390726,0.0026846013,0.002461851,0.012911718],"category_scores_gemma":[0.012216836,0.0012333903,0.0014828242,0.00055181026,0.0011457878,0.0037083768,0.004541052,0.0027750125,0.012068524],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002114892,0.0014683572,0.004570093,0.0010701542,0.0004388191,0.0017003686,0.0014776102,0.049561974,0.11628698,0.062258262,0.10209843,0.6569541],"study_design_scores_gemma":[0.00023878252,0.00030155416,0.0010150198,0.00013166528,0.00012368018,0.0005517156,0.00018981584,0.75204873,0.10997334,0.036570154,0.09870537,0.00015014264],"about_ca_topic_score_codex":0.0026307679,"about_ca_topic_score_gemma":0.0033448471,"teacher_disagreement_score":0.012911718,"about_ca_system_score_codex":0.0006598196,"about_ca_system_score_gemma":0.0032321848,"threshold_uncertainty_score":0.043193936},"labels":[],"label_agreement":null},{"id":"W7127642937","doi":"10.1109/cicn67655.2025.11368293","title":"PromptOS: Study of Natural-Language Shells, Execution-Based Evaluation, Command-Syntax Learning, and Runtime Guardrails","year":2025,"lang":"","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Trinity College","funders":"","keywords":"Undo; Workflow; Cloud computing; Key (lock); Parsing; Overlay; Architecture; Syntax","score_opus":0.00831503984126254,"score_gpt":0.3268386743978949,"score_spread":0.31852363455663235,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7127642937","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0770787,0.0003020802,0.8967358,0.00043916676,0.000047266632,0.00023996057,0.00026963803,0.01834833,0.0065391297],"genre_scores_gemma":[0.5039058,0.00033887135,0.4850558,0.00025775903,0.000033955293,0.0002463169,0.0012187357,0.0041400627,0.004802747],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9922092,0.0037752495,0.00052133354,0.0010760204,0.0021041778,0.00031395926],"domain_scores_gemma":[0.9706293,0.01844524,0.001732823,0.0051523675,0.0034460379,0.0005943049],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008145505,0.0010257082,0.0006451673,0.0011976805,0.0005864285,0.0029387958,0.0019865315,0.0008382837,0.0038988837],"category_scores_gemma":[0.039647046,0.0006854081,0.000740343,0.00081967935,0.0025403048,0.00821838,0.002802755,0.0023041142,0.0011231947],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009943983,0.00055297394,0.017686361,0.0011463276,0.000107733424,0.0004203556,0.005777234,0.08190114,0.028520176,0.26866633,0.009146936,0.58508],"study_design_scores_gemma":[0.00007944377,0.00074722135,0.004779055,0.00030792915,0.00009267267,0.0006587939,0.0013505169,0.7452385,0.10852645,0.09422807,0.043852344,0.00013906536],"about_ca_topic_score_codex":0.003515948,"about_ca_topic_score_gemma":0.0025586756,"teacher_disagreement_score":0.008145505,"about_ca_system_score_codex":0.0017821676,"about_ca_system_score_gemma":0.0031254515,"threshold_uncertainty_score":0.043078125},"labels":[],"label_agreement":null},{"id":"W7130693008","doi":"10.5281/zenodo.18717174","title":"Natural Language Processing Challenges and Opportunities in African Languages of Togo","year":2000,"lang":"en","type":"article","venue":"Open MIND","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Robustness (evolution); Focus (optics); Component (thermodynamics); Natural language; Language model; Estimation; Statistical model; Process (computing)","score_opus":0.05055187660192826,"score_gpt":0.3221405269119667,"score_spread":0.2715886503100384,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7130693008","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.97076875,0.0006684551,0.018982656,0.0036184827,0.00002622564,0.000050023333,0.00025175593,0.00008062873,0.0055530085],"genre_scores_gemma":[0.9856523,0.00042822573,0.012725265,0.00016136961,0.000014828379,0.000059276645,0.00015900968,0.000029891282,0.00076982705],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9983708,0.0011294399,0.000063946005,0.00017334751,0.00011364515,0.00014883895],"domain_scores_gemma":[0.9929028,0.0057929726,0.00043914746,0.00034545394,0.00035151216,0.00016809818],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031779897,0.00038515875,0.00043590454,0.0012849865,0.0026420148,0.002514398,0.0005100765,0.00091311254,0.0025061867],"category_scores_gemma":[0.014756913,0.0002955522,0.00030440136,0.002094796,0.0016455231,0.0039080554,0.0022263762,0.0009893404,0.0003626376],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013775775,0.0003542789,0.15598527,0.0016010443,0.0001170473,0.0078651495,0.14022504,0.03838534,0.021931628,0.122991756,0.0042827805,0.5048831],"study_design_scores_gemma":[0.0001242347,0.0003231844,0.18328057,0.0011240601,0.000115259616,0.0070291925,0.31145042,0.19766691,0.01971088,0.21393688,0.06490101,0.00033747393],"about_ca_topic_score_codex":0.010217288,"about_ca_topic_score_gemma":0.02055801,"teacher_disagreement_score":0.010217288,"about_ca_system_score_codex":0.0011459412,"about_ca_system_score_gemma":0.002142194,"threshold_uncertainty_score":0.020315588},"labels":[],"label_agreement":null},{"id":"W7130703374","doi":"10.5281/zenodo.18717175","title":"Natural Language Processing Challenges and Opportunities in African Languages of Togo","year":2000,"lang":"en","type":"article","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Robustness (evolution); Focus (optics); Component (thermodynamics); Natural language; Language model; Estimation; Statistical model; Process (computing)","score_opus":0.039845601060946714,"score_gpt":0.26698846664947523,"score_spread":0.22714286558852853,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7130703374","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9698942,0.00077302323,0.020046385,0.0033208227,0.000026778263,0.000046992118,0.00035523795,0.000092381,0.005444274],"genre_scores_gemma":[0.9856818,0.0004798522,0.012556431,0.00014401024,0.000016998749,0.000055387303,0.00022888208,0.00003905932,0.0007975943],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99860317,0.0009511155,0.000053734206,0.00014987175,0.00010224284,0.0001399233],"domain_scores_gemma":[0.9948382,0.0040630437,0.00036369156,0.00029769255,0.00030125884,0.00013606722],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026055994,0.0003845354,0.0004329046,0.0012670236,0.0022722278,0.0021339343,0.00045242216,0.0008445494,0.0024748116],"category_scores_gemma":[0.011107924,0.00029108408,0.0002939162,0.0020575223,0.0014498961,0.0031909128,0.0018820156,0.0008026165,0.00039465853],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018544422,0.0003028762,0.16976322,0.0018212434,0.00013951425,0.008793777,0.09756764,0.047459684,0.032252736,0.13751765,0.0049096397,0.49761763],"study_design_scores_gemma":[0.00015525523,0.00032505704,0.2235584,0.0011286822,0.00014587383,0.008384183,0.2386873,0.21318768,0.02707502,0.20415583,0.08283288,0.00036386008],"about_ca_topic_score_codex":0.008137357,"about_ca_topic_score_gemma":0.01522782,"teacher_disagreement_score":0.008137357,"about_ca_system_score_codex":0.0009916272,"about_ca_system_score_gemma":0.0017605552,"threshold_uncertainty_score":0.016179979},"labels":[],"label_agreement":null},{"id":"W7131100807","doi":"10.1109/icrai68431.2025.11396704","title":"Optimized Fine-tuning and Pseudo-Data Strategies for Cross-Domain Low-Resource Language Cantonese-English Neural Machine Translation","year":2025,"lang":"","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Machine translation; Pipeline (software); Translation (biology); BLEU; Parallel corpora; Domain (mathematical analysis); Training set","score_opus":0.020574627652996242,"score_gpt":0.3230310482419668,"score_spread":0.3024564205889706,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7131100807","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.24576482,0.0015538968,0.7103151,0.0010795305,0.00044569123,0.000860726,0.0034520142,0.024539368,0.011988904],"genre_scores_gemma":[0.5962211,0.0002985982,0.3837293,0.00051587354,0.000061800594,0.0015985889,0.011140611,0.0023879702,0.0040461156],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9972799,0.0013583889,0.00020250717,0.00066219445,0.00031498403,0.00018210731],"domain_scores_gemma":[0.9950577,0.0019157887,0.0001725725,0.0015237074,0.001199165,0.00013114233],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0044909655,0.0019136628,0.0010086995,0.0012172086,0.0010282745,0.0014542297,0.002167169,0.0012385354,0.0050128903],"category_scores_gemma":[0.01989746,0.0007314329,0.0009659275,0.001397601,0.0010950441,0.003342335,0.0026149896,0.0025818506,0.0042615356],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00083828927,0.0009802037,0.010020273,0.0008960886,0.00031074186,0.00038611755,0.0007781949,0.32497153,0.037864033,0.011385591,0.026997631,0.5845713],"study_design_scores_gemma":[0.00019383436,0.0002823842,0.002769627,0.000095062715,0.00008732868,0.00020060853,0.00032586473,0.94330686,0.027387451,0.015421046,0.009858707,0.00007130631],"about_ca_topic_score_codex":0.009208038,"about_ca_topic_score_gemma":0.020806272,"teacher_disagreement_score":0.009208038,"about_ca_system_score_codex":0.001294775,"about_ca_system_score_gemma":0.0025956773,"threshold_uncertainty_score":0.023750782},"labels":[],"label_agreement":null},{"id":"W7131219091","doi":"10.1109/acsac67867.2025.00035","title":"PSan: Towards Hybrid Metadata Scheme for Efficient Pointer Checking","year":2025,"lang":"","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Metadata; Memory safety; Pointer (user interface); Memory protection; Metadata management; Overhead (engineering)","score_opus":0.021802416165553812,"score_gpt":0.31875607393008226,"score_spread":0.2969536577645284,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7131219091","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012197964,0.0003484673,0.9499396,0.00021203868,0.00011072076,0.00017869023,0.00024765826,0.034543794,0.0022210004],"genre_scores_gemma":[0.20652665,0.0003621312,0.7806501,0.00062096573,0.00014085899,0.0004796576,0.0012904559,0.0036745823,0.0062546735],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9927268,0.0015192108,0.00079546415,0.0012023041,0.003247858,0.00050832133],"domain_scores_gemma":[0.98925495,0.0024021817,0.0009330069,0.0054941196,0.0016066832,0.00030908952],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004552722,0.0012975693,0.0012884991,0.0026161205,0.0013252571,0.0026652871,0.006764082,0.00143729,0.005666482],"category_scores_gemma":[0.013793964,0.0013788122,0.0018487145,0.0019767908,0.0028873535,0.009559796,0.009062096,0.0035428489,0.0031374579],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0023466568,0.00037419624,0.006666085,0.0014315639,0.00023252949,0.00050555984,0.0009275146,0.034688946,0.0792303,0.2336034,0.02721797,0.6127754],"study_design_scores_gemma":[0.0004623892,0.000810393,0.0014223633,0.000384303,0.0003591714,0.00085235946,0.00018082415,0.57573575,0.14900905,0.19033591,0.08016536,0.00028213413],"about_ca_topic_score_codex":0.0024268692,"about_ca_topic_score_gemma":0.003613322,"teacher_disagreement_score":0.006764082,"about_ca_system_score_codex":0.0017477851,"about_ca_system_score_gemma":0.005124957,"threshold_uncertainty_score":0.024077356},"labels":[],"label_agreement":null},{"id":"W7131272134","doi":"10.1109/icoiics67115.2025.11390508","title":"TranslingoX: Real-Time Translation System for Indian Languages","year":2025,"lang":"","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Horizon College and Seminary","funders":"","keywords":"Machine translation; Speech translation; Rule-based machine translation; Computer-assisted translation; Representation (politics); Translation system; Translation (biology)","score_opus":0.011196135576642004,"score_gpt":0.293154682125705,"score_spread":0.281958546549063,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7131272134","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.097421266,0.0016526832,0.43198475,0.0007060832,0.001674909,0.0006610437,0.012581531,0.41925454,0.034063198],"genre_scores_gemma":[0.4977894,0.0011065023,0.39877942,0.00086027227,0.0002717087,0.00069060834,0.04516758,0.010662968,0.04467145],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9997317,0.00004194695,0.00003411974,0.00008867188,0.000056314217,0.000047160964],"domain_scores_gemma":[0.99973255,0.000035748548,0.000021974362,0.00006490998,0.00010748981,0.00003740418],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00049686834,0.0010906601,0.0005399681,0.00063491374,0.00054861465,0.0009543429,0.00095373247,0.00046095686,0.012625935],"category_scores_gemma":[0.0009053762,0.00028874405,0.0006141072,0.0005571924,0.0002503125,0.0013741212,0.0012918708,0.00093344273,0.008901286],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020746223,0.00042694336,0.0031696951,0.0011810684,0.00019907951,0.0020474899,0.001110674,0.0076859337,0.17309628,0.006607204,0.20207813,0.60032284],"study_design_scores_gemma":[0.00074053043,0.0014800598,0.0068695843,0.00021941663,0.00044159888,0.0027829707,0.0013968981,0.28206435,0.3533386,0.011246888,0.33893898,0.00048014568],"about_ca_topic_score_codex":0.0029908076,"about_ca_topic_score_gemma":0.004107637,"teacher_disagreement_score":0.012625935,"about_ca_system_score_codex":0.0004097803,"about_ca_system_score_gemma":0.0008930648,"threshold_uncertainty_score":0.042237937},"labels":[],"label_agreement":null},{"id":"W7131787357","doi":"10.63317/43zzdod4xaiy","title":"Phonological Transcription of the Canadian Dictionary of ASL as a Language Resource","year":2024,"lang":"","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Transcription (linguistics); Phonetic transcription; Resource (disambiguation); Phonology; Phonetics","score_opus":0.011683064245696533,"score_gpt":0.2570614534594101,"score_spread":0.24537838921371358,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7131787357","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.067118645,0.0022405933,0.051487185,0.005885307,0.0039635007,0.0006161391,0.08877277,0.0063835257,0.77353245],"genre_scores_gemma":[0.6546532,0.0027565013,0.07644179,0.00093976874,0.0003714843,0.00024234099,0.037948534,0.00313822,0.2235082],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9989146,0.00016575404,0.00008724044,0.00018060183,0.00048964925,0.00016217197],"domain_scores_gemma":[0.9967674,0.00024567248,0.00006226781,0.00021847854,0.002507099,0.00019911908],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005963456,0.00076567195,0.00036967595,0.00447447,0.0035341107,0.0038401023,0.0009527857,0.000520422,0.03464709],"category_scores_gemma":[0.0036117963,0.00033704584,0.00020131316,0.0056481557,0.0013959742,0.0010151191,0.00087690953,0.0018056278,0.008002745],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005186929,0.00009835494,0.0032391376,0.0007074759,0.000012678181,0.0006720943,0.00810917,0.0015407884,0.013400435,0.20493615,0.41272593,0.3540392],"study_design_scores_gemma":[0.000027217928,0.000033257937,0.0063050985,0.00021279161,0.000021818327,0.00034440646,0.0029873298,0.0026295825,0.0062256814,0.006359928,0.97475535,0.000097562195],"about_ca_topic_score_codex":0.8926479,"about_ca_topic_score_gemma":0.9459063,"teacher_disagreement_score":0.10735208,"about_ca_system_score_codex":0.014272788,"about_ca_system_score_gemma":0.04670629,"threshold_uncertainty_score":0.21596855},"labels":[],"label_agreement":null},{"id":"W7131955825","doi":"","title":"Evaluation briefs: drawing on translation studies for human evaluation of MT","year":2024,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Machine translation; Translation (biology); Context (archaeology); Set (abstract data type); Machine translation software usability; Evaluation of machine translation; Example-based machine translation","score_opus":0.1616407711959469,"score_gpt":0.44272982507029,"score_spread":0.28108905387434313,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7131955825","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01061882,0.15442276,0.5909013,0.16803318,0.0118972305,0.0054857302,0.0011172841,0.0014175152,0.056106124],"genre_scores_gemma":[0.22168684,0.062816195,0.6224651,0.051695287,0.0085441405,0.016617313,0.0014391846,0.0011400176,0.013596053],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.75900525,0.19604002,0.01896071,0.0040997555,0.020814111,0.00108007],"domain_scores_gemma":[0.41418496,0.461082,0.02863532,0.022899857,0.06912127,0.0040766136],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.25925162,0.0022677837,0.0016738073,0.010338993,0.0038733021,0.009131325,0.0041369037,0.007310715,0.007569962],"category_scores_gemma":[0.45063657,0.0014391334,0.0013945047,0.006458831,0.0066902107,0.028178954,0.006831777,0.0083718235,0.0026043504],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00090125983,0.00051167695,0.0021562104,0.010294575,0.0003604934,0.0004936867,0.011438781,0.0031905114,0.003178788,0.27869835,0.0916622,0.59711355],"study_design_scores_gemma":[0.00036878188,0.0033763717,0.0049855327,0.017626982,0.0006256083,0.001255898,0.008201176,0.0078403335,0.0065707783,0.26110741,0.68758607,0.00045501016],"about_ca_topic_score_codex":0.0039837435,"about_ca_topic_score_gemma":0.0077801794,"teacher_disagreement_score":0.25925162,"about_ca_system_score_codex":0.011725798,"about_ca_system_score_gemma":0.0063045914,"threshold_uncertainty_score":0.9134746},"labels":[],"label_agreement":null},{"id":"W7131958448","doi":"","title":"Some tradeoffs in continual learning for Parliamentary neural machine translation systems","year":2024,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Machine translation; Translation (biology); Simple (philosophy); Artificial neural network; Training (meteorology)","score_opus":0.017842707167898816,"score_gpt":0.27627739885793656,"score_spread":0.25843469169003774,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7131958448","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.42315432,0.002625454,0.5465523,0.014042078,0.00014568889,0.00020077454,0.00012263298,0.0020015314,0.011155307],"genre_scores_gemma":[0.91741574,0.00022563481,0.07931741,0.000378728,0.00010125283,0.00014462817,0.000120430166,0.00022461028,0.0020715746],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9922671,0.0042161453,0.0005098332,0.0011888486,0.0012922909,0.00052586384],"domain_scores_gemma":[0.90519965,0.07622524,0.0027714046,0.009156461,0.0051581482,0.0014890955],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03116213,0.0007648213,0.0012807357,0.00081465964,0.001972884,0.0031942786,0.0043108403,0.002660695,0.0038544112],"category_scores_gemma":[0.105862275,0.0011993565,0.00072698685,0.0010864504,0.0040576374,0.011863943,0.0048849117,0.0066804197,0.0008367164],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0040160525,0.0017748345,0.013690951,0.00046466908,0.00017899073,0.00043418165,0.0025047755,0.5208258,0.011198044,0.15193385,0.003909691,0.28906816],"study_design_scores_gemma":[0.000108322,0.00030668615,0.00075379986,0.00004205264,0.000025160614,0.000101649886,0.00022407112,0.8910199,0.0036444357,0.10280516,0.0009309295,0.000037933216],"about_ca_topic_score_codex":0.003186304,"about_ca_topic_score_gemma":0.0040366645,"teacher_disagreement_score":0.03116213,"about_ca_system_score_codex":0.00280071,"about_ca_system_score_gemma":0.0021464804,"threshold_uncertainty_score":0.16480303},"labels":[],"label_agreement":null},{"id":"W7131965811","doi":"","title":"WeBiText: building large heterogeneous translation memories from parallel Web content","year":2008,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":true,"route_about_ca":true,"ca_institutions":"National Research Council Canada","funders":"Université du Québec en Outaouais","keywords":"Field (mathematics); Parallel corpora; Machine translation; Government (linguistics); Translation (biology); Cover (algebra); Web page","score_opus":0.03832649372604355,"score_gpt":0.2624293158410198,"score_spread":0.22410282211497626,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7131965811","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09852117,0.0011022515,0.825707,0.0008291485,0.00026757622,0.00090323435,0.009919454,0.053548444,0.009201645],"genre_scores_gemma":[0.23239246,0.00064170593,0.7158431,0.0003332689,0.00013914653,0.0008539869,0.036304727,0.0047608446,0.008730671],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9986052,0.0004443125,0.00012814077,0.0004634894,0.00025711255,0.00010175687],"domain_scores_gemma":[0.99392843,0.0027137604,0.00035931636,0.0018649624,0.0009393704,0.00019406108],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024147434,0.0014457887,0.000899036,0.0029068333,0.0013120596,0.0022701782,0.002069089,0.00085481454,0.007093452],"category_scores_gemma":[0.011873999,0.00081584323,0.0012380589,0.0036789163,0.00093165686,0.0059146537,0.0029052414,0.0014893855,0.0057201576],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013061729,0.00044226553,0.005490927,0.0015677493,0.00048762542,0.0013761902,0.0029661441,0.028485818,0.044912603,0.011393464,0.049444694,0.85212636],"study_design_scores_gemma":[0.00047759456,0.0010321317,0.008667016,0.0003860662,0.0007598949,0.0020804717,0.0057752123,0.5148429,0.18476687,0.042572778,0.23833159,0.00030736908],"about_ca_topic_score_codex":0.007407858,"about_ca_topic_score_gemma":0.012900268,"teacher_disagreement_score":0.007407858,"about_ca_system_score_codex":0.0008539956,"about_ca_system_score_gemma":0.0019762528,"threshold_uncertainty_score":0.02372998},"labels":[],"label_agreement":null},{"id":"W7131974924","doi":"","title":"On the stability of system rankings at WMT","year":2021,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Annotation; Task (project management); Context (archaeology); Quality (philosophy); Stability (learning theory); Ranking (information retrieval); Outlier","score_opus":0.014723612855641628,"score_gpt":0.23811382758374872,"score_spread":0.22339021472810708,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7131974924","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6406881,0.0013953886,0.3320015,0.0060130158,0.00041798266,0.00034985266,0.003040064,0.002900036,0.013193937],"genre_scores_gemma":[0.9529908,0.00010237248,0.041615386,0.00037680005,0.0001210391,0.00026080944,0.002286738,0.00067029585,0.0015757763],"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9325464,0.037136126,0.002988085,0.012347139,0.012797339,0.0021849414],"domain_scores_gemma":[0.47325876,0.38770202,0.026076157,0.05950304,0.04896735,0.004492691],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.08150412,0.0007511564,0.0018886791,0.004659323,0.003997456,0.0064885216,0.0024642188,0.0024914867,0.0037682338],"category_scores_gemma":[0.43634456,0.0012569388,0.0010109214,0.0047480855,0.005575934,0.009444869,0.0046478403,0.0052059665,0.0021132615],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.006577061,0.0008787057,0.3361893,0.0012962968,0.0021552595,0.0010144188,0.016014155,0.18088982,0.043052696,0.08218542,0.02834154,0.30140534],"study_design_scores_gemma":[0.0002520392,0.0011204237,0.21795137,0.00019997529,0.00027092607,0.0006323823,0.0033686503,0.6462831,0.031850394,0.08890594,0.008677911,0.00048684888],"about_ca_topic_score_codex":0.013372878,"about_ca_topic_score_gemma":0.013578765,"teacher_disagreement_score":0.08150412,"about_ca_system_score_codex":0.004565067,"about_ca_system_score_gemma":0.0030365228,"threshold_uncertainty_score":0.43104017},"labels":[],"label_agreement":null},{"id":"W7131990533","doi":"","title":"Results of the WMT21 metrics shared task: evaluating metrics with expert-based human evaluations on TED and news domain","year":2021,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"Grantová Agentura České Republiky","keywords":"Robustness (evolution); Task (project management); Annotation; Domain (mathematical analysis); Task analysis; Quality (philosophy); Machine translation","score_opus":0.05253733671464236,"score_gpt":0.35468255224553946,"score_spread":0.3021452155308971,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7131990533","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8332656,0.0067651966,0.068474434,0.001749059,0.0020497662,0.0029200895,0.027108544,0.016955366,0.040711984],"genre_scores_gemma":[0.834196,0.00060752244,0.07572305,0.0008350801,0.0005829863,0.003356528,0.0630923,0.0036640074,0.017942581],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9505237,0.029008815,0.004593099,0.006185222,0.008165775,0.0015235014],"domain_scores_gemma":[0.91467184,0.041302845,0.004485306,0.013512494,0.021146087,0.0048815394],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.027759973,0.0047428426,0.0025362342,0.005205954,0.0018329882,0.003482931,0.0020152654,0.0036990477,0.0053382963],"category_scores_gemma":[0.08903466,0.00056512863,0.0019319087,0.0028171355,0.001751331,0.0044087823,0.005956547,0.0022590947,0.004952998],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.008299509,0.0057729958,0.04588085,0.0067384047,0.0030772858,0.0014270141,0.009889858,0.032979596,0.071520284,0.002845123,0.1860143,0.6255548],"study_design_scores_gemma":[0.004911943,0.020864552,0.34322548,0.0012744123,0.0019443292,0.0048109163,0.0104461545,0.25073963,0.13970092,0.012718964,0.20688324,0.0024794014],"about_ca_topic_score_codex":0.008300084,"about_ca_topic_score_gemma":0.010343053,"teacher_disagreement_score":0.97224003,"about_ca_system_score_codex":0.002041552,"about_ca_system_score_gemma":0.0018016599,"threshold_uncertainty_score":0.14681053},"labels":[],"label_agreement":null},{"id":"W7132025491","doi":"","title":"Terminology in neural machine translation: a case study of the canadian hansard","year":2023,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Terminology; Feature (linguistics); Machine translation; Artificial neural network","score_opus":0.035954949675231995,"score_gpt":0.29639209850994064,"score_spread":0.26043714883470864,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7132025491","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.830475,0.0035011058,0.019093832,0.014577075,0.00038538565,0.00042356233,0.004369743,0.000961658,0.12621261],"genre_scores_gemma":[0.9275852,0.0014450932,0.027272044,0.0010183477,0.000060988245,0.000057084595,0.0021259694,0.00028991167,0.040145382],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99767226,0.0008972834,0.00009398415,0.0002205417,0.0008408355,0.0002751542],"domain_scores_gemma":[0.9960277,0.0018581776,0.00018217134,0.00024888047,0.001481166,0.00020202949],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0021803696,0.00055386877,0.000430986,0.0015185644,0.008109418,0.002265017,0.0014612295,0.0017607698,0.004538819],"category_scores_gemma":[0.0108668385,0.00019562842,0.00030036512,0.0047884043,0.002919166,0.001463211,0.0012060008,0.0012245383,0.0007911435],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010389003,0.00049827155,0.03236294,0.0015167764,0.00012378275,0.0331511,0.11011279,0.028051104,0.016949892,0.11844784,0.14357589,0.51417065],"study_design_scores_gemma":[0.00018019279,0.00026170682,0.052931134,0.00036300937,0.00017475698,0.0076750503,0.096381135,0.035682037,0.02316512,0.01166424,0.7711725,0.00034908138],"about_ca_topic_score_codex":0.9292468,"about_ca_topic_score_gemma":0.96711695,"teacher_disagreement_score":0.07075322,"about_ca_system_score_codex":0.020694746,"about_ca_system_score_gemma":0.02382403,"threshold_uncertainty_score":0.15015161},"labels":[],"label_agreement":null},{"id":"W7132162559","doi":"","title":"Beyond correlation: making sense of the score differences of new MT evaluation metrics","year":2023,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Intuition; Metric (unit); Heuristics; Rule of thumb; BLEU; Meaning (existential)","score_opus":0.056855265097383194,"score_gpt":0.31818590765609955,"score_spread":0.26133064255871635,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7132162559","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.107622445,0.008260908,0.84906626,0.006664225,0.0018985885,0.00031320544,0.001354518,0.0022314535,0.022588443],"genre_scores_gemma":[0.76188797,0.0011885191,0.22831278,0.0019484342,0.0014906103,0.0005766263,0.0013246067,0.0012421125,0.0020283924],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.89351076,0.059577707,0.0072949585,0.0140495645,0.024245145,0.0013219015],"domain_scores_gemma":[0.52041423,0.33968708,0.031702403,0.069448665,0.036046885,0.0027007768],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.088924274,0.001992856,0.0027005703,0.009793767,0.002002687,0.01095997,0.0035570287,0.0026712613,0.0020132372],"category_scores_gemma":[0.46194932,0.0010123112,0.001168768,0.00982909,0.007062799,0.018371182,0.007140138,0.0076034334,0.0013969955],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015024841,0.00029380154,0.12817101,0.0016995881,0.0025052018,0.0004613353,0.005340059,0.029583562,0.008956239,0.23964384,0.018714512,0.56312835],"study_design_scores_gemma":[0.00017516171,0.001228564,0.060163856,0.00083949586,0.0005279792,0.0012199471,0.0015786444,0.15731621,0.014271629,0.7370454,0.02516017,0.00047288567],"about_ca_topic_score_codex":0.0020287973,"about_ca_topic_score_gemma":0.0025334358,"teacher_disagreement_score":0.9110757,"about_ca_system_score_codex":0.0024692523,"about_ca_system_score_gemma":0.002061568,"threshold_uncertainty_score":0.47028214},"labels":[],"label_agreement":null},{"id":"W7132265934","doi":"","title":"NRC-CNRC systems for Upper Sorbian-German and Lower Sorbian-German machine translation 2021","year":2021,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"German; Machine translation; Task (project management); Translation (biology); Training set; Transfer of learning; Resource (disambiguation)","score_opus":0.020705350725108547,"score_gpt":0.2860940658161212,"score_spread":0.26538871509101264,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7132265934","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08178517,0.0038728516,0.5292244,0.0022951248,0.0024330544,0.0017228292,0.036757316,0.26066428,0.081244946],"genre_scores_gemma":[0.20591071,0.00068195123,0.6142698,0.000917594,0.00026958156,0.0011983681,0.12905821,0.008474513,0.03921933],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9963546,0.0009295777,0.00032207154,0.0011209912,0.00088417565,0.00038859106],"domain_scores_gemma":[0.99254805,0.0010746375,0.0002684294,0.002298289,0.0034801462,0.00033043147],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0054389453,0.002302499,0.0013754697,0.002586018,0.0019001875,0.002499292,0.0030490102,0.0021705653,0.03220043],"category_scores_gemma":[0.012181131,0.00076864514,0.001041195,0.002175553,0.00069045037,0.0031626022,0.003004219,0.0023535194,0.034380823],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010221602,0.00043642957,0.0022063863,0.0010019923,0.00022966746,0.00056886306,0.0006363981,0.018538876,0.03682437,0.01771732,0.2791928,0.64162475],"study_design_scores_gemma":[0.00050917105,0.0009324055,0.00718248,0.000353705,0.0002466806,0.0015392565,0.0005723841,0.47722188,0.13352467,0.016240066,0.36132368,0.00035353334],"about_ca_topic_score_codex":0.021861492,"about_ca_topic_score_gemma":0.032320417,"teacher_disagreement_score":0.9781385,"about_ca_system_score_codex":0.00241908,"about_ca_system_score_gemma":0.0038683163,"threshold_uncertainty_score":0.10772115},"labels":[],"label_agreement":null},{"id":"W7132332393","doi":"","title":"Refining an almost clean translation memory helps machine translation","year":2022,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Machine translation; Translation (biology); Focus (optics); Annotation; Contrast (vision); Refining (metallurgy); Example-based machine translation; Transfer-based machine translation","score_opus":0.029091596957341644,"score_gpt":0.2782647380790894,"score_spread":0.24917314112174777,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7132332393","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.18935956,0.0024692789,0.74503154,0.0023121359,0.0005133515,0.00029787023,0.0023398849,0.04638066,0.011295666],"genre_scores_gemma":[0.48340443,0.00090486516,0.48992515,0.0015132742,0.00028378156,0.00029789456,0.0068581775,0.006529768,0.010282789],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9973992,0.00090787234,0.00025562357,0.000749053,0.00046992523,0.0002183851],"domain_scores_gemma":[0.9906551,0.003795901,0.0006336005,0.0032952544,0.0014570707,0.00016310101],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003116183,0.0016664045,0.0016494853,0.00189473,0.0020208238,0.0026829487,0.0017522541,0.0016354993,0.011022981],"category_scores_gemma":[0.017519591,0.0009204705,0.0009777318,0.003020039,0.0012925953,0.0050941743,0.0030454902,0.002568761,0.009572513],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011626375,0.0004899737,0.0042391624,0.00085576234,0.00021404303,0.0005418153,0.0011530046,0.018096924,0.103845246,0.0063105044,0.034345042,0.82874596],"study_design_scores_gemma":[0.0004830946,0.001408797,0.010553441,0.00037681326,0.00077633065,0.0031010758,0.0022598966,0.29026338,0.47830293,0.0470126,0.16516697,0.00029459334],"about_ca_topic_score_codex":0.0025760846,"about_ca_topic_score_gemma":0.006478042,"teacher_disagreement_score":0.011022981,"about_ca_system_score_codex":0.0006680195,"about_ca_system_score_gemma":0.0026768912,"threshold_uncertainty_score":0.036875546},"labels":[],"label_agreement":null},{"id":"W7132432999","doi":"","title":"Translation memories as baselines for low-resource machine translation","year":2022,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Machine translation; Example-based machine translation; Translation (biology); Machine translation software usability; Baseline (sea); Benchmark (surveying); Computer-assisted translation","score_opus":0.017312355936919443,"score_gpt":0.27222515508510203,"score_spread":0.2549127991481826,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7132432999","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.33782998,0.01352779,0.5720881,0.0023304967,0.0019409475,0.0011834663,0.008463682,0.019171461,0.043464176],"genre_scores_gemma":[0.7716276,0.0009249676,0.20650199,0.00057833,0.00038166338,0.0013698278,0.010936962,0.0017698872,0.0059087793],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99005604,0.0057607507,0.000878322,0.0015823228,0.0014029931,0.0003196692],"domain_scores_gemma":[0.9730304,0.014552867,0.0017281487,0.006887208,0.003390499,0.00041086884],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010376022,0.0010761248,0.0010929346,0.003137225,0.0016519411,0.0032069767,0.002511387,0.0019576,0.0075852345],"category_scores_gemma":[0.0542002,0.0005820165,0.0006713286,0.0034703163,0.0014835912,0.007532658,0.003037926,0.0024104351,0.004080087],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0060398453,0.0013153208,0.00948375,0.002191819,0.0009396912,0.00043836565,0.0014454725,0.102144785,0.034693774,0.05752328,0.03820707,0.7455768],"study_design_scores_gemma":[0.0010842847,0.004651052,0.014208009,0.0006732747,0.0010047887,0.0009081903,0.0016802365,0.55437267,0.14067408,0.19344416,0.0869249,0.00037431656],"about_ca_topic_score_codex":0.0014998206,"about_ca_topic_score_gemma":0.002796917,"teacher_disagreement_score":0.010376022,"about_ca_system_score_codex":0.0012711303,"about_ca_system_score_gemma":0.0010141946,"threshold_uncertainty_score":0.0548743},"labels":[],"label_agreement":null},{"id":"W7132436408","doi":"","title":"ᓄᑖᑦ ᓄᐊᑕᐅᓯᒪᔪᑦ ᑎᑎᕋᖅᑕᐅᓯᒪᔪᓂᑦ ᓄᓇᕗᒻᒥ ᒐᕙᒪᖓᑦᑕ ᐱᓕᕆᐊᖏᓐᓂᙶᖅᑐᑦ, ᐃᓄᒃᑎᑑᖅᑐᑦ ᖃᓪᓗᓈᑎᑑᓕᖅᓯᒪᔪᓂᒃ ᑲᑎᙵᓪᓗᑎᒃ - ᐊᒻᒪ ᖃᕋᓴᐅᔭᒃᑯᑦ ᑐᑭᓕᐅᖅᑕᐅᓯᒪᔪᑦ ᐃᓄᒃᑐᑦ ᖃᓪᓗᓈᑐᓪᓗ ᖃᓄᐃᒻᒪᖔᑕ ᓇᓗᓇᐃᖅᓯᔾᔪᑕᐅᓪᓗᓂ","year":2020,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Indigenous; Indigenous language; Machine translation; Sentence; First language; Second language","score_opus":0.0175971438060229,"score_gpt":0.2559683282143584,"score_spread":0.2383711844083355,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7132436408","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.72049224,0.0033154413,0.033910986,0.0014396422,0.0010394731,0.0012827702,0.09069844,0.0023873078,0.14543377],"genre_scores_gemma":[0.74632275,0.0011586775,0.06485623,0.00032936552,0.0001234798,0.0011886465,0.14181504,0.0010491555,0.043156642],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9994791,0.000114661874,0.000054475633,0.00018279284,0.00010681314,0.0000622136],"domain_scores_gemma":[0.9992224,0.00020639997,0.00007110837,0.00008271087,0.00036249307,0.000054859065],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00041936315,0.00043667227,0.00027881723,0.0011766404,0.0027234165,0.0012488175,0.00047110903,0.0002840217,0.0088581],"category_scores_gemma":[0.0017754423,0.00020925127,0.00016897082,0.0023750993,0.00083444716,0.0007474886,0.00076688634,0.0007557033,0.0031957836],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016368764,0.0005996805,0.042412784,0.004036052,0.00010887965,0.0029719952,0.02813791,0.003072013,0.14248566,0.024852928,0.18976022,0.55992496],"study_design_scores_gemma":[0.00006909203,0.00025063514,0.15003744,0.00035797636,0.000092646624,0.0018695679,0.01514528,0.007847326,0.05720508,0.002747678,0.7642681,0.00010919063],"about_ca_topic_score_codex":0.06694722,"about_ca_topic_score_gemma":0.16706505,"teacher_disagreement_score":0.9330528,"about_ca_system_score_codex":0.0015769815,"about_ca_system_score_gemma":0.0028270576,"threshold_uncertainty_score":0.13311511},"labels":[],"label_agreement":null},{"id":"W7132621154","doi":"","title":"Big BiRD: a large, fine-grained, bigram relatedness dataset for examining semantic composition","year":2019,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"","funders":"","keywords":"Bigram; Semantic similarity; Semantics (computer science); Benchmark (surveying); Annotation; Scale (ratio); Meaning (existential)","score_opus":0.02076164190976727,"score_gpt":0.2784811732803737,"score_spread":0.25771953137060644,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7132621154","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.31161338,0.0034908452,0.092871815,0.0012159023,0.0005636924,0.0015547031,0.5470737,0.012438609,0.029177329],"genre_scores_gemma":[0.14957571,0.0004307102,0.15199938,0.00038063288,0.00009932466,0.0016188156,0.68868864,0.0006111887,0.0065956214],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99803036,0.00052633864,0.00029751554,0.00052944483,0.00048739102,0.00012887416],"domain_scores_gemma":[0.99516183,0.0017872937,0.000548232,0.0011588203,0.0009474085,0.00039648663],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014068553,0.000980876,0.0005571978,0.005146919,0.0017560249,0.0009431631,0.0012743494,0.0015986954,0.0052096075],"category_scores_gemma":[0.009725839,0.00031879087,0.00093302486,0.0042126765,0.0005708457,0.0030122332,0.002131646,0.0016553481,0.004554918],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018302585,0.0018546733,0.052764516,0.0074818823,0.0006241955,0.0018128238,0.0061848806,0.006569788,0.09127545,0.024220478,0.41003165,0.39534935],"study_design_scores_gemma":[0.0005729796,0.0012410126,0.19094057,0.00051171466,0.00026095432,0.0034969603,0.0046810145,0.051943284,0.036464583,0.023146506,0.6863325,0.00040787182],"about_ca_topic_score_codex":0.0072523165,"about_ca_topic_score_gemma":0.019129865,"teacher_disagreement_score":0.0072523165,"about_ca_system_score_codex":0.0009003094,"about_ca_system_score_gemma":0.0011862653,"threshold_uncertainty_score":0.017427802},"labels":[],"label_agreement":null},{"id":"W7132866158","doi":"","title":"MPBRQ - A Framework for Mixed-Precision Quantization for Large Language Models","year":2024,"lang":"","type":"dissertation","venue":"TSpace","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"University of Toronto","keywords":"Speedup; Quantization (signal processing); Kernel (algebra); Language model; Acceleration; Vector quantization","score_opus":0.025883225952077957,"score_gpt":0.3958974775084187,"score_spread":0.3700142515563407,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7132866158","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0011693769,0.0003036027,0.98047596,0.00019675857,0.00006337144,0.00006136281,0.00044140895,0.016502336,0.00078576495],"genre_scores_gemma":[0.07269886,0.00041054236,0.917072,0.00043896658,0.000108122615,0.0003692771,0.0020115848,0.0038644222,0.0030262363],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9975255,0.0006541254,0.00019219883,0.00054993184,0.00093123654,0.00014699381],"domain_scores_gemma":[0.9972011,0.0013633488,0.00015440221,0.0006648973,0.000522963,0.000093276656],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033564111,0.0015596546,0.0013933259,0.0012039572,0.00084060896,0.003291246,0.004822819,0.0021051702,0.012494811],"category_scores_gemma":[0.014064633,0.001370332,0.00222196,0.0013286474,0.0014535049,0.0053672483,0.0037288,0.0045583528,0.005858908],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005781594,0.00018630787,0.0012156193,0.00075646595,0.0003264586,0.00029952498,0.00059406675,0.2943867,0.019668499,0.14004332,0.051546212,0.4903986],"study_design_scores_gemma":[0.000048446826,0.000035025438,0.00007393292,0.000032370903,0.000016754244,0.000049965438,0.0000268344,0.9292638,0.0036327722,0.059570402,0.0072195027,0.000030301755],"about_ca_topic_score_codex":0.014410278,"about_ca_topic_score_gemma":0.026467107,"teacher_disagreement_score":0.014410278,"about_ca_system_score_codex":0.002145839,"about_ca_system_score_gemma":0.0030225813,"threshold_uncertainty_score":0.041799307},"labels":[],"label_agreement":null},{"id":"W7132897739","doi":"","title":"Automatically classifying English verb-particle constructions by particle semantics","year":2006,"lang":"","type":"dissertation","venue":"TSpace","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Bank of Canada; Library and Archives Canada","funders":"","keywords":"Focus (optics); Semantics (computer science); Feature (linguistics); Task (project management); Word (group theory); Range (aeronautics); Space (punctuation); Semantic property","score_opus":0.013071277994076327,"score_gpt":0.3159117891287109,"score_spread":0.30284051113463456,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7132897739","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.63127816,0.0005428221,0.35457656,0.00047098479,0.00016225176,0.00018534166,0.001444686,0.0038020161,0.0075371442],"genre_scores_gemma":[0.8655186,0.00016349296,0.1290307,0.00006487573,0.000051997784,0.00008569602,0.0032867312,0.00035686028,0.0014409922],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9988674,0.00028881247,0.00011251641,0.00033251243,0.00027961502,0.00011922266],"domain_scores_gemma":[0.99618965,0.0023828945,0.00039075717,0.000412617,0.0005019151,0.00012235789],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017100656,0.0008810558,0.0008620823,0.004050912,0.0007880146,0.0020972237,0.0009341977,0.0011455794,0.0023378537],"category_scores_gemma":[0.005257699,0.00047090295,0.0011721385,0.002304304,0.0008569044,0.0038837534,0.0013123233,0.0011689484,0.00097006845],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00085768214,0.00042586448,0.07567092,0.0006136006,0.00020704386,0.0011577434,0.0018851786,0.013924205,0.06663949,0.038661834,0.010642979,0.78931344],"study_design_scores_gemma":[0.00014681616,0.00025205457,0.063709386,0.00014014868,0.00025565422,0.0021315585,0.0029976228,0.7138013,0.06040559,0.13401067,0.022013381,0.00013591256],"about_ca_topic_score_codex":0.0021932349,"about_ca_topic_score_gemma":0.0028485698,"teacher_disagreement_score":0.004050912,"about_ca_system_score_codex":0.0006529743,"about_ca_system_score_gemma":0.0008476179,"threshold_uncertainty_score":0.009043813},"labels":[],"label_agreement":null},{"id":"W7132934590","doi":"","title":"Subcategorial Considerations in Statistical Categorial Parsing","year":2022,"lang":"","type":"dissertation","venue":"TSpace","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Combinatory categorial grammar; Categorial grammar; Parsing; Rule-based machine translation; Link grammar; Natural language; Statistical learning; Grammar","score_opus":0.02561535006652994,"score_gpt":0.3742374172771301,"score_spread":0.3486220672106002,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7132934590","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.016066026,0.0008016192,0.9681265,0.0034606394,0.00012995374,0.000034737237,0.000079843805,0.00034973279,0.010950866],"genre_scores_gemma":[0.6012472,0.0015429577,0.38651437,0.0020178608,0.0005400511,0.00020670323,0.00032007866,0.00062200887,0.00698874],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99682295,0.0014560736,0.0002200272,0.0006695301,0.0006298612,0.00020161142],"domain_scores_gemma":[0.98774105,0.00913092,0.00045463126,0.001674897,0.00080898637,0.00018961757],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0060055167,0.00063467596,0.0006860508,0.0016840461,0.0010272665,0.0038263826,0.001744829,0.0019608168,0.0037009267],"category_scores_gemma":[0.020818973,0.00077702274,0.0013345649,0.0011990148,0.005608592,0.010555756,0.003369556,0.005444821,0.0007752621],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000012775966,0.000008392294,0.00061000796,0.00004506379,0.000012332097,0.00010148921,0.00036622141,0.016516939,0.00079005404,0.95581836,0.0009326688,0.024785701],"study_design_scores_gemma":[0.00000158634,0.000005299482,0.00011681732,0.000018537448,0.0000054337975,0.000039970957,0.00004046542,0.04206115,0.0003560358,0.955201,0.0021457002,0.000007916019],"about_ca_topic_score_codex":0.0021879892,"about_ca_topic_score_gemma":0.003096822,"teacher_disagreement_score":0.0060055167,"about_ca_system_score_codex":0.0018699194,"about_ca_system_score_gemma":0.001495036,"threshold_uncertainty_score":0.031760573},"labels":[],"label_agreement":null},{"id":"W7133197112","doi":"10.52202/075280-3402","title":"Metis: Understanding and Enhancing In-Network Regular Expressions","year":2023,"lang":"","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Focus (optics); Set (abstract data type); Feature (linguistics); Equivalence (formal languages); Perspective (graphical)","score_opus":0.04106595006974889,"score_gpt":0.2937412266654791,"score_spread":0.2526752765957302,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7133197112","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03217476,0.00020308749,0.94399536,0.00052472285,0.00012947456,0.0001633272,0.0012338565,0.015953653,0.005621759],"genre_scores_gemma":[0.30094534,0.00034045338,0.68183386,0.00025899487,0.00008210715,0.00016382942,0.0030518728,0.0032420137,0.010081603],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99899846,0.00026898202,0.00007160629,0.00024313759,0.0003419558,0.000075899676],"domain_scores_gemma":[0.997948,0.0009862137,0.00021752805,0.00044957508,0.0003400264,0.00005861549],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011356989,0.00076222926,0.0004757736,0.0008907897,0.00056457194,0.0013196932,0.0013746787,0.00048896443,0.0039999112],"category_scores_gemma":[0.004736893,0.0005250756,0.0007704,0.0006116384,0.0007433872,0.0035940532,0.001487926,0.0013810212,0.0011547721],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00066585717,0.00036487007,0.007137919,0.0009895196,0.00014522154,0.00063394604,0.0022154467,0.0647584,0.05440165,0.30742964,0.041872464,0.5193851],"study_design_scores_gemma":[0.000045934587,0.00013891564,0.0009902469,0.00012188871,0.00011115475,0.00030086207,0.00054058916,0.6957839,0.0752926,0.16868918,0.057932984,0.000051811992],"about_ca_topic_score_codex":0.0031516938,"about_ca_topic_score_gemma":0.0062563694,"teacher_disagreement_score":0.0039999112,"about_ca_system_score_codex":0.00073632604,"about_ca_system_score_gemma":0.0008458364,"threshold_uncertainty_score":0.013381064},"labels":[],"label_agreement":null},{"id":"W7134200768","doi":"10.1109/bigdata66926.2025.11401766","title":"Towards a Graph-Based Agentic Workflow and Framework for Natural Language Directions","year":2025,"lang":"","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Workflow; Natural language; Natural (archaeology); Domain (mathematical analysis); Context (archaeology); Abstraction; Work (physics)","score_opus":0.010567281069264625,"score_gpt":0.30658472774354306,"score_spread":0.29601744667427843,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7134200768","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00072588073,0.000035606998,0.99486464,0.0003187023,0.000035600955,0.000106576226,0.0002143191,0.0021996226,0.0014991185],"genre_scores_gemma":[0.018701432,0.000105890766,0.9778291,0.00011479119,0.0000234854,0.0001658556,0.00066038367,0.0004750812,0.0019239864],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9975647,0.0008791248,0.00028988466,0.0005300398,0.0005666215,0.00016961571],"domain_scores_gemma":[0.9957936,0.0010076002,0.00023211144,0.0013749807,0.0011272616,0.00046447042],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0044208625,0.0009694397,0.0010424639,0.0035587277,0.0023925048,0.007102385,0.0048195086,0.0021078687,0.0071969014],"category_scores_gemma":[0.007736164,0.0012149736,0.003796026,0.0028695958,0.0029258218,0.007774957,0.005522682,0.0050631706,0.0035608353],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008603175,0.00014326099,0.00058467174,0.00021203901,0.000068220535,0.00019378478,0.00076593173,0.030907797,0.0026904957,0.89640194,0.008919526,0.0590264],"study_design_scores_gemma":[0.000038760063,0.000031841533,0.000120812176,0.00009812541,0.00007487718,0.00010672532,0.00029482887,0.2747295,0.0036190755,0.6641994,0.056635533,0.000050520368],"about_ca_topic_score_codex":0.025456954,"about_ca_topic_score_gemma":0.03348658,"teacher_disagreement_score":0.025456954,"about_ca_system_score_codex":0.0023646522,"about_ca_system_score_gemma":0.0060771448,"threshold_uncertainty_score":0.050617576},"labels":[],"label_agreement":null},{"id":"W7135442803","doi":"","title":"LEPOR:A Robust Evaluation Metric for Machine Translation with Augmented Factors","year":2012,"lang":"en","type":"article","venue":"Research Explorer (The University of Manchester)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Open Text (Canada)","funders":"Universidade de Macau","keywords":"Metric (unit); Machine translation; Correlation; Translation (biology); Robustness (evolution)","score_opus":0.19273656479364318,"score_gpt":0.3430886210722153,"score_spread":0.15035205627857215,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7135442803","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.042312503,0.005793117,0.93277407,0.00033301496,0.0002675488,0.0006324626,0.0044636056,0.008154367,0.0052692657],"genre_scores_gemma":[0.34825712,0.0010252164,0.6378902,0.0001716839,0.00023363555,0.0009724998,0.00750102,0.0012329841,0.0027155248],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9828467,0.008047259,0.0016640798,0.001916688,0.0051152543,0.00040997186],"domain_scores_gemma":[0.98146296,0.009146224,0.0020203213,0.0027530596,0.004282612,0.0003347793],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010072094,0.0025566102,0.0021284127,0.0062184744,0.00077213265,0.0022406601,0.0013772879,0.0016639592,0.0023127175],"category_scores_gemma":[0.03317814,0.00033977634,0.0012700568,0.00442351,0.0010521194,0.0043010623,0.0022392694,0.0010503748,0.001992539],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001968347,0.00021711027,0.01226776,0.0021906332,0.00091340346,0.00028413278,0.0003737406,0.047803685,0.044169795,0.010745514,0.014422436,0.8646435],"study_design_scores_gemma":[0.00027721823,0.0063719708,0.034725633,0.0005447084,0.0008602817,0.002494938,0.0005118301,0.74809366,0.08892086,0.04593554,0.07042313,0.0008401286],"about_ca_topic_score_codex":0.0024880925,"about_ca_topic_score_gemma":0.0031765744,"teacher_disagreement_score":0.010072094,"about_ca_system_score_codex":0.0012891847,"about_ca_system_score_gemma":0.0014658573,"threshold_uncertainty_score":0.053267002},"labels":[],"label_agreement":null},{"id":"W7135697395","doi":"","title":"Language-independent Model for Machine Translation Evaluation with Reinforced Factors","year":2013,"lang":"en","type":"article","venue":"Research Explorer (The University of Manchester)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Open Text (Canada)","funders":"Universidade de Macau","keywords":"Metric (unit); Machine translation; Translation (biology); Quality (philosophy); Language model; Work (physics)","score_opus":0.10086607864645425,"score_gpt":0.3227555285243708,"score_spread":0.22188944987791653,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7135697395","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018707003,0.000737613,0.9762947,0.00019971165,0.00007670284,0.00020023095,0.00026193782,0.001719156,0.0018028668],"genre_scores_gemma":[0.6061304,0.0006782611,0.37921003,0.00027713602,0.00022881512,0.0011770414,0.0016617096,0.00057401636,0.010062566],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9949109,0.0022663097,0.00036619313,0.001174035,0.0010091129,0.0002734699],"domain_scores_gemma":[0.9943879,0.0029753712,0.00045470608,0.00055287004,0.0015037699,0.00012528733],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006280708,0.0021398081,0.0018525841,0.0022135645,0.0006070594,0.0017324869,0.0019543811,0.0017029803,0.002982018],"category_scores_gemma":[0.014843337,0.0005998113,0.0017696004,0.0017659048,0.00102325,0.002779504,0.0015154817,0.0020842399,0.0021387527],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000811253,0.00028507048,0.004069851,0.00040079624,0.0004923827,0.00036534978,0.00035179825,0.62987494,0.012909607,0.024067229,0.006650679,0.31972092],"study_design_scores_gemma":[0.000015106717,0.00008327305,0.00044758982,0.000010540538,0.00003710549,0.00005976037,0.000008417967,0.992101,0.0009851198,0.005421173,0.00080997584,0.00002089235],"about_ca_topic_score_codex":0.007732102,"about_ca_topic_score_gemma":0.007683694,"teacher_disagreement_score":0.007732102,"about_ca_system_score_codex":0.0016092798,"about_ca_system_score_gemma":0.0017726471,"threshold_uncertainty_score":0.03321594},"labels":[],"label_agreement":null},{"id":"W7135699072","doi":"","title":"Detection of Verbal Multi-Word Expressions via Conditional Random Fields with Syntactic Dependency Features and Semantic Re-Ranking","year":2017,"lang":"","type":"article","venue":"Research Explorer (The University of Manchester)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Open Text (Canada)","funders":"","keywords":"Conditional random field; Dependency (UML); Semantics (computer science); Pattern recognition (psychology); Field (mathematics)","score_opus":0.05394565159256942,"score_gpt":0.3112105575763141,"score_spread":0.2572649059837447,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7135699072","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.32129252,0.0012798698,0.6498626,0.0010559729,0.00035293296,0.0003506466,0.0063970922,0.012238683,0.00716982],"genre_scores_gemma":[0.7577253,0.00031373906,0.2258994,0.00021838296,0.00023039938,0.00023060403,0.009414882,0.0010508756,0.0049163518],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9977952,0.00062479044,0.00017896041,0.0005619761,0.00056383805,0.0002751503],"domain_scores_gemma":[0.9939487,0.0035303608,0.0005515731,0.0005365889,0.0012280024,0.00020474866],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020879006,0.0012124811,0.001303425,0.005209491,0.0010324729,0.0016741308,0.0015687556,0.0014252973,0.0045481212],"category_scores_gemma":[0.0049576843,0.00046375373,0.0012905827,0.0026122276,0.00060023105,0.002671368,0.0015277446,0.0019839185,0.0029257094],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015113993,0.00063728204,0.01770418,0.0006312635,0.00023385747,0.00153381,0.00057748175,0.013033364,0.1394625,0.016602803,0.022579573,0.7854925],"study_design_scores_gemma":[0.000091518166,0.00023489099,0.014530609,0.00008376135,0.00022342558,0.0012090213,0.0004960527,0.8807902,0.057417776,0.035524067,0.009268257,0.00013044747],"about_ca_topic_score_codex":0.0051516304,"about_ca_topic_score_gemma":0.0069465386,"teacher_disagreement_score":0.005209491,"about_ca_system_score_codex":0.0008114537,"about_ca_system_score_gemma":0.0016644043,"threshold_uncertainty_score":0.015215039},"labels":[],"label_agreement":null},{"id":"W7135940324","doi":"","title":"LEPOR:An Augmented Machine Translation Evaluation Metric - MT Evaluation, Quality Estimation, and Multilingual Treebanks","year":2017,"lang":"en","type":"book","venue":"Research Explorer (The University of Manchester)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Open Text (Canada)","funders":"","keywords":"Machine translation; Metric (unit); Quality (philosophy); Translation (biology); Quality assessment","score_opus":0.22140880087400827,"score_gpt":0.4251339928660098,"score_spread":0.2037251919920015,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7135940324","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0145241795,0.008416149,0.8936912,0.0008242499,0.0009897085,0.00049738336,0.009016526,0.04581964,0.026221063],"genre_scores_gemma":[0.098965205,0.002426423,0.83386636,0.0004288122,0.00041737148,0.00081297586,0.027713448,0.008362554,0.027006906],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.98854095,0.003593088,0.00093611144,0.001168002,0.005447739,0.00031402337],"domain_scores_gemma":[0.9907368,0.0029895082,0.0005660549,0.001846699,0.0035678903,0.00029313262],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0067789652,0.0025003452,0.0025256395,0.0067774532,0.001246572,0.0045938333,0.002891274,0.0018361511,0.012065477],"category_scores_gemma":[0.018207196,0.00086471415,0.0013115834,0.0064206817,0.00081684196,0.007364337,0.0044921213,0.0020532256,0.008075231],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00030992218,0.000101738406,0.0010700834,0.00096191943,0.00019090257,0.0001378896,0.00020066819,0.010548792,0.010669907,0.019791652,0.094135284,0.8618814],"study_design_scores_gemma":[0.00023628204,0.0009536924,0.008479198,0.00092734455,0.00045091126,0.0021611657,0.00028575418,0.47718567,0.056835268,0.08970717,0.3623944,0.00038315103],"about_ca_topic_score_codex":0.0027470146,"about_ca_topic_score_gemma":0.0049197436,"teacher_disagreement_score":0.012065477,"about_ca_system_score_codex":0.001807541,"about_ca_system_score_gemma":0.002041345,"threshold_uncertainty_score":0.040363014},"labels":[],"label_agreement":null},{"id":"W7135967693","doi":"","title":"Proceedings of the 19th Workshop on Multiword Expressions (MWE 2023)","year":2023,"lang":"","type":"article","venue":"Research Explorer (The University of Manchester)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Open Text (Canada)","funders":"","keywords":"Feature (linguistics); Term (time); Semantics (computer science); Computational linguistics; Key (lock)","score_opus":0.10297009081148316,"score_gpt":0.3275999013826305,"score_spread":0.22462981057114734,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7135967693","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02571219,0.028418725,0.75817215,0.022638338,0.026379412,0.00090288074,0.014004635,0.019420536,0.10435103],"genre_scores_gemma":[0.06366476,0.012163355,0.5525546,0.006563005,0.00442415,0.0009525145,0.0510128,0.01843043,0.29023442],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99582267,0.0017833201,0.00036489716,0.0008879902,0.0008328899,0.00030818544],"domain_scores_gemma":[0.99317884,0.0028082235,0.00016327186,0.0015877694,0.001728978,0.0005329155],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006310718,0.0015551532,0.0026435142,0.0019188083,0.0014947853,0.0065909424,0.0027022003,0.002595065,0.09710177],"category_scores_gemma":[0.0117409155,0.0010593584,0.0019197472,0.0021632728,0.0012109223,0.009580194,0.006247618,0.0040597655,0.04598162],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007008892,0.00039840894,0.0005404443,0.00064312364,0.000109745255,0.0002730545,0.00081569486,0.0010980949,0.011356684,0.02665793,0.43538332,0.5220226],"study_design_scores_gemma":[0.00007114507,0.00009135896,0.0008701627,0.0003170795,0.00007682609,0.00042505783,0.00034826578,0.008642503,0.008080582,0.032255247,0.9487753,0.000046456367],"about_ca_topic_score_codex":0.002959364,"about_ca_topic_score_gemma":0.004305009,"teacher_disagreement_score":0.09710177,"about_ca_system_score_codex":0.0013398322,"about_ca_system_score_gemma":0.002337818,"threshold_uncertainty_score":0.3248378},"labels":[],"label_agreement":null},{"id":"W7137882908","doi":"10.18653/v1/2025.ijcnlp-long.22","title":"Ibom NLP: A Step Toward Inclusive Natural Language Processing for Nigeria’s Minority Languages","year":2025,"lang":"","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Institut de Valorisation des Données; Canada First Research Excellence Fund; Carnegie Corporation of New York","keywords":"Natural language; Natural (archaeology); Minority language; Process (computing); Language identification","score_opus":0.008953073104983781,"score_gpt":0.3223032931189689,"score_spread":0.3133502200139851,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7137882908","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.031622477,0.002146717,0.87911457,0.007361767,0.0011896861,0.0013092003,0.007830531,0.03235013,0.03707492],"genre_scores_gemma":[0.11505372,0.0014765972,0.847601,0.0011300493,0.00030563507,0.001110056,0.015896015,0.003960983,0.0134659475],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9988833,0.00036569382,0.0001443479,0.0002496292,0.0002674644,0.00008958312],"domain_scores_gemma":[0.99825305,0.0007467394,0.00008858034,0.0003593811,0.00041323746,0.00013894399],"candidate_categories":["open_science"],"consensus_categories":[],"category_scores_codex":[0.0025681416,0.00080393726,0.00076172646,0.0022830572,0.0026935928,0.0055816644,0.0010091297,0.0011498015,0.010000702],"category_scores_gemma":[0.005598955,0.000704103,0.00069882016,0.0013668344,0.0009009448,0.009221482,0.0067999787,0.002974467,0.007248568],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00048131705,0.00040319478,0.004222869,0.001265795,0.0001109559,0.0012168748,0.0060691414,0.0014723688,0.028016549,0.094611734,0.08886575,0.77326345],"study_design_scores_gemma":[0.00013561538,0.00011763224,0.0029253014,0.0007062358,0.00014819384,0.0011287305,0.006612669,0.061831746,0.038132336,0.15010197,0.7380443,0.00011520805],"about_ca_topic_score_codex":0.0036197484,"about_ca_topic_score_gemma":0.0060561644,"teacher_disagreement_score":0.9989909,"about_ca_system_score_codex":0.00052005734,"about_ca_system_score_gemma":0.0029019157,"threshold_uncertainty_score":0.03345567},"labels":[],"label_agreement":null},{"id":"W7138047150","doi":"10.18653/v1/2025.banglalp-1.14","title":"BLUCK: A Benchmark Dataset for Bengali Linguistic Understanding and Cultural Knowledge","year":2025,"lang":"","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Bengali; Benchmark (surveying); Cultural knowledge; Knowledge-based systems; Knowledge base; Computational linguistics","score_opus":0.05115238458861413,"score_gpt":0.3614718791674825,"score_spread":0.31031949457886837,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7138047150","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1162902,0.0052002114,0.018019224,0.002838109,0.0007098853,0.0015161162,0.7939955,0.022425879,0.03900479],"genre_scores_gemma":[0.065015525,0.0005242593,0.013322006,0.00047976416,0.00007805286,0.00077364023,0.9123678,0.0004935502,0.0069453768],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99758816,0.000772534,0.00029748128,0.0005362183,0.00049639895,0.00030931004],"domain_scores_gemma":[0.9965912,0.0009265666,0.00018552445,0.0008813688,0.0011229465,0.00029237074],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016803601,0.0027214393,0.00097171724,0.0059443116,0.0021104729,0.0022083162,0.0038217034,0.0022666869,0.015668627],"category_scores_gemma":[0.00923968,0.0003693614,0.0013848455,0.006107106,0.0009651051,0.0036519715,0.003330994,0.0019120896,0.013948536],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009274405,0.0006925696,0.021649424,0.0025546788,0.00035214445,0.00076195336,0.0028032572,0.0083631715,0.006722561,0.0058604307,0.75912774,0.19018468],"study_design_scores_gemma":[0.00025854344,0.0003016977,0.060102124,0.0006757769,0.0003044185,0.0011200181,0.007488113,0.05644781,0.015393766,0.008167408,0.8494508,0.00028952622],"about_ca_topic_score_codex":0.13102345,"about_ca_topic_score_gemma":0.15482041,"teacher_disagreement_score":0.13102345,"about_ca_system_score_codex":0.004258421,"about_ca_system_score_gemma":0.003067414,"threshold_uncertainty_score":0.26052165},"labels":[],"label_agreement":null},{"id":"W7138252887","doi":"10.18653/v1/2025.ijcnlp-long.123","title":"Interpreting the Effects of Quantization on LLMs","year":2025,"lang":"","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"Natural Sciences and Engineering Research Council of Canada; Alliance de recherche numérique du Canada; Research Nova Scotia","keywords":"Quantization (signal processing); Natural language; Natural (archaeology); Association (psychology); Computational linguistics","score_opus":0.00428799359950585,"score_gpt":0.27877290594235393,"score_spread":0.2744849123428481,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7138252887","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.35022625,0.0055687865,0.52945775,0.014716144,0.0019400779,0.00016082822,0.002085252,0.0060923873,0.08975248],"genre_scores_gemma":[0.96102077,0.00038083518,0.031751942,0.00038814804,0.00014537892,0.00002793621,0.0003271325,0.0007510132,0.00520676],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9974554,0.0010953832,0.00015721816,0.0003213898,0.0007637751,0.00020692623],"domain_scores_gemma":[0.9811127,0.011324902,0.0010186095,0.002789142,0.0034305544,0.00032398847],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026280158,0.0003876769,0.0004312001,0.0011979777,0.0012212407,0.0036193805,0.0011281181,0.0010081582,0.016006496],"category_scores_gemma":[0.03794388,0.0004577296,0.0004009135,0.0015455429,0.002121516,0.005891806,0.0027241146,0.0015552855,0.0013631001],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015490171,0.0001373536,0.012879854,0.00048831315,0.000088972556,0.0012448857,0.0036025932,0.038941387,0.041336503,0.60829,0.025847988,0.26559302],"study_design_scores_gemma":[0.00012495086,0.00018504904,0.009092489,0.00015489703,0.0001005909,0.00041009413,0.0023784207,0.17933537,0.03412047,0.7406591,0.03331813,0.00012038992],"about_ca_topic_score_codex":0.005097663,"about_ca_topic_score_gemma":0.0059534987,"teacher_disagreement_score":0.016006496,"about_ca_system_score_codex":0.0016712171,"about_ca_system_score_gemma":0.00075408653,"threshold_uncertainty_score":0.053547084},"labels":[],"label_agreement":null},{"id":"W7138376537","doi":"10.18653/v1/2025.ijcnlp-long.107","title":"Breaking Bad: Norms for Valence, Arousal, and Dominance for over 10k English Multiword Expressions","year":2025,"lang":"","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Dominance (genetics); Pragmatics; Focus (optics)","score_opus":0.009931889552590436,"score_gpt":0.3024462780681954,"score_spread":0.29251438851560496,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7138376537","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.61696833,0.002136478,0.29367584,0.0013798085,0.0012427665,0.0008006744,0.040406384,0.010063871,0.033325948],"genre_scores_gemma":[0.83571583,0.00031179923,0.12151469,0.00025050712,0.00013060211,0.0010399816,0.028936593,0.0017909451,0.010309028],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9967951,0.0009462956,0.0005162053,0.00067549414,0.00087588676,0.00019102747],"domain_scores_gemma":[0.9921234,0.0034073486,0.00078176614,0.0012597996,0.0020410293,0.00038661034],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00365466,0.000948305,0.00055422605,0.001214664,0.00093705027,0.0032896523,0.00092663686,0.0005394382,0.0076205973],"category_scores_gemma":[0.019147476,0.00035283595,0.0005098527,0.0010571051,0.00096475764,0.0028524976,0.0018532648,0.001069217,0.003055024],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.005642205,0.00038600096,0.09613236,0.0012226359,0.0003708724,0.00037612082,0.00801711,0.0049771518,0.034129802,0.036266096,0.07865113,0.73382866],"study_design_scores_gemma":[0.00053221156,0.0017665398,0.4332106,0.000710797,0.00034687127,0.0019509399,0.014855372,0.10572773,0.048099805,0.19598953,0.19592582,0.00088375155],"about_ca_topic_score_codex":0.00302054,"about_ca_topic_score_gemma":0.0072303675,"teacher_disagreement_score":0.0076205973,"about_ca_system_score_codex":0.0008611624,"about_ca_system_score_gemma":0.00055392605,"threshold_uncertainty_score":0.025493443},"labels":[],"label_agreement":null},{"id":"W7138861340","doi":"10.5281/zenodo.19111195","title":"Natural Language Processing for African Languages in Ghana: Challenges and Opportunities","year":2014,"lang":"en","type":"article","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Python (programming language); Machine translation; Process (computing); Focus (optics); Natural language; Language technology; Computational linguistics; Field (mathematics); Languages of Africa","score_opus":0.04390762642283109,"score_gpt":0.27234898096365756,"score_spread":0.22844135454082648,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7138861340","genre_codex":"empirical","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8878085,0.0043666735,0.058237802,0.02414947,0.00019539768,0.0004089659,0.0033454099,0.0007029256,0.020784814],"genre_scores_gemma":[0.92472583,0.002960837,0.06536953,0.0006576181,0.000059331633,0.00030160614,0.0021584553,0.00018150787,0.00358526],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.999113,0.00056196534,0.00006826398,0.00009111082,0.000081672464,0.000083994404],"domain_scores_gemma":[0.99489886,0.00366697,0.0004372018,0.00028489117,0.00056302047,0.00014918372],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002029247,0.00035468937,0.00034910464,0.001094373,0.0012933587,0.0012554121,0.00044590287,0.00045214215,0.005859614],"category_scores_gemma":[0.008292155,0.00024138948,0.0001792601,0.0026497394,0.0009394309,0.0032017694,0.0012022273,0.00073928654,0.0011077927],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009527361,0.00034416094,0.09138481,0.0034853164,0.000047566165,0.008857928,0.05363867,0.012141108,0.02651927,0.043344297,0.027194275,0.7320899],"study_design_scores_gemma":[0.00036425682,0.0003712827,0.15252687,0.004171041,0.00010632614,0.0073265075,0.25794587,0.06954083,0.03637046,0.079574734,0.3914693,0.00023254205],"about_ca_topic_score_codex":0.016232077,"about_ca_topic_score_gemma":0.029730761,"teacher_disagreement_score":0.016232077,"about_ca_system_score_codex":0.0013958805,"about_ca_system_score_gemma":0.003961924,"threshold_uncertainty_score":0.0322752},"labels":[],"label_agreement":null},{"id":"W7138994399","doi":"10.5281/zenodo.19111194","title":"Natural Language Processing for African Languages in Ghana: Challenges and Opportunities","year":2014,"lang":"en","type":"article","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Python (programming language); Machine translation; Process (computing); Focus (optics); Natural language; Language technology; Computational linguistics; Field (mathematics); Languages of Africa","score_opus":0.04390762642283109,"score_gpt":0.27234898096365756,"score_spread":0.22844135454082648,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7138994399","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8878085,0.0043666735,0.058237802,0.02414947,0.00019539768,0.0004089659,0.0033454099,0.0007029256,0.020784814],"genre_scores_gemma":[0.92472583,0.002960837,0.06536953,0.0006576181,0.000059331633,0.00030160614,0.0021584553,0.00018150787,0.00358526],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.999113,0.00056196534,0.00006826398,0.00009111082,0.000081672464,0.000083994404],"domain_scores_gemma":[0.99489886,0.00366697,0.0004372018,0.00028489117,0.00056302047,0.00014918372],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002029247,0.00035468937,0.00034910464,0.001094373,0.0012933587,0.0012554121,0.00044590287,0.00045214215,0.005859614],"category_scores_gemma":[0.008292155,0.00024138948,0.0001792601,0.0026497394,0.0009394309,0.0032017694,0.0012022273,0.00073928654,0.0011077927],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009527361,0.00034416094,0.09138481,0.0034853164,0.000047566165,0.008857928,0.05363867,0.012141108,0.02651927,0.043344297,0.027194275,0.7320899],"study_design_scores_gemma":[0.00036425682,0.0003712827,0.15252687,0.004171041,0.00010632614,0.0073265075,0.25794587,0.06954083,0.03637046,0.079574734,0.3914693,0.00023254205],"about_ca_topic_score_codex":0.016232077,"about_ca_topic_score_gemma":0.029730761,"teacher_disagreement_score":0.016232077,"about_ca_system_score_codex":0.0013958805,"about_ca_system_score_gemma":0.003961924,"threshold_uncertainty_score":0.0322752},"labels":[],"label_agreement":null},{"id":"W7139957846","doi":"10.1109/hipc66333.2025.00027","title":"IMBPS - Iterative MLP Blocks with Parameter Splits for Improving LLM Inference","year":2025,"lang":"","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Advanced Micro Devices (Canada)","funders":"","keywords":"Pattern recognition (psychology); Iterative method; Inference; Noise (video); Feature (linguistics)","score_opus":0.012807086463383221,"score_gpt":0.29699344845022924,"score_spread":0.284186361986846,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7139957846","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03059153,0.00025534802,0.959453,0.00026185028,0.00003611665,0.00007182277,0.00014850718,0.00663182,0.002550043],"genre_scores_gemma":[0.33636388,0.00017928272,0.6569142,0.00028170802,0.00005174734,0.00024635158,0.00076953415,0.00069583976,0.004497474],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99935466,0.00022301302,0.000042113836,0.00010355908,0.00018459429,0.00009207082],"domain_scores_gemma":[0.9985123,0.00069148466,0.00010836079,0.0003960724,0.00022142335,0.00007033053],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014036275,0.0011210932,0.00061988033,0.00047817332,0.0004026261,0.0009259285,0.0021090957,0.00080123486,0.0057595912],"category_scores_gemma":[0.0058131763,0.0005783739,0.0005595888,0.0005301625,0.00063553825,0.0028013824,0.0017659216,0.0016504728,0.0033261944],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00074291445,0.00032787112,0.0027876992,0.00021452009,0.00013232898,0.00018425523,0.00033826148,0.32934406,0.048232034,0.021953525,0.009909323,0.5858332],"study_design_scores_gemma":[0.000019978963,0.00004816177,0.00016886141,0.000008061206,0.000011520043,0.000023778452,0.000021007038,0.9805963,0.012515371,0.0052196863,0.0013607427,0.0000065628524],"about_ca_topic_score_codex":0.00540468,"about_ca_topic_score_gemma":0.010289083,"teacher_disagreement_score":0.0057595912,"about_ca_system_score_codex":0.0008074382,"about_ca_system_score_gemma":0.0021809777,"threshold_uncertainty_score":0.019267797},"labels":[],"label_agreement":null},{"id":"W7143345797","doi":"10.1109/icdsinc66221.2025.11448203","title":"From Syntax to Semantics: AI Assisted Computational Linguistics in the Era of Large Computational Language Models","year":2025,"lang":"","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Microsemi (Canada)","funders":"","keywords":"Computational linguistics; Computational model; Syntax; Computational complexity theory; Computational semantics; Natural language","score_opus":0.013943942670997998,"score_gpt":0.323490894681319,"score_spread":0.30954695201032095,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7143345797","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004936085,0.008101979,0.9652008,0.012531955,0.000438756,0.00008592437,0.00036729494,0.0016136629,0.0067235543],"genre_scores_gemma":[0.15154006,0.008928464,0.82731193,0.0034590194,0.0011237997,0.00037609975,0.0009679037,0.00087070797,0.005422041],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99643373,0.0020936227,0.00021161635,0.00051960256,0.00065524597,0.00008618604],"domain_scores_gemma":[0.9874464,0.008745895,0.00043926382,0.0020776254,0.0009594688,0.00033134458],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006898423,0.0010532752,0.0011172619,0.004080434,0.0010673571,0.0066876644,0.002382626,0.0016500776,0.0040369253],"category_scores_gemma":[0.02032768,0.0007670654,0.0015351307,0.0026590985,0.0067346315,0.025269253,0.005708513,0.004784318,0.0015421045],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011341948,0.000080053636,0.0019754712,0.0010318869,0.00015523806,0.00024889674,0.002142424,0.011491962,0.0022039772,0.70051545,0.017303081,0.26273805],"study_design_scores_gemma":[0.000010845256,0.000021866961,0.00025801072,0.00018222495,0.000019723208,0.000076682656,0.00030313022,0.04904449,0.0006323235,0.9212177,0.028203333,0.000029713354],"about_ca_topic_score_codex":0.0027343533,"about_ca_topic_score_gemma":0.0030308925,"teacher_disagreement_score":0.006898423,"about_ca_system_score_codex":0.0019709507,"about_ca_system_score_gemma":0.0029781868,"threshold_uncertainty_score":0.03648275},"labels":[],"label_agreement":null},{"id":"W7144705875","doi":"","title":"機械翻訳の可能性の分析 : Ontologyの必要性","year":2008,"lang":"ja","type":"article","venue":"Institutional Repositories DataBase (IRDB)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Machine translation; Ontology; Futures studies; Meaning (existential); Christian ministry; Machine translation software usability","score_opus":0.026763934650704417,"score_gpt":0.27737334870331004,"score_spread":0.2506094140526056,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7144705875","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.057256106,0.018412897,0.5101966,0.03370069,0.0029739754,0.000613985,0.0025599084,0.0015237728,0.37276196],"genre_scores_gemma":[0.4828735,0.027012967,0.4025269,0.004208734,0.0010708796,0.00065049046,0.004052714,0.00055332785,0.07705053],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9968336,0.0009446839,0.00044712867,0.00048271773,0.0010934514,0.00019836049],"domain_scores_gemma":[0.99797577,0.0007398989,0.00022570859,0.00037025404,0.00057603186,0.00011234139],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004579136,0.00036022792,0.0003584628,0.0022391693,0.0021677385,0.005617836,0.0006126851,0.001170699,0.00700107],"category_scores_gemma":[0.005103899,0.00042593168,0.0004033683,0.003098754,0.0076025655,0.010615342,0.0027467078,0.002459144,0.0040221326],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011370402,0.0000835145,0.0029408024,0.00048528772,0.00003629911,0.00068763713,0.010411643,0.0008154609,0.0056220917,0.66518587,0.026482064,0.28713557],"study_design_scores_gemma":[0.000018871277,0.0000586189,0.0035064123,0.00048767976,0.000034443583,0.0014362468,0.008346253,0.0028509076,0.0066184336,0.30381897,0.6727399,0.000083220715],"about_ca_topic_score_codex":0.0061241826,"about_ca_topic_score_gemma":0.006729141,"teacher_disagreement_score":0.00700107,"about_ca_system_score_codex":0.002852644,"about_ca_system_score_gemma":0.003817182,"threshold_uncertainty_score":0.024217129},"labels":[],"label_agreement":null},{"id":"W7145613160","doi":"","title":"〈Original Papers〉On the Approach to the Quarter System in \"English AI and BI\"","year":2022,"lang":"ja","type":"article","venue":"Institutional Repositories DataBase (IRDB)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Quarter (Canadian coin); Feature (linguistics); Identification (biology); Window (computing)","score_opus":0.013222073840006277,"score_gpt":0.2445449062868035,"score_spread":0.23132283244679722,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7145613160","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.010454293,0.013364976,0.4744077,0.11682175,0.02335331,0.0006142532,0.008222906,0.005541766,0.34721908],"genre_scores_gemma":[0.18256061,0.012594366,0.37090734,0.026772562,0.010477165,0.00066992774,0.018068234,0.005088186,0.3728616],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9944904,0.0024515055,0.0004746108,0.0007633275,0.0015239138,0.00029629463],"domain_scores_gemma":[0.99092454,0.0032648235,0.00024502937,0.0019734446,0.003061309,0.00053091295],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0065529193,0.00046314608,0.0004826744,0.0031498338,0.0022726755,0.0059977053,0.0022549436,0.0014450921,0.043443836],"category_scores_gemma":[0.01817735,0.00046255902,0.00065138744,0.0071277195,0.002316661,0.008844738,0.0026202481,0.0029233485,0.013387853],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011333527,0.00006496452,0.0006433304,0.00039695133,0.000058648544,0.00014621648,0.0007943002,0.0012441024,0.001201425,0.40218684,0.4391363,0.1540136],"study_design_scores_gemma":[0.000025213843,0.000025081077,0.0007358081,0.00010271313,0.00003711214,0.00014857504,0.0003281527,0.0045122053,0.001964354,0.13325667,0.85882604,0.000038055183],"about_ca_topic_score_codex":0.017894227,"about_ca_topic_score_gemma":0.018823942,"teacher_disagreement_score":0.043443836,"about_ca_system_score_codex":0.0027522512,"about_ca_system_score_gemma":0.0034312897,"threshold_uncertainty_score":0.14533406},"labels":[],"label_agreement":null},{"id":"W7155072931","doi":"10.1109/aixse64906.2025.00020","title":"A Configurable Trait-Based API Framework for Enhancing Large Language Model Output","year":2025,"lang":"","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Workplace Health, Safety and Compensation Commission","funders":"","keywords":"Language model; Modeling language; Key (lock); Component (thermodynamics); Data modeling","score_opus":0.014951487869177512,"score_gpt":0.31143392784672896,"score_spread":0.29648243997755147,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7155072931","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0074522286,0.00007779246,0.8889112,0.0002179113,0.00005721196,0.00016853516,0.0013000095,0.10034566,0.0014694593],"genre_scores_gemma":[0.24502662,0.00019212674,0.7084834,0.0005556334,0.00007820491,0.0007618413,0.00919311,0.030575708,0.0051333667],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99703205,0.0007913327,0.00045219576,0.0006944985,0.00087746466,0.00015230986],"domain_scores_gemma":[0.9920353,0.0027256575,0.00050552917,0.0032752808,0.001184242,0.0002739217],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036314977,0.0015697583,0.0006346134,0.0012738664,0.00053090183,0.0026471324,0.0021753884,0.0009339062,0.0064487536],"category_scores_gemma":[0.021765476,0.00088200445,0.0017992604,0.000874874,0.00076129206,0.005079391,0.0047651213,0.002795259,0.0045709275],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014567073,0.00076225505,0.019376561,0.0013920182,0.00043811716,0.0014335844,0.004044347,0.035566207,0.10778345,0.08954639,0.06692463,0.67127573],"study_design_scores_gemma":[0.00013962165,0.00033084565,0.004092041,0.00017858528,0.00021063066,0.0011581307,0.00037190493,0.69439036,0.09553086,0.08052293,0.122756004,0.00031800644],"about_ca_topic_score_codex":0.0015962138,"about_ca_topic_score_gemma":0.0023181515,"teacher_disagreement_score":0.0064487536,"about_ca_system_score_codex":0.0006304071,"about_ca_system_score_gemma":0.001182174,"threshold_uncertainty_score":0.021573186},"labels":[],"label_agreement":null},{"id":"W7160150040","doi":"10.1109/iccv51701.2025.01222","title":"MAVFlow: Preserving Paralinguistic Elements with Conditional Flow Matching for Zero-Shot AV2AV Multilingual Translation","year":2025,"lang":"","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Paralanguage; Matching (statistics); Translation (biology); Flow (mathematics); Feature (linguistics)","score_opus":0.030732551150078002,"score_gpt":0.3371557552435094,"score_spread":0.3064232040934314,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7160150040","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018207954,0.00024005795,0.93504673,0.00012335401,0.00023184331,0.000114912364,0.0015234693,0.037222546,0.0072891084],"genre_scores_gemma":[0.30390966,0.00022054522,0.6663171,0.00023141524,0.00012294627,0.00020035745,0.00696534,0.009677991,0.012354603],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99883384,0.00022296695,0.00010110915,0.00035748805,0.00032715814,0.00015745134],"domain_scores_gemma":[0.99903214,0.0002486907,0.000050175586,0.000399975,0.00022486667,0.000044163742],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008163063,0.0010988043,0.0008376964,0.0012985207,0.00092297216,0.0015882553,0.0016861768,0.0011206314,0.017485244],"category_scores_gemma":[0.002529969,0.0006155258,0.00095109554,0.0011051655,0.00081885926,0.0027213139,0.003382894,0.0015205548,0.0071173706],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009891547,0.00025709564,0.00088494323,0.0008068923,0.000106041036,0.00074503984,0.0008827769,0.011011506,0.081106305,0.063345924,0.037601877,0.80226254],"study_design_scores_gemma":[0.00019785755,0.00038999214,0.001270874,0.00017308387,0.00016553876,0.0010712063,0.0007518572,0.40217298,0.3154076,0.15048787,0.12769255,0.00021862396],"about_ca_topic_score_codex":0.005182476,"about_ca_topic_score_gemma":0.0074128704,"teacher_disagreement_score":0.017485244,"about_ca_system_score_codex":0.00049859384,"about_ca_system_score_gemma":0.0014070751,"threshold_uncertainty_score":0.058493912},"labels":[],"label_agreement":null},{"id":"W7161992278","doi":"10.82308/1860","title":"Speech based machine aided human translation for a document translation task","year":2012,"lang":"en","type":"dissertation","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Human being; Machine translation","score_opus":0.023335719929392134,"score_gpt":0.3260513608269938,"score_spread":0.30271564089760167,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7161992278","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13105242,0.00556538,0.76823574,0.0018694333,0.0021772964,0.00087262667,0.0056922813,0.058245797,0.02628899],"genre_scores_gemma":[0.47324574,0.002091392,0.44127935,0.0006414271,0.0006079351,0.00065399456,0.012877064,0.0011568762,0.067446224],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99921036,0.0002000351,0.000060236496,0.00028490485,0.00014422956,0.000100101664],"domain_scores_gemma":[0.9988845,0.00050219253,0.00005611779,0.0001546114,0.00034845242,0.00005420062],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009194915,0.0015076632,0.0011478323,0.0012069686,0.0010343322,0.0018102273,0.0010824694,0.0021575494,0.02274885],"category_scores_gemma":[0.0026380597,0.00041527065,0.0010645088,0.0013294966,0.0004229018,0.0013400894,0.00092698255,0.0013698648,0.01935492],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012918892,0.00027893033,0.00076523726,0.0007501547,0.00013884957,0.0009060133,0.00037804455,0.018317103,0.066953786,0.0027370718,0.036491908,0.87099105],"study_design_scores_gemma":[0.00027301465,0.00075742503,0.003662911,0.00014205546,0.0002409008,0.0013675184,0.00093999674,0.81016,0.11183616,0.008371798,0.06212493,0.00012330769],"about_ca_topic_score_codex":0.0062288977,"about_ca_topic_score_gemma":0.009039997,"teacher_disagreement_score":0.02274885,"about_ca_system_score_codex":0.00063747365,"about_ca_system_score_gemma":0.0018414004,"threshold_uncertainty_score":0.076102495},"labels":[],"label_agreement":null},{"id":"W7162782628","doi":"10.7202/1125415ar","title":"Association for Machine Translation in the Americas (2024): 16th conference of the association for machine translation in the Americas. Chicago: AMTA. Online: &lt;https://aclanthology.org/events/amta-2024/&gt;","year":2025,"lang":"fr","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Machine translation; Association (psychology); Translation (biology); Association rule learning; Example-based machine translation","score_opus":0.0574773935087738,"score_gpt":0.3365715244853372,"score_spread":0.27909413097656344,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7162782628","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0026922415,0.13586605,0.06563463,0.1531583,0.21423364,0.0014434146,0.07849298,0.020991225,0.32748756],"genre_scores_gemma":[0.016304145,0.058023747,0.08174304,0.024866732,0.039983403,0.0025019702,0.12721133,0.01820783,0.63115793],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9911074,0.0022852253,0.0010074275,0.0011752859,0.0034799315,0.0009446651],"domain_scores_gemma":[0.97920173,0.0029627008,0.0015532408,0.0020832077,0.010839932,0.0033592486],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012640092,0.002941407,0.0023552263,0.008402944,0.0038792365,0.016066335,0.0030441496,0.0059715123,0.23140207],"category_scores_gemma":[0.016003003,0.0017709621,0.0016873418,0.009723135,0.0024026108,0.015356551,0.008468915,0.007992095,0.24383135],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000046773694,0.000013245254,0.000097659584,0.00012612733,0.000010262252,0.000023981795,0.00003112862,0.00003682256,0.0002293914,0.001003338,0.97556096,0.022820164],"study_design_scores_gemma":[0.000019383282,0.000022576556,0.0006594449,0.00025935803,0.000014330864,0.00008426075,0.000093490606,0.00014924548,0.0006137035,0.0023630355,0.9956943,0.000026900936],"about_ca_topic_score_codex":0.0149600655,"about_ca_topic_score_gemma":0.022362769,"teacher_disagreement_score":0.23140207,"about_ca_system_score_codex":0.003696748,"about_ca_system_score_gemma":0.012316119,"threshold_uncertainty_score":0.77411705},"labels":[],"label_agreement":null},{"id":"W7163158916","doi":"10.1109/icsm64417.2025.11541536","title":"RAST 4.0: Restorable Arbitrary Style Transfer via Content Leakage Correction","year":2025,"lang":"","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Leakage (economics); Quality (philosophy); Reliability (semiconductor); Metrology","score_opus":0.019746289080262176,"score_gpt":0.25908038534801064,"score_spread":0.23933409626774846,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7163158916","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013584367,0.0004055504,0.96095836,0.00010531619,0.00022330976,0.00021614962,0.0002534092,0.01867701,0.005576542],"genre_scores_gemma":[0.14060897,0.00046232092,0.8420447,0.00030477365,0.00011017117,0.00020495187,0.0009716175,0.0037023916,0.011590012],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993468,0.000100170626,0.00004046329,0.00013157025,0.00032616447,0.00005471979],"domain_scores_gemma":[0.99900407,0.00020523697,0.00007902138,0.00043500512,0.00021625517,0.000060322123],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009981124,0.0011771758,0.0005488168,0.0012236031,0.00050883676,0.001425754,0.0018435303,0.00091555144,0.007764517],"category_scores_gemma":[0.0026828402,0.00045472733,0.0014930229,0.00057042635,0.0009946015,0.0014589757,0.0018694255,0.0015887521,0.0035516028],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00051168626,0.00023719878,0.0009954222,0.00072762155,0.00018404964,0.00050384173,0.00059956376,0.047699023,0.23102379,0.022556756,0.021399243,0.6735619],"study_design_scores_gemma":[0.00013028347,0.00048524066,0.0012230377,0.000104225575,0.00011952291,0.0015607076,0.00015380027,0.62143666,0.27254567,0.022549728,0.07951619,0.0001749932],"about_ca_topic_score_codex":0.0012588282,"about_ca_topic_score_gemma":0.002038502,"teacher_disagreement_score":0.007764517,"about_ca_system_score_codex":0.00044043327,"about_ca_system_score_gemma":0.00071209884,"threshold_uncertainty_score":0.02597493},"labels":[],"label_agreement":null},{"id":"W7164340957","doi":"10.1109/iciics67880.2026.11483479","title":"Intelligent Document Querying with LLMs and Speech Input: A Scalable Semantic Framework","year":2005,"lang":"","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Scalability; Key (lock); Semantics (computer science); Component (thermodynamics); Term (time); Ontology","score_opus":0.01053403847569881,"score_gpt":0.27135799202872884,"score_spread":0.26082395355303,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7164340957","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.027213467,0.00046281426,0.91473305,0.0004845247,0.00004350406,0.00016529887,0.001636025,0.051663417,0.0035978602],"genre_scores_gemma":[0.36884433,0.000321267,0.6221092,0.000265969,0.00009165033,0.00022352386,0.003587434,0.0009583749,0.0035983594],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99913675,0.00016832435,0.00008361468,0.00019310023,0.0003398592,0.000078416546],"domain_scores_gemma":[0.9990097,0.00034083996,0.0000641319,0.00027847398,0.0002525782,0.00005417553],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013168218,0.0007078427,0.0010670755,0.0018042739,0.00071704306,0.0025375434,0.0020124442,0.00095197634,0.0038721669],"category_scores_gemma":[0.0028981282,0.00029908292,0.0009030539,0.001609325,0.0007875001,0.0044884575,0.0015160425,0.00079213444,0.0017855631],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018459796,0.00038494458,0.002146809,0.00042921232,0.00012681977,0.0003209545,0.0006994525,0.07029455,0.05589425,0.04101505,0.030079149,0.7967628],"study_design_scores_gemma":[0.00009389981,0.000091835704,0.0005345882,0.000021271982,0.00005147361,0.000117922515,0.0002025555,0.9322874,0.030368874,0.023904229,0.012278907,0.000047037367],"about_ca_topic_score_codex":0.01367936,"about_ca_topic_score_gemma":0.013734523,"teacher_disagreement_score":0.01367936,"about_ca_system_score_codex":0.0010381215,"about_ca_system_score_gemma":0.0013811213,"threshold_uncertainty_score":0.027199447},"labels":[],"label_agreement":null},{"id":"W7164753396","doi":"10.26443/ijwpc.v12i2.595","title":"Machine Translation Use as a Purposeful Activity","year":2025,"lang":"en","type":"article","venue":"International Journal of Whole Person Care","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"","score_opus":0.01930758434692293,"score_gpt":0.3147265869226583,"score_spread":0.2954190025757354,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7164753396","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.051931076,0.0005465067,0.775806,0.0019445654,0.0008826997,0.0011616093,0.0032122955,0.08324381,0.08127148],"genre_scores_gemma":[0.39686948,0.00044309057,0.5162697,0.00097532227,0.0003287572,0.0005292197,0.006741283,0.011906473,0.06593665],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9949662,0.0026751366,0.00029080093,0.00075013103,0.001042532,0.00027513396],"domain_scores_gemma":[0.99025106,0.0036982987,0.00031434113,0.0041665086,0.0014050043,0.00016479575],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025342433,0.0013225664,0.0009542602,0.001909545,0.0015230378,0.003978689,0.0013455529,0.0016508233,0.04058043],"category_scores_gemma":[0.008745182,0.00074135367,0.0012627484,0.0016575743,0.0008497414,0.004647236,0.003127379,0.0012949352,0.028983569],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00070570415,0.00075258594,0.0033098056,0.0011129571,0.00014957071,0.0013694569,0.0034694427,0.0025189463,0.10125599,0.03146969,0.06095595,0.7929299],"study_design_scores_gemma":[0.00013137475,0.00062002026,0.005876614,0.00037458603,0.00023382902,0.0042493674,0.0026322654,0.09231925,0.38274285,0.0643176,0.4463093,0.00019291988],"about_ca_topic_score_codex":0.00073805783,"about_ca_topic_score_gemma":0.0010768967,"teacher_disagreement_score":0.04058043,"about_ca_system_score_codex":0.00046264305,"about_ca_system_score_gemma":0.0012506602,"threshold_uncertainty_score":0.13575506},"labels":[],"label_agreement":null},{"id":"W71776421","doi":"","title":"Casting Implicit Role Linking as an Anaphora Resolution Task","year":2012,"lang":"en","type":"article","venue":"RECERCAT (Consorci de Serveis Universitaris de Catalunya)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":39,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Coreference; Anaphora (linguistics); Computer science; Task (project management); Natural language processing; Heuristic; Resolution (logic); Artificial intelligence; Baseline (sea); Machine learning; Engineering","score_opus":0.012766195486713122,"score_gpt":0.25589942160458984,"score_spread":0.24313322611787672,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W71776421","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.056018546,0.0018174896,0.9019263,0.0044169417,0.00068443624,0.0002517498,0.00066174404,0.0032032416,0.031019527],"genre_scores_gemma":[0.4589304,0.0012169061,0.50194925,0.00065782527,0.0007308992,0.00018512165,0.0016925533,0.0013546845,0.033282306],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9944975,0.0026069514,0.00030834676,0.0010307005,0.0011046799,0.0004518494],"domain_scores_gemma":[0.9892079,0.006111936,0.00038256802,0.002954819,0.0010167412,0.00032606348],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0062433975,0.0009813957,0.0018190424,0.0020045328,0.0034385908,0.0073831975,0.004157363,0.0045539145,0.01631132],"category_scores_gemma":[0.020958683,0.0016430691,0.001919621,0.0031515486,0.0026403754,0.016283624,0.009164859,0.0052376958,0.0038322632],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011010467,0.0005138903,0.0013532474,0.0006547616,0.00013901401,0.0005893164,0.002323038,0.024491144,0.012239145,0.65032405,0.022136016,0.28413525],"study_design_scores_gemma":[0.00013690145,0.00009490821,0.0004217374,0.0001005408,0.00017543024,0.00030373232,0.0006223814,0.2571263,0.015078891,0.68480545,0.041058358,0.00007535201],"about_ca_topic_score_codex":0.0038817392,"about_ca_topic_score_gemma":0.0065132394,"teacher_disagreement_score":0.01631132,"about_ca_system_score_codex":0.0016182238,"about_ca_system_score_gemma":0.0024645058,"threshold_uncertainty_score":0.0545668},"labels":[],"label_agreement":null},{"id":"W72644278","doi":"10.1007/978-3-642-21640-4_33","title":"ONTECTAS: Bridging the Gap between Collaborative Tagging Systems and Structured Data","year":2011,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Bridging (networking); Ontology; Construct (python library); Semantic Web; Block (permutation group theory); Information retrieval; Set (abstract data type); Data mining; Programming language","score_opus":0.03863034186774695,"score_gpt":0.2814982791313524,"score_spread":0.24286793726360545,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W72644278","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0071708946,0.0012707774,0.96960056,0.0010042066,0.0005985513,0.00013146158,0.001098094,0.010764567,0.008360961],"genre_scores_gemma":[0.069975436,0.0018722541,0.8854583,0.0007031784,0.00040595827,0.00025113413,0.009539971,0.003806812,0.027986972],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9949751,0.0020320178,0.00044085164,0.0010060503,0.0013262474,0.0002197398],"domain_scores_gemma":[0.98564917,0.006711713,0.00048721462,0.0056900787,0.00093175174,0.0005299265],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0057650907,0.0009657683,0.0011933441,0.0025054358,0.0021200327,0.0068294993,0.0033374876,0.0024893687,0.010753307],"category_scores_gemma":[0.013179706,0.0008813668,0.0014386429,0.005253857,0.0020094898,0.017757496,0.009118279,0.0030057372,0.0059503447],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006542351,0.00029209015,0.0017596855,0.00078703713,0.00017352705,0.00043053547,0.0028292711,0.0040516616,0.012280332,0.18219061,0.05611012,0.7384409],"study_design_scores_gemma":[0.00009708228,0.00026951538,0.0016758908,0.00034658544,0.00019177567,0.0010281099,0.0013534952,0.092967995,0.02859631,0.4040123,0.46926594,0.00019505923],"about_ca_topic_score_codex":0.0029716515,"about_ca_topic_score_gemma":0.0047415122,"teacher_disagreement_score":0.010753307,"about_ca_system_score_codex":0.0009084793,"about_ca_system_score_gemma":0.0023835134,"threshold_uncertainty_score":0.03597337},"labels":[],"label_agreement":null},{"id":"W72985070","doi":"10.21437/interspeech.2009-407","title":"Online discriminative training for grapheme-to-phoneme conversion","year":2009,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":34,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Grapheme; Discriminative model; Computer science; Artificial intelligence; Context (archaeology); Sequence (biology); Training set; Dependency (UML); Context model; Segmentation; Speech recognition; Pattern recognition (psychology); Natural language processing; Machine learning; Object (grammar); Engineering","score_opus":0.03768294263616557,"score_gpt":0.32247931722436374,"score_spread":0.2847963745881982,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W72985070","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.026434766,0.0002490759,0.96408147,0.00009827043,0.000058349924,0.00004475645,0.00019431251,0.006556403,0.002282532],"genre_scores_gemma":[0.551043,0.00024456054,0.43991995,0.00025317553,0.00006340626,0.00016259703,0.0015155125,0.0004820537,0.006315732],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9997002,0.000072973235,0.000009227078,0.00012400198,0.00006035034,0.00003336994],"domain_scores_gemma":[0.99945253,0.00033205136,0.00003251298,0.00010380496,0.00005195341,0.000027179682],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00031975552,0.00067061535,0.0005173602,0.0004085931,0.00028692957,0.00032147087,0.0012179734,0.00056581234,0.0042656614],"category_scores_gemma":[0.001362171,0.00030829135,0.00030203754,0.00064658374,0.00037852547,0.0008305609,0.00069198717,0.0012047096,0.0016688418],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018162982,0.00020650777,0.00069104286,0.00011928057,0.000035139095,0.00016338057,0.00006416027,0.09299507,0.0574212,0.0025171044,0.0044450555,0.8411605],"study_design_scores_gemma":[0.000014622698,0.0000749258,0.00065652514,0.0000057268985,0.000012370278,0.00016193917,0.000019768853,0.9681588,0.025292825,0.0032260276,0.0023660425,0.0000104044175],"about_ca_topic_score_codex":0.0026545601,"about_ca_topic_score_gemma":0.006090905,"teacher_disagreement_score":0.0042656614,"about_ca_system_score_codex":0.0003197991,"about_ca_system_score_gemma":0.0004588866,"threshold_uncertainty_score":0.014270067},"labels":[],"label_agreement":null},{"id":"W73771754","doi":"","title":"Annotation of the Complex Terms in Multilingual Corpora.","year":2006,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Trois-Rivières","funders":"","keywords":"Rule-based machine translation; Computer science; Categorial grammar; Natural language processing; Multilingualism; Linguistics; Artificial intelligence; Combinatory categorial grammar; Information extraction; Generative grammar; Tree-adjoining grammar; Context-free grammar; Mildly context-sensitive grammar formalism; Phrase structure rules","score_opus":0.015876426531492897,"score_gpt":0.27226501982824763,"score_spread":0.25638859329675473,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W73771754","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06468855,0.010119755,0.7703413,0.0029664224,0.0012427045,0.0007511135,0.07129985,0.008151332,0.07043898],"genre_scores_gemma":[0.2067073,0.003910994,0.64836025,0.00042761216,0.00039415763,0.0005663611,0.110432394,0.002673629,0.026527347],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9979236,0.00055750273,0.00028661266,0.0005331109,0.00062070676,0.000078398305],"domain_scores_gemma":[0.9940147,0.0028332479,0.00052974734,0.0013203914,0.001168916,0.00013289705],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018416088,0.0007304292,0.0007064827,0.0070816474,0.0016519978,0.0025499675,0.00088140415,0.0008495358,0.01048212],"category_scores_gemma":[0.008326615,0.00044341222,0.0007630883,0.0071645523,0.00091573026,0.0031230873,0.002516147,0.0015046907,0.0076836613],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036486593,0.00019039432,0.009550703,0.004335703,0.00027159302,0.0024796096,0.006346947,0.0036015888,0.09831376,0.12767416,0.09545869,0.651412],"study_design_scores_gemma":[0.00003010743,0.000056615336,0.017810412,0.00060906477,0.00021486227,0.0024825905,0.001592134,0.021167805,0.039545435,0.0570761,0.859271,0.00014386154],"about_ca_topic_score_codex":0.003008359,"about_ca_topic_score_gemma":0.0073756045,"teacher_disagreement_score":0.01048212,"about_ca_system_score_codex":0.0009487488,"about_ca_system_score_gemma":0.0019761438,"threshold_uncertainty_score":0.035066247},"labels":[],"label_agreement":null},{"id":"W74649020","doi":"","title":"Efficient Parsing for Word Structure.","year":2001,"lang":"en","type":"article","venue":"NLPRS","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Parsing; Computer science; Natural language processing; Covert; Morpheme; Artificial intelligence; Word (group theory); Selection (genetic algorithm); Linguistics","score_opus":0.013491939885216,"score_gpt":0.27993489328081017,"score_spread":0.26644295339559415,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W74649020","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0012969563,0.00021521124,0.9869398,0.00022360252,0.00006824567,0.00006544595,0.00040192588,0.0058315275,0.0049573276],"genre_scores_gemma":[0.07692629,0.0006057006,0.9069836,0.0003876949,0.0002244446,0.00022263722,0.0028292476,0.0025911867,0.009229143],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99833125,0.00051986566,0.00014424692,0.0003421701,0.0005330369,0.00012941757],"domain_scores_gemma":[0.99781346,0.0010024161,0.000119186974,0.0006322007,0.00039276434,0.000039884162],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018987688,0.00092925923,0.0007393193,0.001368467,0.001011155,0.0024645622,0.0016807329,0.0010178316,0.015353212],"category_scores_gemma":[0.004577035,0.0008621961,0.0014059205,0.0011048575,0.0012199269,0.005157597,0.002524561,0.0016911671,0.012300034],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000080729806,0.000053130185,0.0006156477,0.00053295086,0.00007145283,0.00033901518,0.0005464825,0.0062380424,0.023223374,0.61960185,0.041885853,0.30681145],"study_design_scores_gemma":[0.000024034503,0.000037587735,0.00054381654,0.000089577065,0.000061542116,0.00051843823,0.00011922584,0.08821229,0.019167643,0.78613055,0.10504054,0.00005472385],"about_ca_topic_score_codex":0.00074907055,"about_ca_topic_score_gemma":0.0013320069,"teacher_disagreement_score":0.015353212,"about_ca_system_score_codex":0.0007516315,"about_ca_system_score_gemma":0.0010648232,"threshold_uncertainty_score":0.05136156},"labels":[],"label_agreement":null},{"id":"W74822525","doi":"","title":"Automatic translation of formal data specifications to voice data-input applications.","year":2006,"lang":"en","type":"article","venue":"Scholarship at UWindsor (University of Windsor)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Translation (biology); Speech recognition; Natural language processing; Programming language; Artificial intelligence","score_opus":0.06490298255985248,"score_gpt":0.2763369195410419,"score_spread":0.2114339369811894,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W74822525","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.003416612,0.00009003822,0.9898164,0.00014698372,0.00008429537,0.0001904808,0.00032378803,0.0042757248,0.0016556972],"genre_scores_gemma":[0.070361346,0.00035284917,0.92002255,0.00028130622,0.00005186287,0.00040753704,0.0022132809,0.0019453457,0.004364069],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9961659,0.0011671213,0.00043243493,0.00058145606,0.0014479196,0.00020528545],"domain_scores_gemma":[0.9926192,0.004407252,0.00038059993,0.0011265738,0.0013861931,0.00008015768],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003263574,0.00094854535,0.00069178804,0.0011049886,0.0007700569,0.002597443,0.0015243595,0.0010436146,0.0058640265],"category_scores_gemma":[0.012296725,0.0009554792,0.0015414541,0.0006804356,0.0011384073,0.0021539326,0.0024779178,0.0017891404,0.002916679],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00039130397,0.00033896364,0.0019112091,0.0019071044,0.00018438978,0.0015126212,0.0032028086,0.048369005,0.07882618,0.25588456,0.024094352,0.58337754],"study_design_scores_gemma":[0.00015873519,0.00019420008,0.0005720255,0.0005396263,0.00014064895,0.0013836438,0.001023832,0.3552477,0.22290502,0.18921223,0.2284767,0.00014571036],"about_ca_topic_score_codex":0.0013279414,"about_ca_topic_score_gemma":0.0014172595,"teacher_disagreement_score":0.0058640265,"about_ca_system_score_codex":0.0009925687,"about_ca_system_score_gemma":0.0024844669,"threshold_uncertainty_score":0.01961714},"labels":[],"label_agreement":null},{"id":"W75593468","doi":"10.1007/978-1-4419-9863-7_100988","title":"Natural Language Understanding","year":2013,"lang":"en","type":"book-chapter","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":50,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Natural (archaeology); Linguistics; Natural language processing; Computer science; Geography; Philosophy; Archaeology","score_opus":0.025995609444127543,"score_gpt":0.2564845495278901,"score_spread":0.23048894008376258,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W75593468","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0014648074,0.015698863,0.2733068,0.0031863584,0.0011743755,0.000185036,0.001417267,0.0032380614,0.70032847],"genre_scores_gemma":[0.02352667,0.021740453,0.1572214,0.0014873291,0.0008792332,0.00030109385,0.0077684526,0.002086577,0.7849887],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99963295,0.00006072129,0.000024305491,0.00010535152,0.00015424768,0.00002242657],"domain_scores_gemma":[0.9996778,0.00014639267,0.000010772605,0.00007399982,0.00007844547,0.000012699781],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00038656406,0.0010541416,0.0006371848,0.0016911493,0.00086842023,0.003047106,0.0012505644,0.0007656005,0.06569211],"category_scores_gemma":[0.0013775823,0.0005299607,0.0007810058,0.0016583445,0.0012174138,0.005772001,0.001550395,0.0021749227,0.033841345],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000012464972,0.000031869,0.00006473385,0.0004196236,0.000011424613,0.00006596619,0.0004691725,0.0005707561,0.0018563513,0.23269667,0.23099263,0.5328083],"study_design_scores_gemma":[0.0000028729633,0.0000055687046,0.00010070557,0.00019102269,0.000008464394,0.00020475808,0.00009517197,0.0012694818,0.0012932294,0.13518791,0.86163116,0.000009689144],"about_ca_topic_score_codex":0.0018739126,"about_ca_topic_score_gemma":0.0028020246,"teacher_disagreement_score":0.06569211,"about_ca_system_score_codex":0.0010046518,"about_ca_system_score_gemma":0.0012472983,"threshold_uncertainty_score":0.21976203},"labels":[],"label_agreement":null},{"id":"W761563068","doi":"10.63317/5jhson5fhahj","title":"MISTRAL: a Statistical Machine Translation Decoder for Speech Recognition Lattices","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Machine translation; Computer science; Phrase; Natural language processing; Speech recognition; Artificial intelligence; Speech translation; Example-based machine translation; Translation (biology); Word (group theory); Rule-based machine translation; Transfer-based machine translation; Machine translation software usability; Language model; Linguistics","score_opus":0.05867933553657155,"score_gpt":0.31983480875084963,"score_spread":0.26115547321427807,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W761563068","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007668347,0.00034107326,0.90709776,0.00025595588,0.00037719973,0.00011532545,0.0040593343,0.07464647,0.0054384503],"genre_scores_gemma":[0.14453723,0.0002738796,0.82068276,0.00043367821,0.00025551525,0.00038697876,0.011660682,0.009399468,0.01236984],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9988286,0.00034362258,0.00010977885,0.00026637173,0.0003482917,0.00010340235],"domain_scores_gemma":[0.9977126,0.0008906229,0.000097303644,0.00045976165,0.0007284743,0.00011127632],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010683256,0.0011634362,0.0010987577,0.0013274173,0.0007514825,0.0018723363,0.0016036219,0.0013170494,0.018784396],"category_scores_gemma":[0.0054902923,0.00085228233,0.0008002041,0.0010254951,0.0005504143,0.0015305819,0.0017720339,0.0016194476,0.016077464],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019390995,0.00025563085,0.0013811787,0.00074011146,0.00018906737,0.0006916298,0.00034296425,0.02833233,0.07413831,0.04744985,0.15191412,0.69262564],"study_design_scores_gemma":[0.00038248696,0.00040891572,0.00062211516,0.000105307576,0.00010354964,0.0007722575,0.00019340786,0.72896737,0.13687022,0.047784403,0.08365286,0.0001370216],"about_ca_topic_score_codex":0.0022782127,"about_ca_topic_score_gemma":0.006212021,"teacher_disagreement_score":0.018784396,"about_ca_system_score_codex":0.0006622832,"about_ca_system_score_gemma":0.0022134536,"threshold_uncertainty_score":0.062840104},"labels":[],"label_agreement":null},{"id":"W782030620","doi":"10.63317/2464o68iyiio","title":"A review corpus annotated for negation, speculation and their scope","year":2012,"lang":"en","type":"review","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":85,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Scope (computer science); Computer science; Speculation; Negation; Natural language processing; Sentence; Consistency (knowledge bases); Artificial intelligence; Security token; Product (mathematics); Resource (disambiguation); Information retrieval; Programming language","score_opus":0.06136005070524941,"score_gpt":0.3528080662813044,"score_spread":0.291448015576055,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W782030620","genre_codex":"dataset","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13906364,0.19265765,0.08514404,0.0038644588,0.0037622668,0.00922502,0.475682,0.0047147805,0.08588626],"genre_scores_gemma":[0.17035748,0.055769034,0.18229437,0.0011732619,0.0012835048,0.017436307,0.54060465,0.00092617655,0.030155277],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99691725,0.000906153,0.00062963367,0.0003909216,0.0010746638,0.000081370184],"domain_scores_gemma":[0.9817276,0.009240003,0.0014090027,0.000988889,0.006339038,0.0002954453],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003669403,0.0006246658,0.0010282006,0.01112696,0.0010733544,0.0008874408,0.0008523941,0.00071169954,0.008332957],"category_scores_gemma":[0.017906783,0.00045572015,0.00047447011,0.0096553955,0.0004900535,0.0009019442,0.0012023271,0.00067820307,0.004020585],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008259268,0.00022948132,0.005866162,0.044351198,0.000285539,0.002777246,0.0030073926,0.0010733011,0.06913429,0.00515143,0.323634,0.5436641],"study_design_scores_gemma":[0.00019468777,0.0001577559,0.033240046,0.0026290559,0.0005412565,0.002030803,0.0006337067,0.0012044107,0.01180577,0.0013148816,0.9461525,0.0000951384],"about_ca_topic_score_codex":0.0057028076,"about_ca_topic_score_gemma":0.013358649,"teacher_disagreement_score":0.01112696,"about_ca_system_score_codex":0.0007851798,"about_ca_system_score_gemma":0.0042171776,"threshold_uncertainty_score":0.027876556},"labels":[],"label_agreement":null},{"id":"W78848553","doi":"10.21437/interspeech.2008-606","title":"Using latent Dirichlet allocation to incorporate domain knowledge for topic transition detection","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Latent Dirichlet allocation; Literal (mathematical logic); Computer science; Similarity (geometry); Topic model; Artificial intelligence; Domain (mathematical analysis); Transition (genetics); Natural language processing; Matching (statistics); Space (punctuation); Latent semantic analysis; Pattern recognition (psychology); Image (mathematics); Algorithm; Mathematics","score_opus":0.04733201078399718,"score_gpt":0.3001867433196963,"score_spread":0.2528547325356991,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W78848553","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0074364124,0.000379788,0.99058753,0.00013849749,0.000049966482,0.00007920905,0.00010534434,0.000748695,0.0004746392],"genre_scores_gemma":[0.27265024,0.00077070325,0.71959126,0.00025722763,0.00038210148,0.0006807947,0.0018523153,0.0003545187,0.0034608827],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9937743,0.0038465415,0.0002882406,0.001205402,0.0006249441,0.000260599],"domain_scores_gemma":[0.9921014,0.0061027403,0.00046825432,0.0006671438,0.0005038882,0.00015657587],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0059177526,0.001408252,0.0013781085,0.003450669,0.0009312291,0.0022824914,0.0019686578,0.0019024261,0.0019098805],"category_scores_gemma":[0.016122757,0.00080327835,0.0016117678,0.0031481618,0.0011499702,0.0036682507,0.0022045497,0.0033209028,0.001953851],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000605787,0.00048808596,0.005419977,0.0004774137,0.00037248826,0.0002349505,0.0009503959,0.1677577,0.019546924,0.01408499,0.0066183447,0.783443],"study_design_scores_gemma":[0.00003564836,0.000064304884,0.0012150157,0.0000387934,0.00005085611,0.00010473719,0.00010209789,0.9724577,0.0057195555,0.017509872,0.0026524162,0.000048956932],"about_ca_topic_score_codex":0.004117857,"about_ca_topic_score_gemma":0.004131709,"teacher_disagreement_score":0.0059177526,"about_ca_system_score_codex":0.0013281697,"about_ca_system_score_gemma":0.0014076312,"threshold_uncertainty_score":0.03129649},"labels":[],"label_agreement":null},{"id":"W795346602","doi":"10.1558/equinox.25350","title":"Semantic networks: the description of linguistic meaning in SFL","year":2005,"lang":"en","type":"article","venue":"Equinox eBooks Publishing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Meaning (existential); Linguistics; Semantics (computer science); Perspective (graphical); Computer science; Computational semantics; Formal semantics (linguistics); Artificial intelligence; Epistemology; Philosophy; Operational semantics","score_opus":0.023576897698249017,"score_gpt":0.2535127691030703,"score_spread":0.22993587140482127,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W795346602","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012054007,0.006819222,0.8462601,0.009569438,0.00047728105,0.00011898755,0.00062883925,0.00034327697,0.12372892],"genre_scores_gemma":[0.56455797,0.007847025,0.39500725,0.0016351416,0.0012449903,0.0007689018,0.0010709049,0.00037460274,0.0274932],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99769706,0.0012676608,0.0001544142,0.00031841948,0.00043115774,0.00013135982],"domain_scores_gemma":[0.99743986,0.0015671178,0.00022673613,0.0003545047,0.00028294805,0.00012877827],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0038684446,0.00089146046,0.0006244777,0.0045666727,0.0023739268,0.0068172263,0.0016875233,0.0024013584,0.006812195],"category_scores_gemma":[0.00829074,0.00046675265,0.0011677264,0.005146733,0.0092730895,0.02561911,0.0030278019,0.0029693746,0.0009649276],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000039401302,0.0000018222983,0.000047715264,0.000018823446,0.0000029236808,0.000028648972,0.00041716523,0.00061093684,0.000051024217,0.9949062,0.0004916193,0.0034192184],"study_design_scores_gemma":[0.0000016424866,0.0000017514518,0.000037480346,0.000020688449,0.0000027526285,0.00003169403,0.00015187824,0.0038887153,0.00003857956,0.9836373,0.012184014,0.0000036594015],"about_ca_topic_score_codex":0.00433425,"about_ca_topic_score_gemma":0.0034114413,"teacher_disagreement_score":0.0068172263,"about_ca_system_score_codex":0.0033232016,"about_ca_system_score_gemma":0.0012586826,"threshold_uncertainty_score":0.024111569},"labels":[],"label_agreement":null},{"id":"W800351510","doi":"10.71781/9768","title":"Génération de résumés par abstraction","year":2013,"lang":"fr","type":"dissertation","venue":"Open MIND","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Fonds de recherche du Québec – Nature et technologies; National Institute of Standards and Technology; Tsinghua University","keywords":"Abstraction; Mathematics; Combinatorics; Philosophy; Epistemology","score_opus":0.04583745383580591,"score_gpt":0.3582150979952278,"score_spread":0.31237764415942193,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W800351510","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021714143,0.00176524,0.89662284,0.00070569746,0.00069777516,0.0006149866,0.0043986626,0.059484966,0.013995568],"genre_scores_gemma":[0.13563246,0.0019196119,0.8021867,0.0005156509,0.00035079126,0.0005853428,0.014161697,0.0068374863,0.03781021],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9984085,0.00028805828,0.00016751124,0.00035463515,0.00068463734,0.00009676997],"domain_scores_gemma":[0.994994,0.0021840176,0.00027108687,0.0012162027,0.001218143,0.000116627365],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013918391,0.0018519097,0.0009979735,0.001750685,0.0007289988,0.0025141623,0.0014973454,0.00086931477,0.021906141],"category_scores_gemma":[0.007320204,0.0006226746,0.0015648268,0.001395475,0.00059273286,0.002987089,0.0021345792,0.0013504068,0.009960451],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00070410804,0.000099644276,0.0016664637,0.0020664337,0.00019712039,0.00079298584,0.0018196083,0.010996722,0.08142626,0.013043564,0.040227138,0.84695995],"study_design_scores_gemma":[0.00023636328,0.0006137314,0.0050521083,0.00054861064,0.00049407576,0.0012876248,0.001369359,0.16858776,0.24320251,0.025091683,0.5532129,0.000303346],"about_ca_topic_score_codex":0.0028116286,"about_ca_topic_score_gemma":0.003136025,"teacher_disagreement_score":0.021906141,"about_ca_system_score_codex":0.00061284547,"about_ca_system_score_gemma":0.0013049677,"threshold_uncertainty_score":0.073283315},"labels":[],"label_agreement":null},{"id":"W803028973","doi":"","title":"Generating Natural Language Questions to Support Learning On-Line","year":2013,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":119,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Natural language; Context (archaeology); Task (project management); Natural language processing; Artificial intelligence; Natural language understanding; Question answering; Engineering","score_opus":0.013573088263604672,"score_gpt":0.30141558890242065,"score_spread":0.287842500638816,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W803028973","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024855947,0.00013887211,0.92452335,0.0005402515,0.00011075863,0.00065637304,0.0014044056,0.04155922,0.006210795],"genre_scores_gemma":[0.11778588,0.00015603947,0.86937976,0.00023166303,0.000062896106,0.000375704,0.005647841,0.0011396571,0.005220518],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9970866,0.0015054931,0.0002233933,0.0005128962,0.00057146593,0.00010024027],"domain_scores_gemma":[0.98387665,0.010477241,0.00096543983,0.0021716224,0.002188853,0.0003202746],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003253278,0.0011161381,0.0005587565,0.0009590121,0.00040831688,0.0016175483,0.0026016692,0.0021074845,0.009669248],"category_scores_gemma":[0.015734127,0.0003809326,0.00073687435,0.0005965133,0.0005143047,0.002826259,0.0013747254,0.0013200092,0.006420213],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00047125854,0.0014175188,0.0071331216,0.0012158052,0.000095061085,0.0011057191,0.0022832376,0.019977322,0.11422559,0.018054811,0.042574868,0.7914456],"study_design_scores_gemma":[0.00027257012,0.00071039,0.0063994853,0.00025944487,0.0001154739,0.001519779,0.0007649049,0.53685254,0.235082,0.05475404,0.16306658,0.00020284185],"about_ca_topic_score_codex":0.0011438347,"about_ca_topic_score_gemma":0.0018025287,"teacher_disagreement_score":0.009669248,"about_ca_system_score_codex":0.00063716393,"about_ca_system_score_gemma":0.0009914148,"threshold_uncertainty_score":0.032346845},"labels":[],"label_agreement":null},{"id":"W82889130","doi":"10.1007/11424918_35","title":"Adjectives: A Uniform Semantic Approach","year":2005,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Windsor","funders":"","keywords":"Computer science; Natural language processing; Linguistics; Philosophy","score_opus":0.013424798239769206,"score_gpt":0.2516046370746645,"score_spread":0.2381798388348953,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W82889130","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0067894985,0.0037567704,0.82277495,0.0023487085,0.0011602403,0.00029555344,0.0010946185,0.0021648007,0.15961485],"genre_scores_gemma":[0.23434237,0.0067013255,0.6452173,0.0022783093,0.0028550576,0.00086961105,0.0033916065,0.0023669475,0.10197753],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9981659,0.0005185762,0.0002657577,0.0004974895,0.0004244533,0.00012789643],"domain_scores_gemma":[0.99915504,0.00022252939,0.00004369282,0.0002584315,0.0002534228,0.00006686522],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013895313,0.0015423826,0.00144367,0.004454165,0.0026451326,0.0061617927,0.0023650099,0.0021094698,0.013496905],"category_scores_gemma":[0.0021476878,0.0012096725,0.0022992406,0.004110989,0.0046196147,0.017257094,0.0050012567,0.00400142,0.008447756],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000021414151,0.000022682438,0.00007125393,0.00011888647,0.00001213263,0.000095862895,0.00042754033,0.00015096921,0.0011867901,0.9576807,0.0056541804,0.03455753],"study_design_scores_gemma":[0.000021233727,0.000025868629,0.00022153163,0.000117026335,0.00006322472,0.00056308036,0.00044809788,0.0036948356,0.0011115884,0.8344097,0.1592886,0.00003524208],"about_ca_topic_score_codex":0.0010047125,"about_ca_topic_score_gemma":0.0012453301,"teacher_disagreement_score":0.013496905,"about_ca_system_score_codex":0.0008893997,"about_ca_system_score_gemma":0.0009803659,"threshold_uncertainty_score":0.04515165},"labels":[],"label_agreement":null},{"id":"W83007297","doi":"","title":"Handling Pronouns Intelligently","year":2005,"lang":"en","type":"article","venue":"New Trends in Software Methodologies, Tools and Techniques","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Automatic summarization; Computer science; Anaphora (linguistics); Focus (optics); Identification (biology); Question answering; Interpretation (philosophy); Natural language processing; Resolution (logic); Artificial intelligence; Information retrieval; Programming language","score_opus":0.12023128723289653,"score_gpt":0.3808686564178361,"score_spread":0.2606373691849396,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W83007297","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.030908586,0.0012245066,0.93105656,0.0029128357,0.0004562277,0.0001358432,0.00034773603,0.0024099643,0.030547692],"genre_scores_gemma":[0.33692074,0.0020284592,0.60932446,0.0017818067,0.0010740587,0.00018343603,0.0017659765,0.0015994204,0.045321636],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9920415,0.0020702367,0.0007071901,0.0014878116,0.003122313,0.00057087245],"domain_scores_gemma":[0.99340665,0.0024459725,0.00070454384,0.0015307913,0.0017946432,0.00011750822],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0042523313,0.001252149,0.0013722988,0.0028800538,0.0031752337,0.005262971,0.0024195472,0.002058684,0.0069081034],"category_scores_gemma":[0.011902526,0.0010929597,0.00089521933,0.0028660614,0.002193571,0.011842329,0.0056843897,0.0029989078,0.004514494],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020526211,0.00011782672,0.0026571697,0.0006653525,0.00010416056,0.0014002514,0.0072097606,0.006435109,0.030050414,0.5652947,0.027038068,0.35882193],"study_design_scores_gemma":[0.000042320007,0.00006312497,0.00054109347,0.00019593353,0.00013216725,0.0010707013,0.00251758,0.045772806,0.044532314,0.6254976,0.2795443,0.00009008563],"about_ca_topic_score_codex":0.0010941369,"about_ca_topic_score_gemma":0.0013421164,"teacher_disagreement_score":0.0069081034,"about_ca_system_score_codex":0.0013833954,"about_ca_system_score_gemma":0.0015775934,"threshold_uncertainty_score":0.023109853},"labels":[],"label_agreement":null},{"id":"W830770281","doi":"10.4000/trajectoires.2157","title":"Construction de corpus généraux et spécialisés à partir du Web (Ad hoc and general-purpose corpus construction from web sources)","year":2016,"lang":"fr","type":"article","venue":"Tr@jectoires","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Barrie Urology Group","funders":"","keywords":"Humanities; Corpus linguistics; Art; Philosophy; Linguistics","score_opus":0.016160062854339425,"score_gpt":0.25501382275061696,"score_spread":0.23885375989627752,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W830770281","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10169073,0.0015920314,0.81116956,0.0009812455,0.00077220576,0.003584694,0.027924523,0.013551923,0.038733065],"genre_scores_gemma":[0.11675371,0.0010909408,0.7941478,0.00021116566,0.00022469593,0.0048557357,0.05982592,0.0034838466,0.019406255],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99633116,0.0011989095,0.0005122457,0.0009800204,0.0008077854,0.00016995681],"domain_scores_gemma":[0.98875767,0.006032248,0.00050998735,0.002095084,0.002357906,0.00024715438],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003132667,0.0010312933,0.0009016121,0.011130585,0.0027173131,0.0031878399,0.0013685197,0.001264378,0.013089548],"category_scores_gemma":[0.014456778,0.0009954026,0.0011173753,0.008076438,0.0016923042,0.0032786718,0.0037222994,0.0017958593,0.0061400356],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006465568,0.00031516957,0.007493494,0.0041747633,0.00020188538,0.0032120175,0.01240198,0.0058788764,0.1166637,0.07380717,0.06608447,0.70911986],"study_design_scores_gemma":[0.00017627752,0.0003149348,0.01751217,0.00085473625,0.00036893212,0.0034122418,0.006795644,0.06464841,0.1437029,0.040850442,0.7211719,0.00019150188],"about_ca_topic_score_codex":0.0032777537,"about_ca_topic_score_gemma":0.005175614,"teacher_disagreement_score":0.013089548,"about_ca_system_score_codex":0.001169957,"about_ca_system_score_gemma":0.003862606,"threshold_uncertainty_score":0.04378885},"labels":[],"label_agreement":null},{"id":"W831928984","doi":"","title":"Exploiting a Multilingual Web-based Encyclopedia for Bilingual Terminology Extraction","year":2010,"lang":"en","type":"article","venue":"Institutional Repositories DataBase (IRDB)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Computer science; Terminology; Encyclopedia; Natural language processing; Rank (graph theory); Information retrieval; Exploit; Artificial intelligence; Ontology; Filter (signal processing); Term (time); Information extraction; World Wide Web; Linguistics","score_opus":0.01890596775989091,"score_gpt":0.31368904551802557,"score_spread":0.29478307775813467,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W831928984","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11460355,0.004003382,0.8140163,0.00092075893,0.00036937377,0.0015170502,0.0226511,0.011252894,0.030665709],"genre_scores_gemma":[0.13781445,0.001870542,0.8152937,0.00016063217,0.00014116084,0.00081824063,0.038478263,0.00095943833,0.004463533],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99871314,0.0003125033,0.0003614204,0.00031325794,0.00022575281,0.000073930234],"domain_scores_gemma":[0.9971584,0.0011517614,0.0002819146,0.00036503864,0.000922002,0.00012089596],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001331081,0.0012519128,0.0009869721,0.016681097,0.0012662262,0.002553696,0.00070457114,0.00048089572,0.006010252],"category_scores_gemma":[0.0065559554,0.0005139925,0.0008500471,0.011897693,0.0004279181,0.004175402,0.0021805698,0.000708042,0.004788286],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003533464,0.0002515715,0.00892714,0.003264825,0.00030110686,0.0024974504,0.002016248,0.0029733207,0.10909439,0.01954703,0.01819306,0.83258057],"study_design_scores_gemma":[0.00027469796,0.0008232248,0.05197795,0.0018251475,0.0014989563,0.009060941,0.0074712285,0.10799182,0.21327713,0.02887394,0.5763076,0.0006173629],"about_ca_topic_score_codex":0.0038481038,"about_ca_topic_score_gemma":0.0074352706,"teacher_disagreement_score":0.016681097,"about_ca_system_score_codex":0.00066106365,"about_ca_system_score_gemma":0.002988437,"threshold_uncertainty_score":0.020106316},"labels":[],"label_agreement":null},{"id":"W8504697","doi":"10.63317/2c6sh4qi5md8","title":"Romanian Zero Pronoun Distribution: A Comparative Study","year":2010,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Anaphora (linguistics); Pronoun; Romanian; Computer science; Zero (linguistics); Subject pronoun; Natural language processing; Antecedent (behavioral psychology); Linguistics; Artificial intelligence; Identification (biology); Resolution (logic); Personal pronoun; Scope (computer science); Psychology; Philosophy","score_opus":0.01640505086969652,"score_gpt":0.30125803413104263,"score_spread":0.2848529832613461,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W8504697","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98364985,0.0014295055,0.0017590027,0.00013111308,0.00002784308,0.000010552149,0.0003051696,0.00003516461,0.012651867],"genre_scores_gemma":[0.997398,0.00026242697,0.00045947684,0.000023334247,0.000019395127,0.000006581765,0.00027865998,0.000027577707,0.0015245232],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99854314,0.00072790385,0.00011840091,0.00023832735,0.00021832451,0.00015385356],"domain_scores_gemma":[0.9980981,0.0010573256,0.00023591958,0.0002495743,0.00026852798,0.00009055031],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011850349,0.00018028138,0.00038906344,0.0022032808,0.0007674236,0.0008997357,0.00031374535,0.000191231,0.011905],"category_scores_gemma":[0.0060733156,0.00011344394,0.0002186432,0.0018851937,0.0006156951,0.0006717242,0.0005179252,0.0003053131,0.0016920521],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0026848828,0.00031518086,0.5171343,0.00045867864,0.00035734774,0.0087857675,0.024444494,0.0007573981,0.014805942,0.021745028,0.006310288,0.4022007],"study_design_scores_gemma":[0.00012327453,0.0007548823,0.87372905,0.00015385557,0.00016795965,0.0418914,0.016435718,0.0045179846,0.00474474,0.0076927603,0.0497241,0.00006435012],"about_ca_topic_score_codex":0.0013075661,"about_ca_topic_score_gemma":0.0010205309,"teacher_disagreement_score":0.011905,"about_ca_system_score_codex":0.00043432054,"about_ca_system_score_gemma":0.00023831234,"threshold_uncertainty_score":0.039826155},"labels":[],"label_agreement":null},{"id":"W850942346","doi":"","title":"Finding Negative Key Phrases for Internet Advertising Campaigns using Wikipedia","year":2011,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Key (lock); Computer science; Phrase; The Internet; Context (archaeology); Natural language processing; Artificial intelligence; Online advertising; World Wide Web; Information retrieval; Advertising; Computer security","score_opus":0.06449796609774279,"score_gpt":0.3067233470942443,"score_spread":0.24222538099650154,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W850942346","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.85088855,0.009016898,0.10290285,0.00097006955,0.0007549612,0.0006415145,0.014677,0.0027776517,0.017370492],"genre_scores_gemma":[0.91531444,0.0011505773,0.06783408,0.00017136073,0.00031806366,0.00029022485,0.013071532,0.0002973928,0.0015522258],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99790394,0.00041531987,0.00040787942,0.0005752757,0.00053752866,0.00015993029],"domain_scores_gemma":[0.9899485,0.0051146327,0.0019067918,0.0004907107,0.0021442277,0.00039501808],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001250197,0.0011530366,0.00055302924,0.008961699,0.0014146415,0.0020736973,0.0005886736,0.0010452524,0.0015704168],"category_scores_gemma":[0.010575916,0.00052161864,0.0007027039,0.0043869806,0.0009654247,0.0032915932,0.0010998459,0.00087833154,0.0010832966],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021614712,0.0008301707,0.26670143,0.005755878,0.0007002381,0.0075997594,0.008209963,0.0061831903,0.16064313,0.016809054,0.043704398,0.48070136],"study_design_scores_gemma":[0.00026597516,0.001049537,0.5124281,0.0013632056,0.0017162634,0.021069385,0.010419849,0.16402891,0.09233283,0.048215345,0.1463006,0.0008100933],"about_ca_topic_score_codex":0.0036038493,"about_ca_topic_score_gemma":0.006557173,"teacher_disagreement_score":0.008961699,"about_ca_system_score_codex":0.00062464946,"about_ca_system_score_gemma":0.0012167136,"threshold_uncertainty_score":0.00716573},"labels":[],"label_agreement":null},{"id":"W85461845","doi":"10.1007/978-3-642-54903-8_33","title":"How Document Properties Affect Document Relatedness Measures","year":2014,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Trigram; Computer science; Document classification; Quality (philosophy); Information retrieval; Vector space model; Natural language processing; Word (group theory); Property (philosophy); Affect (linguistics); Artificial intelligence; Space (punctuation); Weighting; Linguistics","score_opus":0.016428412065574563,"score_gpt":0.24364657990157457,"score_spread":0.22721816783600002,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W85461845","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.78535104,0.0114463,0.16967924,0.0011174863,0.00058000913,0.00024954948,0.004707719,0.0041127494,0.022755899],"genre_scores_gemma":[0.9405009,0.0013877945,0.048714496,0.00013689896,0.00022769117,0.00008075106,0.004424816,0.0008825583,0.003644101],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9957159,0.0017710705,0.000352887,0.00082466495,0.001135909,0.0001994732],"domain_scores_gemma":[0.9482888,0.041211743,0.002182024,0.0029059432,0.0047510657,0.0006604442],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005212633,0.00056042837,0.0006433406,0.0036387327,0.000861801,0.004201354,0.00056564086,0.0009375276,0.0035628746],"category_scores_gemma":[0.055002984,0.00034774505,0.0006274694,0.004182603,0.0005477081,0.0065123443,0.0008275325,0.0011488184,0.00237315],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0028765178,0.000595161,0.1124243,0.0020205562,0.00081264845,0.00037826155,0.0020831593,0.018893896,0.13666575,0.015723718,0.025959346,0.68156666],"study_design_scores_gemma":[0.00025800092,0.0020590727,0.28371495,0.00063116144,0.003069496,0.0031253507,0.0029038799,0.33413672,0.22450313,0.08665848,0.058465097,0.0004747233],"about_ca_topic_score_codex":0.0013361363,"about_ca_topic_score_gemma":0.002096633,"teacher_disagreement_score":0.005212633,"about_ca_system_score_codex":0.0007319654,"about_ca_system_score_gemma":0.00043666683,"threshold_uncertainty_score":0.027567387},"labels":[],"label_agreement":null},{"id":"W85714267","doi":"","title":"A multiple dominance analysis of sharing coordination constructions using tree adjoining grammar","year":2010,"lang":"en","type":"dissertation","venue":"Summit (Simon Fraser University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada","keywords":"Ellipsis (linguistics); Dominance (genetics); Computer science; Linguistics; Grammar; Syntax; Natural language processing; Artificial intelligence; Philosophy","score_opus":0.012751468375682862,"score_gpt":0.24432162986535022,"score_spread":0.23157016148966736,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W85714267","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14679302,0.00023457194,0.814234,0.000998914,0.000041040057,0.000041866944,0.00008395759,0.0003921184,0.03718043],"genre_scores_gemma":[0.89912856,0.00021052592,0.09390128,0.0001539527,0.000060210426,0.000041799314,0.0001188841,0.00028483884,0.0060999086],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990332,0.00035463815,0.000045807425,0.00019479178,0.00026277523,0.00010875409],"domain_scores_gemma":[0.99875975,0.0007244176,0.00008215862,0.00017792525,0.0002037683,0.000052013886],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009809658,0.00033818086,0.00040450503,0.000980728,0.0013853665,0.0021259745,0.00069313927,0.0005843238,0.003413913],"category_scores_gemma":[0.0021943161,0.00038846288,0.00073361833,0.0009975947,0.0032275761,0.0057752477,0.0018921944,0.0015605579,0.00049592555],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000009851296,0.00000894021,0.0004234708,0.000017558938,0.000004331201,0.00017331043,0.0011186883,0.001969651,0.0026243234,0.9844599,0.00046313336,0.008726906],"study_design_scores_gemma":[0.00000936944,0.0000133891945,0.0004536825,0.000014139246,0.0000146006405,0.00021218043,0.00041419684,0.036621038,0.0023269996,0.9541058,0.0057981736,0.00001626182],"about_ca_topic_score_codex":0.002615925,"about_ca_topic_score_gemma":0.0021992214,"teacher_disagreement_score":0.003413913,"about_ca_system_score_codex":0.0011756474,"about_ca_system_score_gemma":0.0009184977,"threshold_uncertainty_score":0.011420667},"labels":[],"label_agreement":null},{"id":"W86816073","doi":"","title":"A distributional account of the semantics of multiword expressions","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Lexicon; Natural language processing; Artificial intelligence; Meaning (existential); Semantics (computer science); Homogeneous; Distributional semantics; Relation (database); Linguistics; Mathematics; Semantic similarity; Psychology; Philosophy; Programming language; Database","score_opus":0.016461616708681858,"score_gpt":0.25791613188452417,"score_spread":0.24145451517584232,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W86816073","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.19300392,0.0015592139,0.76322156,0.0025862094,0.00017822614,0.00006339153,0.0007199075,0.00071853574,0.037948962],"genre_scores_gemma":[0.9367146,0.0005330874,0.057935465,0.00026193602,0.00026886462,0.00008017539,0.00054737396,0.00022423599,0.0034342052],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9987729,0.00039795536,0.00016961074,0.00028201556,0.00028288024,0.00009449087],"domain_scores_gemma":[0.9970011,0.0013344624,0.0003785082,0.00050993485,0.0006513463,0.00012464258],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011377247,0.0004587515,0.00056842633,0.002887919,0.0013202142,0.0033882244,0.0013989973,0.0009855533,0.0032746675],"category_scores_gemma":[0.00589394,0.00039721213,0.0008078845,0.0033731495,0.0031684902,0.01052407,0.0016652113,0.0011068428,0.00065061246],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000060321163,0.00002661983,0.0025385118,0.0001470846,0.000028012486,0.00019349567,0.001485074,0.0027129813,0.005968012,0.94457,0.0009932132,0.041276813],"study_design_scores_gemma":[0.000010049018,0.000025621574,0.0022070725,0.000024516989,0.00001764401,0.0004595334,0.0003987151,0.021049961,0.0016039116,0.96786994,0.006310157,0.000022967839],"about_ca_topic_score_codex":0.00085024606,"about_ca_topic_score_gemma":0.0009363577,"teacher_disagreement_score":0.0033882244,"about_ca_system_score_codex":0.0012005388,"about_ca_system_score_gemma":0.0004764207,"threshold_uncertainty_score":0.0109549165},"labels":[],"label_agreement":null},{"id":"W878889913","doi":"","title":"Multi-Metric Optimization Using Ensemble Tuning","year":2013,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Metric (unit); Computer science; Pareto principle; Multi-objective optimization; Machine translation; Performance metric; Mathematical optimization; Artificial intelligence; Machine learning; Algorithm; Mathematics; Engineering","score_opus":0.026187872872148972,"score_gpt":0.2813989874716865,"score_spread":0.25521111459953755,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W878889913","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.028938357,0.0009322553,0.96534735,0.00015731253,0.00008393435,0.00009788226,0.000096771604,0.0015446586,0.002801503],"genre_scores_gemma":[0.6038643,0.0003766579,0.3911193,0.00024954398,0.00011893609,0.0003684586,0.00062484335,0.0006373441,0.0026407365],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9955617,0.002218229,0.00025910372,0.00072341115,0.0010307463,0.00020683889],"domain_scores_gemma":[0.9929213,0.004281123,0.0004094546,0.0011038781,0.001123964,0.0001603055],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006560385,0.0020822443,0.0019578226,0.002171996,0.0006609128,0.0015056769,0.001473616,0.0015536804,0.0016276374],"category_scores_gemma":[0.016543204,0.00051815604,0.0011672693,0.001961249,0.00078976847,0.0024500112,0.0018453836,0.0017243697,0.0007658303],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012420678,0.0001255869,0.0021050188,0.00010576284,0.00024094281,0.00004765777,0.000085396845,0.7201294,0.0051723514,0.005265125,0.0022253115,0.26437327],"study_design_scores_gemma":[0.000008208838,0.00006074223,0.0002483316,0.000007890464,0.000014598069,0.000026911153,0.000011765596,0.9935488,0.0014429546,0.0040239855,0.0005943416,0.000011447867],"about_ca_topic_score_codex":0.002436064,"about_ca_topic_score_gemma":0.0029786201,"teacher_disagreement_score":0.006560385,"about_ca_system_score_codex":0.0012110192,"about_ca_system_score_gemma":0.0010977921,"threshold_uncertainty_score":0.03469503},"labels":[],"label_agreement":null},{"id":"W91653113","doi":"10.30827/digibug.31745","title":"Post-editing machine translation as an FSL exercise","year":2008,"lang":"en","type":"article","venue":"Porta Linguarum Revista Interuniversitaria de Didáctica de las Lenguas Extranjeras","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"","keywords":"Humanities; Philosophy; Art","score_opus":0.013230425694117436,"score_gpt":0.2583183515785562,"score_spread":0.24508792588443876,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W91653113","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8529434,0.00030636622,0.089585215,0.0022473016,0.00035418544,0.00075331907,0.0005183397,0.004271209,0.049020626],"genre_scores_gemma":[0.91348255,0.00017032409,0.055897124,0.00066715933,0.00012950036,0.0003753154,0.0006709315,0.0007873167,0.027819907],"study_design_codex":"design_other","study_design_gemma":"qualitative","domain_scores_codex":[0.9934121,0.003906472,0.00037664495,0.0009900973,0.000908638,0.00040615493],"domain_scores_gemma":[0.91463155,0.06600293,0.002507197,0.009033706,0.0061756824,0.0016489363],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009339526,0.0010462018,0.0010411331,0.0008310815,0.002042168,0.0028740335,0.0015781132,0.0014338831,0.017010089],"category_scores_gemma":[0.065621145,0.00035118894,0.000461347,0.0010386439,0.0010669525,0.0031026055,0.0024075648,0.0021925722,0.0068277856],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016369967,0.005395288,0.017116627,0.00066719623,0.00004188887,0.0011107579,0.072018616,0.0032017739,0.03732037,0.0080802,0.020549895,0.83286047],"study_design_scores_gemma":[0.0010722033,0.023269884,0.09176597,0.0010466211,0.0002946318,0.007335593,0.078054,0.06842013,0.20858344,0.049203906,0.47022474,0.0007288631],"about_ca_topic_score_codex":0.0005129559,"about_ca_topic_score_gemma":0.0011915665,"teacher_disagreement_score":0.017010089,"about_ca_system_score_codex":0.0008757084,"about_ca_system_score_gemma":0.0009835806,"threshold_uncertainty_score":0.056904435},"labels":[],"label_agreement":null},{"id":"W94318738","doi":"","title":"Synonym-Based Expansion and Boosting-Based Re-Ranking: A Two-phase Approach for Genomic Information Retrieval.","year":2005,"lang":"en","type":"article","venue":"Text REtrieval Conference","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Boosting (machine learning); WordNet; Information retrieval; Artificial intelligence; Ranking (information retrieval); Machine learning","score_opus":0.028944597508011552,"score_gpt":0.29892116513885053,"score_spread":0.269976567630839,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W94318738","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020568717,0.0019851525,0.9645828,0.00033935695,0.00025238236,0.0007570464,0.00064776954,0.0069979136,0.0038689112],"genre_scores_gemma":[0.110476315,0.00065534306,0.88060045,0.0003231873,0.00022671843,0.00042634655,0.0025158406,0.00035018308,0.0044255173],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9958961,0.001704541,0.00025276517,0.00060632377,0.0013168644,0.0002234168],"domain_scores_gemma":[0.99586296,0.0012840399,0.00031416182,0.00086569303,0.0015126924,0.00016051173],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005392751,0.00117083,0.0018086787,0.0070267385,0.0010401411,0.0014569007,0.0018941008,0.0011594048,0.0029867794],"category_scores_gemma":[0.009490428,0.0005584182,0.0011778836,0.0051608398,0.0005774494,0.0026875571,0.0017724633,0.0011827106,0.004304155],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037447255,0.0004446272,0.0027035302,0.00044761598,0.00021301955,0.0002111561,0.0004088,0.009178377,0.05300895,0.005177511,0.020609353,0.90722257],"study_design_scores_gemma":[0.00029836592,0.0012384828,0.016729902,0.00015802663,0.00057219015,0.0022932221,0.0006190575,0.69964385,0.117107995,0.04860067,0.11232475,0.00041357768],"about_ca_topic_score_codex":0.0019374285,"about_ca_topic_score_gemma":0.0046644583,"teacher_disagreement_score":0.0070267385,"about_ca_system_score_codex":0.0006187105,"about_ca_system_score_gemma":0.0013053512,"threshold_uncertainty_score":0.028519928},"labels":[],"label_agreement":null},{"id":"W95321849","doi":"10.1007/978-3-642-30353-1_43","title":"Populating a Knowledge Base from a Dictionary","year":2012,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; USable; Knowledge base; Encyclopedia; Artificial intelligence; Natural language processing; Task (project management); Question answering; Knowledge-based systems; Natural language; Information retrieval; World Wide Web","score_opus":0.020120521833711682,"score_gpt":0.2706545788491563,"score_spread":0.2505340570154446,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W95321849","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04628026,0.0013326169,0.90672207,0.0011615196,0.000402592,0.000741881,0.008384967,0.01410822,0.020865813],"genre_scores_gemma":[0.10819171,0.0018092103,0.85573894,0.00041584502,0.000113009446,0.0003213323,0.019977998,0.001454339,0.011977582],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9993585,0.00010582652,0.000074813506,0.0001685151,0.00023555454,0.000056884077],"domain_scores_gemma":[0.9977168,0.0013563582,0.000058657035,0.00040775235,0.0003653387,0.0000950928],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00074067817,0.0008787672,0.0014246446,0.004444095,0.0011329887,0.0033604265,0.0023341107,0.0012841179,0.014038579],"category_scores_gemma":[0.0066835,0.0008900067,0.0018644246,0.0043800985,0.0006988705,0.0057626255,0.0053924876,0.0015731697,0.007746225],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004892472,0.0002425414,0.0017103322,0.0017063755,0.00022943722,0.0025238711,0.0012921492,0.012491203,0.028944306,0.037830368,0.038866162,0.873674],"study_design_scores_gemma":[0.00026467713,0.00046346642,0.0026970305,0.0014637709,0.0013225456,0.0041789063,0.0033502276,0.31005242,0.08099115,0.18810847,0.4068397,0.00026763623],"about_ca_topic_score_codex":0.003320663,"about_ca_topic_score_gemma":0.0050553144,"teacher_disagreement_score":0.014038579,"about_ca_system_score_codex":0.00060473674,"about_ca_system_score_gemma":0.0016515495,"threshold_uncertainty_score":0.04696375},"labels":[],"label_agreement":null},{"id":"W95866547","doi":"","title":"A tool for detecting French-English cognates and false friends","year":2007,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Political science; Humanities; Philosophy","score_opus":0.009646752142278874,"score_gpt":0.26678175244387237,"score_spread":0.2571350003015935,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W95866547","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16148186,0.0020759432,0.68341726,0.000985631,0.0005262263,0.0012036358,0.020404732,0.10622387,0.023680784],"genre_scores_gemma":[0.32537603,0.0005543032,0.6270237,0.0003462352,0.00013135954,0.0009045321,0.018267604,0.0017582853,0.025637994],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9977156,0.00039947664,0.0002761309,0.00078259135,0.0006381826,0.00018799197],"domain_scores_gemma":[0.99085593,0.00548012,0.0006829241,0.000707165,0.0019883162,0.0002855901],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018962368,0.0015760005,0.00094244134,0.009220742,0.0015775431,0.0036378736,0.0012828532,0.0017516179,0.016560499],"category_scores_gemma":[0.013553949,0.00065121765,0.00091811526,0.0037154884,0.0005861421,0.0033062415,0.0016357524,0.0008258255,0.006706474],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008486766,0.00020638976,0.039010134,0.0009538921,0.00020407572,0.0013899069,0.003091747,0.0015668249,0.027582496,0.007201907,0.045493457,0.8724505],"study_design_scores_gemma":[0.00021577987,0.00066997064,0.12517373,0.0007196014,0.0006476095,0.008479456,0.009560328,0.2441412,0.15348397,0.017682033,0.4386409,0.00058534375],"about_ca_topic_score_codex":0.018416196,"about_ca_topic_score_gemma":0.02578693,"teacher_disagreement_score":0.018416196,"about_ca_system_score_codex":0.0011077572,"about_ca_system_score_gemma":0.001802237,"threshold_uncertainty_score":0.05540037},"labels":[],"label_agreement":null},{"id":"W96216588","doi":"10.7202/1030018ar","title":"Thésaurus et systèmes de traitement automatique de la langue","year":2015,"lang":"fr","type":"article","venue":"Documentation et bibliothèques","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":true,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Humanities; Philosophy","score_opus":0.026939888622284615,"score_gpt":0.36333728754509365,"score_spread":0.336397398922809,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W96216588","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.033182785,0.009041648,0.9026779,0.0019779261,0.00056250306,0.00065515534,0.0040117977,0.022163419,0.025726816],"genre_scores_gemma":[0.1245681,0.005373337,0.83663595,0.00039105487,0.0002586696,0.00046031608,0.0092470925,0.0028097758,0.020255752],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9957457,0.0012195436,0.0007866654,0.0008297329,0.001296511,0.000121948484],"domain_scores_gemma":[0.9911999,0.0041962913,0.0007808325,0.0018068743,0.001865205,0.0001509466],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004184387,0.0010392051,0.0014978225,0.01471221,0.0025893718,0.00618118,0.0017207846,0.0017009354,0.0093618715],"category_scores_gemma":[0.026276333,0.0008864319,0.0018609228,0.011218231,0.001704453,0.00851191,0.0027715296,0.0017999354,0.0075808885],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003138152,0.000115939896,0.0049054557,0.0039301473,0.00037753294,0.0008021649,0.0058516315,0.0055842944,0.03862354,0.07419202,0.02501968,0.8402838],"study_design_scores_gemma":[0.0001197721,0.00027408838,0.011436099,0.0019020009,0.0007367235,0.00414778,0.0034118448,0.073513106,0.09164887,0.07141742,0.74091554,0.00047672796],"about_ca_topic_score_codex":0.0151850125,"about_ca_topic_score_gemma":0.010279706,"teacher_disagreement_score":0.0151850125,"about_ca_system_score_codex":0.0024240424,"about_ca_system_score_gemma":0.0032779223,"threshold_uncertainty_score":0.031318545},"labels":[],"label_agreement":null},{"id":"W963255036","doi":"","title":"A Tale of Two Cultures: Bringing Literary Analysis and Computational Linguistics Together","year":2013,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":30,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Context (archaeology); Computational linguistics; Field (mathematics); Linguistics; Computer science; Sociology; Work (physics); Applied linguistics; Literary criticism; Literature; Epistemology; History; Engineering; Art; Philosophy; Archaeology; Mechanical engineering; Mathematics","score_opus":0.006794630645730028,"score_gpt":0.2813034283208954,"score_spread":0.27450879767516534,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W963255036","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08673522,0.061347075,0.18400207,0.3043657,0.0065536513,0.00019459586,0.00016553618,0.0007942778,0.35584188],"genre_scores_gemma":[0.8511312,0.016510552,0.08101234,0.02147915,0.0034711445,0.00041570293,0.00011796155,0.0007069437,0.025154943],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.98125684,0.015351771,0.0003857821,0.0009169728,0.001490337,0.00059824914],"domain_scores_gemma":[0.9736143,0.018864522,0.0008310487,0.0032323531,0.0016958793,0.001761931],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.017136674,0.001117567,0.001193509,0.007873033,0.02054176,0.038556326,0.0026812612,0.005476417,0.0074873352],"category_scores_gemma":[0.023894746,0.001122245,0.00078306923,0.0047745155,0.066552065,0.050638575,0.02526377,0.010120833,0.0012946571],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000059190737,0.000043405256,0.0014643673,0.00033405775,0.000060401755,0.0005439833,0.24109429,0.00024221226,0.0005891605,0.68860245,0.01364032,0.05332611],"study_design_scores_gemma":[0.000015264057,0.000042920503,0.0008326651,0.001050765,0.00003475835,0.00049170066,0.15954424,0.0007053587,0.0005866551,0.656753,0.17987883,0.000063904416],"about_ca_topic_score_codex":0.0038089536,"about_ca_topic_score_gemma":0.005746426,"teacher_disagreement_score":0.038556326,"about_ca_system_score_codex":0.0044204867,"about_ca_system_score_gemma":0.0045547234,"threshold_uncertainty_score":0.090628505},"labels":[],"label_agreement":null},{"id":"W98931015","doi":"10.1038/npre.2008.2310.1","title":"Performance of the Charniak-Lease parser on biological text using different training corpora","year":2008,"lang":"en","type":"preprint","venue":"Nature Precedings","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Nature Conservancy of Canada; Carleton University; University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada; Universities Space Research Association; Carleton University","keywords":"Computer science; Natural language processing; Parsing; Artificial intelligence; Text corpus","score_opus":0.05159216876529966,"score_gpt":0.28611959377811313,"score_spread":0.23452742501281348,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W98931015","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.89516973,0.0010638094,0.04212957,0.0011094683,0.00037234815,0.000383857,0.009919724,0.03474252,0.015109057],"genre_scores_gemma":[0.87042314,0.00033213475,0.08761029,0.0003575254,0.00005746493,0.00035160605,0.031902947,0.0028610285,0.006103789],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9964006,0.0014995799,0.00031475018,0.0010199745,0.00055512274,0.00020992132],"domain_scores_gemma":[0.9771652,0.017767822,0.00046238766,0.0019112842,0.0024030006,0.00029031493],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0074723433,0.0010518874,0.0009540997,0.0017547674,0.0011176155,0.0022739086,0.0013232833,0.0017964595,0.0050020567],"category_scores_gemma":[0.020120552,0.00080963667,0.00074831775,0.0020083953,0.0010452445,0.0033977297,0.0015689087,0.0015738254,0.0029711837],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.006693775,0.0019964387,0.031549394,0.0027322127,0.0009481391,0.0021598674,0.0042529637,0.12084252,0.23399934,0.0073351283,0.08209561,0.50539464],"study_design_scores_gemma":[0.00077283575,0.0012784704,0.043118175,0.0002827056,0.00046641906,0.00085166603,0.0018930127,0.591645,0.32707447,0.00593813,0.026313035,0.00036618122],"about_ca_topic_score_codex":0.009861857,"about_ca_topic_score_gemma":0.012132936,"teacher_disagreement_score":0.009861857,"about_ca_system_score_codex":0.0014647928,"about_ca_system_score_gemma":0.0013572489,"threshold_uncertainty_score":0.03951794},"labels":[],"label_agreement":null},{"id":"W99633038","doi":"10.3233/978-1-60750-643-0-3","title":"Towards Language-Competent Web Search","year":2010,"lang":"en","type":"book-chapter","venue":"Frontiers in artificial intelligence and applications","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"World Wide Web; Computer science","score_opus":0.02770905852442436,"score_gpt":0.29609781863537293,"score_spread":0.2683887601109486,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W99633038","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011433547,0.0024157579,0.9297506,0.0016527297,0.00023072612,0.00016343559,0.00026255683,0.00462147,0.0494692],"genre_scores_gemma":[0.22276795,0.002527065,0.7173921,0.0013255922,0.00043550154,0.00028098832,0.0015030405,0.0014667135,0.052301046],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9974643,0.0008504358,0.0001830615,0.00033137406,0.00094879576,0.00022195757],"domain_scores_gemma":[0.99673307,0.0014086878,0.00010143117,0.0007706861,0.0008276452,0.00015846566],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027459331,0.00074390776,0.0012291546,0.0014161956,0.0008387915,0.0042288536,0.0023212116,0.00189061,0.008807337],"category_scores_gemma":[0.010023626,0.00089621526,0.0007951735,0.0013600185,0.0023709724,0.0079826275,0.0045512007,0.0026256456,0.007831313],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023425966,0.00031010897,0.00039117946,0.00083750515,0.00006143014,0.00027627693,0.0013854477,0.02122782,0.025989944,0.52767956,0.04328765,0.37831885],"study_design_scores_gemma":[0.0000667928,0.00009728756,0.00032194622,0.00016433638,0.0000468297,0.00032905524,0.00046317253,0.3170649,0.021762393,0.59091467,0.06870728,0.000061388106],"about_ca_topic_score_codex":0.0025686743,"about_ca_topic_score_gemma":0.0031296217,"teacher_disagreement_score":0.008807337,"about_ca_system_score_codex":0.0009780918,"about_ca_system_score_gemma":0.0024928201,"threshold_uncertainty_score":0.02946347},"labels":[],"label_agreement":null}]}